From cec930c69c35ac2c86953640b73edbfe4cb4d614 Mon Sep 17 00:00:00 2001 From: kim Date: Sat, 11 Jul 2026 02:04:20 +0900 Subject: [PATCH] =?UTF-8?q?fix:=20=EC=9D=8C=EC=84=B1=ED=86=B5=ED=99=94=20S?= =?UTF-8?q?TT=20=EC=9E=A1=EC=9D=8C/=EA=B2=B9=EC=B9=A8=20=EB=AA=A9=EC=86=8C?= =?UTF-8?q?=EB=A6=AC=20=ED=99=98=EA=B0=81=20=EC=99=84=ED=99=94?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit no_speech_prob가 높은 세그먼트를 텍스트 조합에서 제외하고 hallucination_silence_threshold를 걸어, 배경 소음이나 다른 사람 목소리가 섞였을 때 엉뚱한 텍스트가 연속으로 나오는 문제를 완화. Co-Authored-By: Claude Sonnet 5 --- src/tools/voice_engine.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/src/tools/voice_engine.py b/src/tools/voice_engine.py index c34aec0..1199b39 100644 --- a/src/tools/voice_engine.py +++ b/src/tools/voice_engine.py @@ -53,6 +53,11 @@ log = logging.getLogger('voice_engine') logging.getLogger('faster_whisper').setLevel(logging.WARNING) MIN_PARTIAL_BYTES = int(16000 * 2 * 0.6) # ~0.6s of 16kHz mono PCM16 before we bother decoding +# Segments where Whisper itself flags no_speech_prob above this are dropped before +# joining the transcript — catches noise/background-voice hallucination (a garbled +# guess is still emitted with low no_speech_prob when a second voice overlaps the +# caller's, so this isn't a full fix, just a confidence floor). +NO_SPEECH_PROB_CUTOFF = 0.6 def is_unrecoverable_cuda_error(e: Exception) -> bool: @@ -140,8 +145,9 @@ class Engine: try: segments, info = self.stt_model.transcribe( path, language=language or None, beam_size=5, vad_filter=True, condition_on_previous_text=False, + hallucination_silence_threshold=2.0, ) - text = ''.join(seg.text for seg in segments).strip() + text = ''.join(seg.text for seg in segments if seg.no_speech_prob < NO_SPEECH_PROB_CUTOFF).strip() return text, (info.language if info else language) finally: try: @@ -173,8 +179,9 @@ class Engine: audio = pcm16_bytes_to_float(pcm) segments, info = self.stt_model.transcribe( audio, language=language or None, beam_size=1, vad_filter=True, condition_on_previous_text=False, + hallucination_silence_threshold=2.0, ) - text = ''.join(seg.text for seg in segments).strip() + text = ''.join(seg.text for seg in segments if seg.no_speech_prob < NO_SPEECH_PROB_CUTOFF).strip() return text, (info.language if info else language) # ── "Streaming" TTS ─────────────────────────────────────────────────