fix: 음성통화 STT 잡음/겹침 목소리 환각 완화

no_speech_prob가 높은 세그먼트를 텍스트 조합에서 제외하고
hallucination_silence_threshold를 걸어, 배경 소음이나 다른 사람
목소리가 섞였을 때 엉뚱한 텍스트가 연속으로 나오는 문제를 완화.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-11 02:04:20 +09:00
co-authored by Claude Sonnet 5
parent 64f7e5ec3b
commit cec930c69c
+9 -2
View File
@@ -53,6 +53,11 @@ log = logging.getLogger('voice_engine')
logging.getLogger('faster_whisper').setLevel(logging.WARNING)
MIN_PARTIAL_BYTES = int(16000 * 2 * 0.6) # ~0.6s of 16kHz mono PCM16 before we bother decoding
# Segments where Whisper itself flags no_speech_prob above this are dropped before
# joining the transcript — catches noise/background-voice hallucination (a garbled
# guess is still emitted with low no_speech_prob when a second voice overlaps the
# caller's, so this isn't a full fix, just a confidence floor).
NO_SPEECH_PROB_CUTOFF = 0.6
def is_unrecoverable_cuda_error(e: Exception) -> bool:
@@ -140,8 +145,9 @@ class Engine:
try:
segments, info = self.stt_model.transcribe(
path, language=language or None, beam_size=5, vad_filter=True, condition_on_previous_text=False,
hallucination_silence_threshold=2.0,
)
text = ''.join(seg.text for seg in segments).strip()
text = ''.join(seg.text for seg in segments if seg.no_speech_prob < NO_SPEECH_PROB_CUTOFF).strip()
return text, (info.language if info else language)
finally:
try:
@@ -173,8 +179,9 @@ class Engine:
audio = pcm16_bytes_to_float(pcm)
segments, info = self.stt_model.transcribe(
audio, language=language or None, beam_size=1, vad_filter=True, condition_on_previous_text=False,
hallucination_silence_threshold=2.0,
)
text = ''.join(seg.text for seg in segments).strip()
text = ''.join(seg.text for seg in segments if seg.no_speech_prob < NO_SPEECH_PROB_CUTOFF).strip()
return text, (info.language if info else language)
# ── "Streaming" TTS ─────────────────────────────────────────────────