fix: 음성통화 STT 잡음/겹침 목소리 환각 완화
no_speech_prob가 높은 세그먼트를 텍스트 조합에서 제외하고 hallucination_silence_threshold를 걸어, 배경 소음이나 다른 사람 목소리가 섞였을 때 엉뚱한 텍스트가 연속으로 나오는 문제를 완화. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -53,6 +53,11 @@ log = logging.getLogger('voice_engine')
|
|||||||
logging.getLogger('faster_whisper').setLevel(logging.WARNING)
|
logging.getLogger('faster_whisper').setLevel(logging.WARNING)
|
||||||
|
|
||||||
MIN_PARTIAL_BYTES = int(16000 * 2 * 0.6) # ~0.6s of 16kHz mono PCM16 before we bother decoding
|
MIN_PARTIAL_BYTES = int(16000 * 2 * 0.6) # ~0.6s of 16kHz mono PCM16 before we bother decoding
|
||||||
|
# Segments where Whisper itself flags no_speech_prob above this are dropped before
|
||||||
|
# joining the transcript — catches noise/background-voice hallucination (a garbled
|
||||||
|
# guess is still emitted with low no_speech_prob when a second voice overlaps the
|
||||||
|
# caller's, so this isn't a full fix, just a confidence floor).
|
||||||
|
NO_SPEECH_PROB_CUTOFF = 0.6
|
||||||
|
|
||||||
|
|
||||||
def is_unrecoverable_cuda_error(e: Exception) -> bool:
|
def is_unrecoverable_cuda_error(e: Exception) -> bool:
|
||||||
@@ -140,8 +145,9 @@ class Engine:
|
|||||||
try:
|
try:
|
||||||
segments, info = self.stt_model.transcribe(
|
segments, info = self.stt_model.transcribe(
|
||||||
path, language=language or None, beam_size=5, vad_filter=True, condition_on_previous_text=False,
|
path, language=language or None, beam_size=5, vad_filter=True, condition_on_previous_text=False,
|
||||||
|
hallucination_silence_threshold=2.0,
|
||||||
)
|
)
|
||||||
text = ''.join(seg.text for seg in segments).strip()
|
text = ''.join(seg.text for seg in segments if seg.no_speech_prob < NO_SPEECH_PROB_CUTOFF).strip()
|
||||||
return text, (info.language if info else language)
|
return text, (info.language if info else language)
|
||||||
finally:
|
finally:
|
||||||
try:
|
try:
|
||||||
@@ -173,8 +179,9 @@ class Engine:
|
|||||||
audio = pcm16_bytes_to_float(pcm)
|
audio = pcm16_bytes_to_float(pcm)
|
||||||
segments, info = self.stt_model.transcribe(
|
segments, info = self.stt_model.transcribe(
|
||||||
audio, language=language or None, beam_size=1, vad_filter=True, condition_on_previous_text=False,
|
audio, language=language or None, beam_size=1, vad_filter=True, condition_on_previous_text=False,
|
||||||
|
hallucination_silence_threshold=2.0,
|
||||||
)
|
)
|
||||||
text = ''.join(seg.text for seg in segments).strip()
|
text = ''.join(seg.text for seg in segments if seg.no_speech_prob < NO_SPEECH_PROB_CUTOFF).strip()
|
||||||
return text, (info.language if info else language)
|
return text, (info.language if info else language)
|
||||||
|
|
||||||
# ── "Streaming" TTS ─────────────────────────────────────────────────
|
# ── "Streaming" TTS ─────────────────────────────────────────────────
|
||||||
|
|||||||
Reference in New Issue
Block a user