diff --git a/src/tools/voice_engine.py b/src/tools/voice_engine.py index c34aec0..1199b39 100644 --- a/src/tools/voice_engine.py +++ b/src/tools/voice_engine.py @@ -53,6 +53,11 @@ log = logging.getLogger('voice_engine') logging.getLogger('faster_whisper').setLevel(logging.WARNING) MIN_PARTIAL_BYTES = int(16000 * 2 * 0.6) # ~0.6s of 16kHz mono PCM16 before we bother decoding +# Segments where Whisper itself flags no_speech_prob above this are dropped before +# joining the transcript — catches noise/background-voice hallucination (a garbled +# guess is still emitted with low no_speech_prob when a second voice overlaps the +# caller's, so this isn't a full fix, just a confidence floor). +NO_SPEECH_PROB_CUTOFF = 0.6 def is_unrecoverable_cuda_error(e: Exception) -> bool: @@ -140,8 +145,9 @@ class Engine: try: segments, info = self.stt_model.transcribe( path, language=language or None, beam_size=5, vad_filter=True, condition_on_previous_text=False, + hallucination_silence_threshold=2.0, ) - text = ''.join(seg.text for seg in segments).strip() + text = ''.join(seg.text for seg in segments if seg.no_speech_prob < NO_SPEECH_PROB_CUTOFF).strip() return text, (info.language if info else language) finally: try: @@ -173,8 +179,9 @@ class Engine: audio = pcm16_bytes_to_float(pcm) segments, info = self.stt_model.transcribe( audio, language=language or None, beam_size=1, vad_filter=True, condition_on_previous_text=False, + hallucination_silence_threshold=2.0, ) - text = ''.join(seg.text for seg in segments).strip() + text = ''.join(seg.text for seg in segments if seg.no_speech_prob < NO_SPEECH_PROB_CUTOFF).strip() return text, (info.language if info else language) # ── "Streaming" TTS ─────────────────────────────────────────────────