fix: 음성통화 STT 환각 완화 + 툴콜 턴 TTS 누락 수정 + 볼륨 슬라이더

- voice_engine.py: condition_on_previous_text=False로 whisper가 애매한
  구간에서 그럴듯한 문장을 지어내는 환각 완화
- app.js/voice-call.js: 툴 사용 턴(웹검색 등)에서 finalAnswer가 partialContent를
  통째로 덮어쓰면서 음성통화의 spokenUpTo 오프셋이 무효화돼 텍스트는 나오는데
  음성이 안 나오는 버그 수정
- index.html/voice-call.js: 통화 중 GainNode 기반 인앱 볼륨 슬라이더 추가
  (일부 모바일 브라우저에서 마이크 사용 중엔 하드웨어 볼륨 버튼이 통화
  오디오에 반영 안 되는 문제 우회)

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-09 23:58:25 +09:00
co-authored by Claude Sonnet 5
parent 73967d1f55
commit b8bfb501b9
4 changed files with 58 additions and 5 deletions
+26 -3
View File
@@ -19,7 +19,7 @@
let ws = null;
let captureCtx = null, playCtx = null;
let micStream = null;
let captureNode = null, playerNode = null;
let captureNode = null, playerNode = null, gainNode = null;
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
let speechChunks = 0;
let silenceMs = 0;
@@ -62,6 +62,13 @@
else startVoiceCall();
};
window.voiceCallSetVolume = function (v) {
const vol = parseFloat(v);
if (!Number.isFinite(vol)) return;
localStorage.setItem('voiceCallVolume', String(vol));
if (gainNode) gainNode.gain.value = vol;
};
window.voiceCallHangup = function () {
active = false;
state = 'idle';
@@ -70,7 +77,7 @@
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
try { captureCtx && captureCtx.close(); } catch {}
try { playCtx && playCtx.close(); } catch {}
captureCtx = playCtx = micStream = captureNode = playerNode = null;
captureCtx = playCtx = micStream = captureNode = playerNode = gainNode = null;
try { wakeLock && wakeLock.release(); } catch {}
wakeLock = null;
ttsQueue = []; ttsBusy = false; currentTtsId = null;
@@ -110,7 +117,16 @@
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
playerNode.connect(playCtx.destination);
// A GainNode we control from the in-app slider — on some mobile browsers
// the hardware volume rocker maps to the call/mic audio session while a
// getUserMedia stream is open, not to this Web Audio output, so the phone's
// own volume slider doesn't reliably affect playback here.
gainNode = playCtx.createGain();
const savedVolume = parseFloat(localStorage.getItem('voiceCallVolume') || '1');
gainNode.gain.value = savedVolume;
const volumeSlider = document.getElementById('voice-call-volume');
if (volumeSlider) volumeSlider.value = String(savedVolume);
playerNode.connect(gainNode).connect(playCtx.destination);
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
const token = getAuthToken();
@@ -298,6 +314,13 @@
return { sentences, consumedLength: start };
}
// Hooked from app.js when partialContent is replaced wholesale rather than
// appended to (e.g. a multi-step tool turn's finalAnswer) — spokenUpTo was an
// offset into the old string and must be dropped before the new text arrives.
window._voiceCallResetSpoken = function () {
spokenUpTo = 0;
};
// Hooked from app.js's SSE token handler — only acts while a call is active.
window._voiceCallOnToken = function (fullPartialContent) {
if (!active) return;