fix: 음성통화 STT 환각 완화 + 툴콜 턴 TTS 누락 수정 + 볼륨 슬라이더
- voice_engine.py: condition_on_previous_text=False로 whisper가 애매한 구간에서 그럴듯한 문장을 지어내는 환각 완화 - app.js/voice-call.js: 툴 사용 턴(웹검색 등)에서 finalAnswer가 partialContent를 통째로 덮어쓰면서 음성통화의 spokenUpTo 오프셋이 무효화돼 텍스트는 나오는데 음성이 안 나오는 버그 수정 - index.html/voice-call.js: 통화 중 GainNode 기반 인앱 볼륨 슬라이더 추가 (일부 모바일 브라우저에서 마이크 사용 중엔 하드웨어 볼륨 버튼이 통화 오디오에 반영 안 되는 문제 우회) Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
+26
-3
@@ -19,7 +19,7 @@
|
||||
let ws = null;
|
||||
let captureCtx = null, playCtx = null;
|
||||
let micStream = null;
|
||||
let captureNode = null, playerNode = null;
|
||||
let captureNode = null, playerNode = null, gainNode = null;
|
||||
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
|
||||
let speechChunks = 0;
|
||||
let silenceMs = 0;
|
||||
@@ -62,6 +62,13 @@
|
||||
else startVoiceCall();
|
||||
};
|
||||
|
||||
window.voiceCallSetVolume = function (v) {
|
||||
const vol = parseFloat(v);
|
||||
if (!Number.isFinite(vol)) return;
|
||||
localStorage.setItem('voiceCallVolume', String(vol));
|
||||
if (gainNode) gainNode.gain.value = vol;
|
||||
};
|
||||
|
||||
window.voiceCallHangup = function () {
|
||||
active = false;
|
||||
state = 'idle';
|
||||
@@ -70,7 +77,7 @@
|
||||
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
|
||||
try { captureCtx && captureCtx.close(); } catch {}
|
||||
try { playCtx && playCtx.close(); } catch {}
|
||||
captureCtx = playCtx = micStream = captureNode = playerNode = null;
|
||||
captureCtx = playCtx = micStream = captureNode = playerNode = gainNode = null;
|
||||
try { wakeLock && wakeLock.release(); } catch {}
|
||||
wakeLock = null;
|
||||
ttsQueue = []; ttsBusy = false; currentTtsId = null;
|
||||
@@ -110,7 +117,16 @@
|
||||
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
|
||||
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
|
||||
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
|
||||
playerNode.connect(playCtx.destination);
|
||||
// A GainNode we control from the in-app slider — on some mobile browsers
|
||||
// the hardware volume rocker maps to the call/mic audio session while a
|
||||
// getUserMedia stream is open, not to this Web Audio output, so the phone's
|
||||
// own volume slider doesn't reliably affect playback here.
|
||||
gainNode = playCtx.createGain();
|
||||
const savedVolume = parseFloat(localStorage.getItem('voiceCallVolume') || '1');
|
||||
gainNode.gain.value = savedVolume;
|
||||
const volumeSlider = document.getElementById('voice-call-volume');
|
||||
if (volumeSlider) volumeSlider.value = String(savedVolume);
|
||||
playerNode.connect(gainNode).connect(playCtx.destination);
|
||||
|
||||
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
|
||||
const token = getAuthToken();
|
||||
@@ -298,6 +314,13 @@
|
||||
return { sentences, consumedLength: start };
|
||||
}
|
||||
|
||||
// Hooked from app.js when partialContent is replaced wholesale rather than
|
||||
// appended to (e.g. a multi-step tool turn's finalAnswer) — spokenUpTo was an
|
||||
// offset into the old string and must be dropped before the new text arrives.
|
||||
window._voiceCallResetSpoken = function () {
|
||||
spokenUpTo = 0;
|
||||
};
|
||||
|
||||
// Hooked from app.js's SSE token handler — only acts while a call is active.
|
||||
window._voiceCallOnToken = function (fullPartialContent) {
|
||||
if (!active) return;
|
||||
|
||||
Reference in New Issue
Block a user