fix: 음성통화 STT 환각 완화 + 툴콜 턴 TTS 누락 수정 + 볼륨 슬라이더

- voice_engine.py: condition_on_previous_text=False로 whisper가 애매한
  구간에서 그럴듯한 문장을 지어내는 환각 완화
- app.js/voice-call.js: 툴 사용 턴(웹검색 등)에서 finalAnswer가 partialContent를
  통째로 덮어쓰면서 음성통화의 spokenUpTo 오프셋이 무효화돼 텍스트는 나오는데
  음성이 안 나오는 버그 수정
- index.html/voice-call.js: 통화 중 GainNode 기반 인앱 볼륨 슬라이더 추가
  (일부 모바일 브라우저에서 마이크 사용 중엔 하드웨어 볼륨 버튼이 통화
  오디오에 반영 안 되는 문제 우회)

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
kim
2026-07-09 23:58:25 +09:00
co-authored by Claude Sonnet 5
parent 73967d1f55
commit b8bfb501b9
4 changed files with 58 additions and 5 deletions
+8
View File
@@ -1561,6 +1561,7 @@ async function sendChat(queuedMessage = null) {
const token = event.text || event.content || event.delta || '';
if (token) {
partialContent += token;
if (window._voiceCallOnToken) window._voiceCallOnToken(partialContent);
}
break;
}
@@ -1813,6 +1814,11 @@ async function sendChat(queuedMessage = null) {
if (s.finalAnswer) {
addProcessEntry('final', s.finalAnswer);
partialContent = s.finalAnswer; // track for stop
// partialContent just got replaced wholesale (not appended to), so
// any voice-call spokenUpTo offset computed against the old string
// is now meaningless — reset before feeding the new text through.
if (window._voiceCallResetSpoken) window._voiceCallResetSpoken();
if (window._voiceCallOnToken) window._voiceCallOnToken(partialContent);
}
break;
}
@@ -1903,6 +1909,8 @@ async function sendChat(queuedMessage = null) {
}
}
if (window._voiceCallOnTurnDone) window._voiceCallOnTurnDone(finalReply || partialContent);
if (finalReply) {
// Append file cards for any attachments sent via the 'files' SSE event
if (turnFileLinks.length > 0) {
+18
View File
@@ -387,10 +387,18 @@
</div>
<!-- 첨부 파일 카드 영역 (전송 전 표시) -->
<div id="staged-attachments" style="display:none;flex-wrap:wrap;gap:8px;padding:8px 12px 0;border-top:1px solid var(--line)"></div>
<div id="voice-call-bar" style="display:none;align-items:center;gap:8px;padding:8px 12px;border-top:1px solid var(--line);font-size:13px;color:var(--text-dim,#888)">
<span id="voice-call-status">통화 연결 중…</span>
<span id="voice-call-caption" style="flex:1;opacity:0.85;font-style:italic"></span>
<span aria-hidden="true">🔈</span>
<input type="range" id="voice-call-volume" min="0" max="1.5" step="0.05" value="1" oninput="voiceCallSetVolume(this.value)" title="통화 음량" style="width:70px" />
<button class="send-btn" onclick="voiceCallHangup()" title="통화 종료" style="font-size:14px;width:32px;height:32px;background:#c0392b" aria-label="통화 종료">✕</button>
</div>
<div class="chat-input-row">
<input type="file" id="file-upload" accept="image/*,.pdf,.txt,.md,.csv,.xlsx,.xls,.docx,.pptx,.mid,.midi" style="display:none" onchange="handleFileUpload(this)" />
<button class="send-btn" onclick="document.getElementById('file-upload').click()" title="파일 첨부 (이미지·PDF·Excel·txt 등)" style="font-size:18px;width:44px;height:44px" aria-label="파일 첨부">📎</button>
<button class="send-btn" id="mic-btn" onclick="toggleMicRecording()" title="음성 입력 (클릭: 시작/중지)" style="font-size:18px;width:44px;height:44px" aria-label="음성 입력">🎤</button>
<button class="send-btn" id="voice-call-btn" onclick="toggleVoiceCall()" title="실시간 음성 대화 (연속 대화 모드)" style="font-size:18px;width:44px;height:44px" aria-label="실시간 음성 대화">📞</button>
<textarea
id="chat-input"
placeholder="메시지를 입력하세요... (Enter로 전송, Shift+Enter로 줄바꿈)"
@@ -1905,6 +1913,15 @@
<div id="cv-sessions-list" style="display:flex;flex-direction:column;gap:10px">
<div style="color:var(--muted);font-size:13px;text-align:center;padding:20px">로딩 중...</div>
</div>
<div style="display:flex;align-items:center;justify-content:space-between;margin:20px 0 6px">
<div class="right-section-title">fail2ban 차단 IP</div>
<button class="btn btn-sm" style="background:var(--panel-2);border:1px solid var(--line);color:var(--muted)" onclick="loadBannedIpsTab()">새로고침</button>
</div>
<div id="cv-banned-ips-stat" style="font-size:11px;color:var(--muted);margin-bottom:8px"></div>
<div id="cv-banned-ips-list" style="display:flex;flex-direction:column;gap:6px">
<div style="color:var(--muted);font-size:13px;text-align:center;padding:20px">로딩 중...</div>
</div>
</div>
<div style="display:flex;justify-content:flex-end;gap:8px;padding:12px 16px;border-top:1px solid var(--line)">
@@ -1914,6 +1931,7 @@
</div>
</div>
<script src="voice-call.js"></script>
<script src="app.js"></script>
<!-- Code folder browser dropdown (fixed, rendered outside any overflow:hidden container) -->
+26 -3
View File
@@ -19,7 +19,7 @@
let ws = null;
let captureCtx = null, playCtx = null;
let micStream = null;
let captureNode = null, playerNode = null;
let captureNode = null, playerNode = null, gainNode = null;
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
let speechChunks = 0;
let silenceMs = 0;
@@ -62,6 +62,13 @@
else startVoiceCall();
};
window.voiceCallSetVolume = function (v) {
const vol = parseFloat(v);
if (!Number.isFinite(vol)) return;
localStorage.setItem('voiceCallVolume', String(vol));
if (gainNode) gainNode.gain.value = vol;
};
window.voiceCallHangup = function () {
active = false;
state = 'idle';
@@ -70,7 +77,7 @@
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
try { captureCtx && captureCtx.close(); } catch {}
try { playCtx && playCtx.close(); } catch {}
captureCtx = playCtx = micStream = captureNode = playerNode = null;
captureCtx = playCtx = micStream = captureNode = playerNode = gainNode = null;
try { wakeLock && wakeLock.release(); } catch {}
wakeLock = null;
ttsQueue = []; ttsBusy = false; currentTtsId = null;
@@ -110,7 +117,16 @@
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
playerNode.connect(playCtx.destination);
// A GainNode we control from the in-app slider — on some mobile browsers
// the hardware volume rocker maps to the call/mic audio session while a
// getUserMedia stream is open, not to this Web Audio output, so the phone's
// own volume slider doesn't reliably affect playback here.
gainNode = playCtx.createGain();
const savedVolume = parseFloat(localStorage.getItem('voiceCallVolume') || '1');
gainNode.gain.value = savedVolume;
const volumeSlider = document.getElementById('voice-call-volume');
if (volumeSlider) volumeSlider.value = String(savedVolume);
playerNode.connect(gainNode).connect(playCtx.destination);
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
const token = getAuthToken();
@@ -298,6 +314,13 @@
return { sentences, consumedLength: start };
}
// Hooked from app.js when partialContent is replaced wholesale rather than
// appended to (e.g. a multi-step tool turn's finalAnswer) — spokenUpTo was an
// offset into the old string and must be dropped before the new text arrives.
window._voiceCallResetSpoken = function () {
spokenUpTo = 0;
};
// Hooked from app.js's SSE token handler — only acts while a call is active.
window._voiceCallOnToken = function (fullPartialContent) {
if (!active) return;