fix: 음성통화 STT 환각 완화 + 툴콜 턴 TTS 누락 수정 + 볼륨 슬라이더
- voice_engine.py: condition_on_previous_text=False로 whisper가 애매한 구간에서 그럴듯한 문장을 지어내는 환각 완화 - app.js/voice-call.js: 툴 사용 턴(웹검색 등)에서 finalAnswer가 partialContent를 통째로 덮어쓰면서 음성통화의 spokenUpTo 오프셋이 무효화돼 텍스트는 나오는데 음성이 안 나오는 버그 수정 - index.html/voice-call.js: 통화 중 GainNode 기반 인앱 볼륨 슬라이더 추가 (일부 모바일 브라우저에서 마이크 사용 중엔 하드웨어 볼륨 버튼이 통화 오디오에 반영 안 되는 문제 우회) Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -119,7 +119,9 @@ class Engine:
|
||||
f.write(audio_bytes)
|
||||
path = f.name
|
||||
try:
|
||||
segments, info = self.stt_model.transcribe(path, language=language or None, beam_size=5, vad_filter=True)
|
||||
segments, info = self.stt_model.transcribe(
|
||||
path, language=language or None, beam_size=5, vad_filter=True, condition_on_previous_text=False,
|
||||
)
|
||||
text = ''.join(seg.text for seg in segments).strip()
|
||||
return text, (info.language if info else language)
|
||||
finally:
|
||||
@@ -140,7 +142,9 @@ class Engine:
|
||||
# ── Streaming STT (partial decode of an accumulating PCM buffer) ──────
|
||||
def transcribe_pcm(self, pcm: bytes, language: str):
|
||||
audio = pcm16_bytes_to_float(pcm)
|
||||
segments, info = self.stt_model.transcribe(audio, language=language or None, beam_size=1, vad_filter=True)
|
||||
segments, info = self.stt_model.transcribe(
|
||||
audio, language=language or None, beam_size=1, vad_filter=True, condition_on_previous_text=False,
|
||||
)
|
||||
text = ''.join(seg.text for seg in segments).strip()
|
||||
return text, (info.language if info else language)
|
||||
|
||||
|
||||
@@ -1561,6 +1561,7 @@ async function sendChat(queuedMessage = null) {
|
||||
const token = event.text || event.content || event.delta || '';
|
||||
if (token) {
|
||||
partialContent += token;
|
||||
if (window._voiceCallOnToken) window._voiceCallOnToken(partialContent);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -1813,6 +1814,11 @@ async function sendChat(queuedMessage = null) {
|
||||
if (s.finalAnswer) {
|
||||
addProcessEntry('final', s.finalAnswer);
|
||||
partialContent = s.finalAnswer; // track for stop
|
||||
// partialContent just got replaced wholesale (not appended to), so
|
||||
// any voice-call spokenUpTo offset computed against the old string
|
||||
// is now meaningless — reset before feeding the new text through.
|
||||
if (window._voiceCallResetSpoken) window._voiceCallResetSpoken();
|
||||
if (window._voiceCallOnToken) window._voiceCallOnToken(partialContent);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -1903,6 +1909,8 @@ async function sendChat(queuedMessage = null) {
|
||||
}
|
||||
}
|
||||
|
||||
if (window._voiceCallOnTurnDone) window._voiceCallOnTurnDone(finalReply || partialContent);
|
||||
|
||||
if (finalReply) {
|
||||
// Append file cards for any attachments sent via the 'files' SSE event
|
||||
if (turnFileLinks.length > 0) {
|
||||
|
||||
@@ -387,10 +387,18 @@
|
||||
</div>
|
||||
<!-- 첨부 파일 카드 영역 (전송 전 표시) -->
|
||||
<div id="staged-attachments" style="display:none;flex-wrap:wrap;gap:8px;padding:8px 12px 0;border-top:1px solid var(--line)"></div>
|
||||
<div id="voice-call-bar" style="display:none;align-items:center;gap:8px;padding:8px 12px;border-top:1px solid var(--line);font-size:13px;color:var(--text-dim,#888)">
|
||||
<span id="voice-call-status">통화 연결 중…</span>
|
||||
<span id="voice-call-caption" style="flex:1;opacity:0.85;font-style:italic"></span>
|
||||
<span aria-hidden="true">🔈</span>
|
||||
<input type="range" id="voice-call-volume" min="0" max="1.5" step="0.05" value="1" oninput="voiceCallSetVolume(this.value)" title="통화 음량" style="width:70px" />
|
||||
<button class="send-btn" onclick="voiceCallHangup()" title="통화 종료" style="font-size:14px;width:32px;height:32px;background:#c0392b" aria-label="통화 종료">✕</button>
|
||||
</div>
|
||||
<div class="chat-input-row">
|
||||
<input type="file" id="file-upload" accept="image/*,.pdf,.txt,.md,.csv,.xlsx,.xls,.docx,.pptx,.mid,.midi" style="display:none" onchange="handleFileUpload(this)" />
|
||||
<button class="send-btn" onclick="document.getElementById('file-upload').click()" title="파일 첨부 (이미지·PDF·Excel·txt 등)" style="font-size:18px;width:44px;height:44px" aria-label="파일 첨부">📎</button>
|
||||
<button class="send-btn" id="mic-btn" onclick="toggleMicRecording()" title="음성 입력 (클릭: 시작/중지)" style="font-size:18px;width:44px;height:44px" aria-label="음성 입력">🎤</button>
|
||||
<button class="send-btn" id="voice-call-btn" onclick="toggleVoiceCall()" title="실시간 음성 대화 (연속 대화 모드)" style="font-size:18px;width:44px;height:44px" aria-label="실시간 음성 대화">📞</button>
|
||||
<textarea
|
||||
id="chat-input"
|
||||
placeholder="메시지를 입력하세요... (Enter로 전송, Shift+Enter로 줄바꿈)"
|
||||
@@ -1905,6 +1913,15 @@
|
||||
<div id="cv-sessions-list" style="display:flex;flex-direction:column;gap:10px">
|
||||
<div style="color:var(--muted);font-size:13px;text-align:center;padding:20px">로딩 중...</div>
|
||||
</div>
|
||||
|
||||
<div style="display:flex;align-items:center;justify-content:space-between;margin:20px 0 6px">
|
||||
<div class="right-section-title">fail2ban 차단 IP</div>
|
||||
<button class="btn btn-sm" style="background:var(--panel-2);border:1px solid var(--line);color:var(--muted)" onclick="loadBannedIpsTab()">새로고침</button>
|
||||
</div>
|
||||
<div id="cv-banned-ips-stat" style="font-size:11px;color:var(--muted);margin-bottom:8px"></div>
|
||||
<div id="cv-banned-ips-list" style="display:flex;flex-direction:column;gap:6px">
|
||||
<div style="color:var(--muted);font-size:13px;text-align:center;padding:20px">로딩 중...</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div style="display:flex;justify-content:flex-end;gap:8px;padding:12px 16px;border-top:1px solid var(--line)">
|
||||
@@ -1914,6 +1931,7 @@
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<script src="voice-call.js"></script>
|
||||
<script src="app.js"></script>
|
||||
|
||||
<!-- Code folder browser dropdown (fixed, rendered outside any overflow:hidden container) -->
|
||||
|
||||
+26
-3
@@ -19,7 +19,7 @@
|
||||
let ws = null;
|
||||
let captureCtx = null, playCtx = null;
|
||||
let micStream = null;
|
||||
let captureNode = null, playerNode = null;
|
||||
let captureNode = null, playerNode = null, gainNode = null;
|
||||
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
|
||||
let speechChunks = 0;
|
||||
let silenceMs = 0;
|
||||
@@ -62,6 +62,13 @@
|
||||
else startVoiceCall();
|
||||
};
|
||||
|
||||
window.voiceCallSetVolume = function (v) {
|
||||
const vol = parseFloat(v);
|
||||
if (!Number.isFinite(vol)) return;
|
||||
localStorage.setItem('voiceCallVolume', String(vol));
|
||||
if (gainNode) gainNode.gain.value = vol;
|
||||
};
|
||||
|
||||
window.voiceCallHangup = function () {
|
||||
active = false;
|
||||
state = 'idle';
|
||||
@@ -70,7 +77,7 @@
|
||||
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
|
||||
try { captureCtx && captureCtx.close(); } catch {}
|
||||
try { playCtx && playCtx.close(); } catch {}
|
||||
captureCtx = playCtx = micStream = captureNode = playerNode = null;
|
||||
captureCtx = playCtx = micStream = captureNode = playerNode = gainNode = null;
|
||||
try { wakeLock && wakeLock.release(); } catch {}
|
||||
wakeLock = null;
|
||||
ttsQueue = []; ttsBusy = false; currentTtsId = null;
|
||||
@@ -110,7 +117,16 @@
|
||||
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
|
||||
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
|
||||
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
|
||||
playerNode.connect(playCtx.destination);
|
||||
// A GainNode we control from the in-app slider — on some mobile browsers
|
||||
// the hardware volume rocker maps to the call/mic audio session while a
|
||||
// getUserMedia stream is open, not to this Web Audio output, so the phone's
|
||||
// own volume slider doesn't reliably affect playback here.
|
||||
gainNode = playCtx.createGain();
|
||||
const savedVolume = parseFloat(localStorage.getItem('voiceCallVolume') || '1');
|
||||
gainNode.gain.value = savedVolume;
|
||||
const volumeSlider = document.getElementById('voice-call-volume');
|
||||
if (volumeSlider) volumeSlider.value = String(savedVolume);
|
||||
playerNode.connect(gainNode).connect(playCtx.destination);
|
||||
|
||||
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
|
||||
const token = getAuthToken();
|
||||
@@ -298,6 +314,13 @@
|
||||
return { sentences, consumedLength: start };
|
||||
}
|
||||
|
||||
// Hooked from app.js when partialContent is replaced wholesale rather than
|
||||
// appended to (e.g. a multi-step tool turn's finalAnswer) — spokenUpTo was an
|
||||
// offset into the old string and must be dropped before the new text arrives.
|
||||
window._voiceCallResetSpoken = function () {
|
||||
spokenUpTo = 0;
|
||||
};
|
||||
|
||||
// Hooked from app.js's SSE token handler — only acts while a call is active.
|
||||
window._voiceCallOnToken = function (fullPartialContent) {
|
||||
if (!active) return;
|
||||
|
||||
Reference in New Issue
Block a user