Files
homeclaw/web-ui/voice-call.js
T
kimandClaude Sonnet 5 34fac14b20 fix: 음성통화 TTS 괄호 문자 처리 + error 응답 시 큐 멈춤 버그 수정
OmniVoice가 "(" 문자(특히 한글에 바로 붙은 경우, 예: 영웅(Heroic))를 합성하다
멈춰버리는 문제가 있어 sanitizeForSpeech에서 괄호를 제거하도록 변경.
또한 TTS 엔진이 합성 실패 시 보내는 error 메시지를 클라이언트가 로그만 찍고
ttsBusy를 리셋하지 않아, 이후 통화 내내 TTS가 먹통이 되는 버그를 tts_end와
동일하게 큐를 복구하도록 수정.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-07-10 00:42:16 +09:00

360 lines
14 KiB
JavaScript

// ── Voice: 실시간 연속 대화 모드 ───────────────────────────────────────────
// AudioWorklet 캡처(16kHz PCM16) → /ws/voice-realtime → GPU STT/TTS 스트리밍.
// VAD(에너지 기반)로 발화 시작/끝을 로컬에서 판단해 stt_start/stop을 보내고,
// LLM 응답 SSE 토큰에서 완성되는 문장 단위로 잘라 TTS를 파이프라이닝한다.
// barge-in: 재생 중 사용자가 다시 말하면 즉시 재생을 끊고 새 인식을 시작한다.
(function () {
const RMS_SPEECH_ON = 0.020;
const RMS_SPEECH_OFF = 0.012;
const SPEECH_CONFIRM_CHUNKS = 2; // ~200ms of sustained energy to confirm speech start
const SILENCE_HANGOVER_MS = 700; // silence before we consider the utterance finished
const CHUNK_MS = 100;
const PREROLL_CHUNKS = 3; // ~300ms of lead-in kept before speech is confirmed
// XTTS's streaming inference_stream() can crash (CUDA device-side assert,
// poisons the whole GPU context) on very short standalone text — merge short
// sentence fragments together before sending them off as one TTS request.
const MIN_TTS_CHARS = 15;
let ws = null;
let captureCtx = null, playCtx = null;
let micStream = null;
let captureNode = null, playerNode = null, gainNode = null;
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
let speechChunks = 0;
let silenceMs = 0;
let preRoll = [];
let ttsQueue = [];
let ttsBusy = false;
let currentTtsId = null;
let spokenUpTo = 0;
let pendingTts = '';
let active = false;
let wakeLock = null;
// The Wake Lock is auto-released whenever the tab loses visibility (screen off,
// app-switch) — without it, mobile browsers throttle/suspend the background tab
// and the call silently drops. Re-acquire it once the tab is visible again.
async function acquireWakeLock() {
if (!('wakeLock' in navigator)) return;
try {
wakeLock = await navigator.wakeLock.request('screen');
wakeLock.addEventListener('release', () => { wakeLock = null; });
} catch (e) {
console.warn('[voice-call] wakeLock request failed:', e.message);
}
}
document.addEventListener('visibilitychange', () => {
if (active && wakeLock === null && document.visibilityState === 'visible') acquireWakeLock();
});
function setStatus(text) {
const el = document.getElementById('voice-call-status');
if (el) el.textContent = text;
}
function setCaption(text) {
const el = document.getElementById('voice-call-caption');
if (el) el.textContent = text || '';
}
window.toggleVoiceCall = function () {
if (active) voiceCallHangup();
else startVoiceCall();
};
window.voiceCallSetVolume = function (v) {
const vol = parseFloat(v);
if (!Number.isFinite(vol)) return;
localStorage.setItem('voiceCallVolume', String(vol));
if (gainNode) gainNode.gain.value = vol;
};
window.voiceCallHangup = function () {
active = false;
state = 'idle';
try { ws && ws.close(); } catch {}
ws = null;
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
try { captureCtx && captureCtx.close(); } catch {}
try { playCtx && playCtx.close(); } catch {}
captureCtx = playCtx = micStream = captureNode = playerNode = gainNode = null;
try { wakeLock && wakeLock.release(); } catch {}
wakeLock = null;
ttsQueue = []; ttsBusy = false; currentTtsId = null;
const bar = document.getElementById('voice-call-bar');
if (bar) bar.style.display = 'none';
const btn = document.getElementById('voice-call-btn');
if (btn) { btn.textContent = '📞'; btn.classList.remove('recording'); }
};
async function startVoiceCall() {
const btn = document.getElementById('voice-call-btn');
const bar = document.getElementById('voice-call-bar');
try {
micStream = await navigator.mediaDevices.getUserMedia({ audio: { channelCount: 1, echoCancellation: true, noiseSuppression: true } });
} catch (e) {
alert('마이크 권한이 필요합니다: ' + e.message);
return;
}
active = true;
acquireWakeLock();
if (btn) { btn.textContent = '📵'; btn.classList.add('recording'); }
if (bar) bar.style.display = 'flex';
setStatus('통화 연결 중…');
setCaption('');
state = 'listening';
speechChunks = 0; silenceMs = 0; preRoll = [];
spokenUpTo = 0; pendingTts = ''; ttsQueue = []; ttsBusy = false; currentTtsId = null;
captureCtx = new (window.AudioContext || window.webkitAudioContext)();
await captureCtx.audioWorklet.addModule('voice-call-worklets.js');
const source = captureCtx.createMediaStreamSource(micStream);
captureNode = new AudioWorkletNode(captureCtx, 'mic-capture-processor');
captureNode.port.onmessage = handleMicChunk;
source.connect(captureNode);
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
// A GainNode we control from the in-app slider — on some mobile browsers
// the hardware volume rocker maps to the call/mic audio session while a
// getUserMedia stream is open, not to this Web Audio output, so the phone's
// own volume slider doesn't reliably affect playback here.
gainNode = playCtx.createGain();
const savedVolume = parseFloat(localStorage.getItem('voiceCallVolume') || '1');
gainNode.gain.value = savedVolume;
const volumeSlider = document.getElementById('voice-call-volume');
if (volumeSlider) volumeSlider.value = String(savedVolume);
playerNode.connect(gainNode).connect(playCtx.destination);
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
const token = getAuthToken();
const qs = token ? `?token=${encodeURIComponent(token)}` : '';
ws = new WebSocket(`${proto}://${API.replace(/^https?:\/\//, '')}/ws/voice-realtime${qs}`);
ws.binaryType = 'arraybuffer';
ws.onopen = () => setStatus('듣는 중…');
ws.onmessage = handleWsMessage;
ws.onerror = () => setStatus('연결 오류');
ws.onclose = () => { if (active) voiceCallHangup(); };
}
function handleMicChunk(e) {
if (!active) return;
const { pcm, rms } = e.data;
if (state === 'listening' || state === 'assistant_speaking') {
preRoll.push(pcm);
if (preRoll.length > PREROLL_CHUNKS) preRoll.shift();
if (rms > RMS_SPEECH_ON) {
speechChunks++;
if (speechChunks >= SPEECH_CONFIRM_CHUNKS) onSpeechStart();
} else {
speechChunks = 0;
}
} else if (state === 'user_speaking') {
if (ws && ws.readyState === WebSocket.OPEN) ws.send(pcm.buffer);
if (rms < RMS_SPEECH_OFF) {
silenceMs += CHUNK_MS;
if (silenceMs >= SILENCE_HANGOVER_MS) onSpeechEnd();
} else {
silenceMs = 0;
}
}
}
function onSpeechStart() {
if (state === 'assistant_speaking') {
// barge-in: cut playback immediately and cancel whatever is still generating
ttsQueue = [];
if (playerNode) playerNode.port.postMessage({ type: 'clear' });
if (ws && ws.readyState === WebSocket.OPEN && currentTtsId) {
ws.send(JSON.stringify({ type: 'tts_cancel', id: currentTtsId }));
}
ttsBusy = false;
currentTtsId = null;
}
state = 'user_speaking';
silenceMs = 0;
setStatus('듣는 중…');
if (ws && ws.readyState === WebSocket.OPEN) {
ws.send(JSON.stringify({ type: 'stt_start', language: 'ko' }));
for (const chunk of preRoll) ws.send(chunk.buffer);
}
preRoll = [];
}
function onSpeechEnd() {
state = 'processing';
speechChunks = 0;
setStatus('생각 중…');
if (ws && ws.readyState === WebSocket.OPEN) {
ws.send(JSON.stringify({ type: 'stt_stop', id: 'turn-' + Date.now() }));
}
}
function handleWsMessage(e) {
if (e.data instanceof ArrayBuffer) {
if (playerNode) playerNode.port.postMessage({ type: 'push', pcm: new Int16Array(e.data) }, [e.data]);
return;
}
let msg;
try { msg = JSON.parse(e.data); } catch { return; }
switch (msg.type) {
case 'stt_partial':
setCaption(msg.text || '');
break;
case 'stt_final': {
setCaption('');
const text = (msg.text || '').trim();
if (!text) { state = 'listening'; setStatus('듣는 중…'); break; }
spokenUpTo = 0; pendingTts = '';
const input = document.getElementById('chat-input');
if (input) {
input.value = text;
input.dispatchEvent(new Event('input'));
}
setStatus('생각 중…');
setTimeout(() => (window._appSendFn || handleSendStop)(), 30);
break;
}
case 'tts_stream_start':
currentTtsId = msg.id;
state = 'assistant_speaking';
setStatus('말하는 중…');
break;
case 'tts_end':
ttsBusy = false;
currentTtsId = null;
pumpTtsQueue();
if (ttsQueue.length === 0) {
state = 'listening';
setStatus('듣는 중…');
}
break;
case 'error':
console.warn('[voice-call] engine error:', msg.message);
// The failed request may have been the one holding ttsBusy — without
// resetting it here, pumpTtsQueue() early-returns forever and the
// call goes silent for the rest of the session.
if (!msg.id || msg.id === currentTtsId) {
ttsBusy = false;
currentTtsId = null;
pumpTtsQueue();
if (ttsQueue.length === 0) {
state = 'listening';
setStatus('듣는 중…');
}
}
break;
}
}
function pumpTtsQueue() {
if (ttsBusy || ttsQueue.length === 0) return;
if (!ws || ws.readyState !== WebSocket.OPEN) return;
const text = ttsQueue.shift();
ttsBusy = true;
const id = 'tts-' + Date.now() + '-' + Math.random().toString(36).slice(2, 7);
currentTtsId = id;
ws.send(JSON.stringify({ type: 'tts_start', text, id }));
}
// Strips markdown syntax, tables, URLs and emoji so raw LLM output
// (**bold**, bullet markers, code blocks, | table | pipes |, links, 🦞
// etc.) doesn't get force-pronounced by XTTS under language='ko' — those
// symbols were producing garbled non-Korean-sounding artifacts mid-sentence
// (tables were the worst offender: pipes + dash separator rows + raw URLs).
function sanitizeForSpeech(text) {
return text
.replace(/```[\s\S]*?```/g, ' ')
.replace(/`([^`]+)`/g, '$1')
.replace(/!\[[^\]]*\]\([^)]*\)/g, ' ')
.replace(/\[([^\]]+)\]\([^)]*\)/g, '$1')
// markdown table separator rows, e.g. "|---|:--:|---|"
.replace(/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?\s*$/gm, ' ')
// remaining table rows: "| a | b |" -> "a, b" so cells read as a list
.replace(/^\s*\|(.+)\|\s*$/gm, (_m, row) => row.split('|').map((c) => c.trim()).filter(Boolean).join(', '))
.replace(/https?:\/\/\S+/g, ' ')
// OmniVoice chokes on a bare "(" — especially Hangul-adjacent, e.g.
// "영웅(Heroic)" — and stops generating audio entirely instead of just
// mispronouncing it. Drop the parens but keep their contents.
.replace(/[()()]/g, ' ')
.replace(/^\s{0,3}#{1,6}\s+/gm, '')
.replace(/^\s*[-*+]\s+/gm, '')
.replace(/^\s*>\s?/gm, '')
.replace(/\*\*([^*]+)\*\*/g, '$1')
.replace(/\*([^*]+)\*/g, '$1')
.replace(/__([^_]+)__/g, '$1')
.replace(/_([^_]+)_/g, '$1')
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}]/gu, '')
.replace(/\s+/g, ' ')
.trim();
}
function enqueueSentence(text) {
text = sanitizeForSpeech((text || '').trim());
if (!text) return;
ttsQueue.push(text);
pumpTtsQueue();
}
// Accumulates sentence fragments until there's enough text to be safe to
// stream to XTTS, then enqueues the merged chunk.
function bufferSentence(sentence) {
pendingTts = pendingTts ? pendingTts + ' ' + sentence : sentence;
if (pendingTts.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_TTS_CHARS) {
enqueueSentence(pendingTts);
pendingTts = '';
}
}
function splitCompleteSentences(text) {
const sentences = [];
let start = 0;
for (let i = 0; i < text.length; i++) {
if (/[.!?\n]/.test(text[i])) {
let j = i + 1;
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
const piece = text.slice(start, j).trim();
if (piece) sentences.push(piece);
start = j;
i = j - 1;
}
}
return { sentences, consumedLength: start };
}
// Hooked from app.js when partialContent is replaced wholesale rather than
// appended to (e.g. a multi-step tool turn's finalAnswer) — spokenUpTo was an
// offset into the old string and must be dropped before the new text arrives.
window._voiceCallResetSpoken = function () {
spokenUpTo = 0;
};
// Hooked from app.js's SSE token handler — only acts while a call is active.
window._voiceCallOnToken = function (fullPartialContent) {
if (!active) return;
const tail = fullPartialContent.slice(spokenUpTo);
const { sentences, consumedLength } = splitCompleteSentences(tail);
if (consumedLength > 0) {
spokenUpTo += consumedLength;
for (const s of sentences) bufferSentence(s);
}
};
// Hooked from app.js when the SSE stream for a turn finishes.
window._voiceCallOnTurnDone = function (finalText) {
if (!active) return;
const remaining = (finalText || '').slice(spokenUpTo).trim();
if (remaining) bufferSentence(remaining);
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
spokenUpTo = 0;
};
})();