Files
homeclaw/web-ui/voice-call.js
T
kimandClaude Sonnet 5 64f7e5ec3b fix: TTS 발음/타이밍 대규모 개선 (숫자, 단위, 방위, 줄바꿈 pause)
- 숫자를 완전한 한글 표기로 변환 (사이노-한국어 기본, 시각은 순우리말)
- °C/°F/%/m/s/mm/km 등 단위를 한글로 스펠아웃
- 괄호 앞뒤 pause 강화, 방위 약어(SSW 등) 한글 변환
- 마크다운 표/리스트를 줄바꿈 전에 분할 후 정제하도록 파이프라인 재설계
- 구조적 개행(리스트/표 셀) 유래 조각은 병합 금지, 대화체 문장만 병합
- 소수점(27.7) 오탐 문장경계 버그 수정
- num_step 16, speed 1.15로 튜닝 (품질 손실 없이 생성속도 개선)
- 대기열 위치 표시 (tts_queue 메시지)

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-07-10 16:39:46 +09:00

497 lines
21 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// ── Voice: 실시간 연속 대화 모드 ───────────────────────────────────────────
// AudioWorklet 캡처(16kHz PCM16) → /ws/voice-realtime → GPU STT/TTS 스트리밍.
// VAD(에너지 기반)로 발화 시작/끝을 로컬에서 판단해 stt_start/stop을 보내고,
// LLM 응답 SSE 토큰에서 완성되는 문장 단위로 잘라 TTS를 파이프라이닝한다.
// barge-in: 재생 중 사용자가 다시 말하면 즉시 재생을 끊고 새 인식을 시작한다.
(function () {
const RMS_SPEECH_ON = 0.020;
const RMS_SPEECH_OFF = 0.012;
const SPEECH_CONFIRM_CHUNKS = 2; // ~200ms of sustained energy to confirm speech start
const SILENCE_HANGOVER_MS = 700; // silence before we consider the utterance finished
const CHUNK_MS = 100;
const PREROLL_CHUNKS = 3; // ~300ms of lead-in kept before speech is confirmed
// XTTS's streaming inference_stream() can crash (CUDA device-side assert,
// poisons the whole GPU context) on very short standalone text — merge short
// sentence fragments together before sending them off as one TTS request.
const MIN_TTS_CHARS = 15;
let ws = null;
let captureCtx = null, playCtx = null;
let micStream = null;
let captureNode = null, playerNode = null, gainNode = null;
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
let speechChunks = 0;
let silenceMs = 0;
let preRoll = [];
let ttsQueue = [];
let ttsBusy = false;
let currentTtsId = null;
let spokenUpTo = 0;
let pendingTts = '';
let active = false;
let wakeLock = null;
// The Wake Lock is auto-released whenever the tab loses visibility (screen off,
// app-switch) — without it, mobile browsers throttle/suspend the background tab
// and the call silently drops. Re-acquire it once the tab is visible again.
async function acquireWakeLock() {
if (!('wakeLock' in navigator)) return;
try {
wakeLock = await navigator.wakeLock.request('screen');
wakeLock.addEventListener('release', () => { wakeLock = null; });
} catch (e) {
console.warn('[voice-call] wakeLock request failed:', e.message);
}
}
document.addEventListener('visibilitychange', () => {
if (active && wakeLock === null && document.visibilityState === 'visible') acquireWakeLock();
});
function setStatus(text) {
const el = document.getElementById('voice-call-status');
if (el) el.textContent = text;
}
function setCaption(text) {
const el = document.getElementById('voice-call-caption');
if (el) el.textContent = text || '';
}
window.toggleVoiceCall = function () {
if (active) voiceCallHangup();
else startVoiceCall();
};
window.voiceCallSetVolume = function (v) {
const vol = parseFloat(v);
if (!Number.isFinite(vol)) return;
localStorage.setItem('voiceCallVolume', String(vol));
if (gainNode) gainNode.gain.value = vol;
};
window.voiceCallHangup = function () {
active = false;
state = 'idle';
try { ws && ws.close(); } catch {}
ws = null;
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
try { captureCtx && captureCtx.close(); } catch {}
try { playCtx && playCtx.close(); } catch {}
captureCtx = playCtx = micStream = captureNode = playerNode = gainNode = null;
try { wakeLock && wakeLock.release(); } catch {}
wakeLock = null;
ttsQueue = []; ttsBusy = false; currentTtsId = null;
const bar = document.getElementById('voice-call-bar');
if (bar) bar.style.display = 'none';
const btn = document.getElementById('voice-call-btn');
if (btn) { btn.textContent = '📞'; btn.classList.remove('recording'); }
};
async function startVoiceCall() {
const btn = document.getElementById('voice-call-btn');
const bar = document.getElementById('voice-call-bar');
try {
micStream = await navigator.mediaDevices.getUserMedia({ audio: { channelCount: 1, echoCancellation: true, noiseSuppression: true } });
} catch (e) {
alert('마이크 권한이 필요합니다: ' + e.message);
return;
}
active = true;
acquireWakeLock();
if (btn) { btn.textContent = '📵'; btn.classList.add('recording'); }
if (bar) bar.style.display = 'flex';
setStatus('통화 연결 중…');
setCaption('');
state = 'listening';
speechChunks = 0; silenceMs = 0; preRoll = [];
spokenUpTo = 0; pendingTts = ''; ttsQueue = []; ttsBusy = false; currentTtsId = null;
captureCtx = new (window.AudioContext || window.webkitAudioContext)();
await captureCtx.audioWorklet.addModule('voice-call-worklets.js');
const source = captureCtx.createMediaStreamSource(micStream);
captureNode = new AudioWorkletNode(captureCtx, 'mic-capture-processor');
captureNode.port.onmessage = handleMicChunk;
source.connect(captureNode);
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
// A GainNode we control from the in-app slider — on some mobile browsers
// the hardware volume rocker maps to the call/mic audio session while a
// getUserMedia stream is open, not to this Web Audio output, so the phone's
// own volume slider doesn't reliably affect playback here.
gainNode = playCtx.createGain();
const savedVolume = parseFloat(localStorage.getItem('voiceCallVolume') || '1');
gainNode.gain.value = savedVolume;
const volumeSlider = document.getElementById('voice-call-volume');
if (volumeSlider) volumeSlider.value = String(savedVolume);
playerNode.connect(gainNode).connect(playCtx.destination);
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
const token = getAuthToken();
const qs = token ? `?token=${encodeURIComponent(token)}` : '';
ws = new WebSocket(`${proto}://${API.replace(/^https?:\/\//, '')}/ws/voice-realtime${qs}`);
ws.binaryType = 'arraybuffer';
ws.onopen = () => setStatus('듣는 중…');
ws.onmessage = handleWsMessage;
ws.onerror = () => setStatus('연결 오류');
ws.onclose = () => { if (active) voiceCallHangup(); };
}
function handleMicChunk(e) {
if (!active) return;
const { pcm, rms } = e.data;
if (state === 'listening' || state === 'assistant_speaking') {
preRoll.push(pcm);
if (preRoll.length > PREROLL_CHUNKS) preRoll.shift();
if (rms > RMS_SPEECH_ON) {
speechChunks++;
if (speechChunks >= SPEECH_CONFIRM_CHUNKS) onSpeechStart();
} else {
speechChunks = 0;
}
} else if (state === 'user_speaking') {
if (ws && ws.readyState === WebSocket.OPEN) ws.send(pcm.buffer);
if (rms < RMS_SPEECH_OFF) {
silenceMs += CHUNK_MS;
if (silenceMs >= SILENCE_HANGOVER_MS) onSpeechEnd();
} else {
silenceMs = 0;
}
}
}
function onSpeechStart() {
if (state === 'assistant_speaking') {
// barge-in: cut playback immediately and cancel whatever is still generating
ttsQueue = [];
if (playerNode) playerNode.port.postMessage({ type: 'clear' });
if (ws && ws.readyState === WebSocket.OPEN && currentTtsId) {
ws.send(JSON.stringify({ type: 'tts_cancel', id: currentTtsId }));
}
ttsBusy = false;
currentTtsId = null;
}
state = 'user_speaking';
silenceMs = 0;
setStatus('듣는 중…');
if (ws && ws.readyState === WebSocket.OPEN) {
ws.send(JSON.stringify({ type: 'stt_start', language: 'ko' }));
for (const chunk of preRoll) ws.send(chunk.buffer);
}
preRoll = [];
}
function onSpeechEnd() {
state = 'processing';
speechChunks = 0;
setStatus('생각 중…');
if (ws && ws.readyState === WebSocket.OPEN) {
ws.send(JSON.stringify({ type: 'stt_stop', id: 'turn-' + Date.now() }));
}
}
function handleWsMessage(e) {
if (e.data instanceof ArrayBuffer) {
if (playerNode) playerNode.port.postMessage({ type: 'push', pcm: new Int16Array(e.data) }, [e.data]);
return;
}
let msg;
try { msg = JSON.parse(e.data); } catch { return; }
switch (msg.type) {
case 'stt_partial':
setCaption(msg.text || '');
break;
case 'stt_final': {
setCaption('');
const text = (msg.text || '').trim();
if (!text) { state = 'listening'; setStatus('듣는 중…'); break; }
spokenUpTo = 0; pendingTts = '';
const input = document.getElementById('chat-input');
if (input) {
input.value = text;
input.dispatchEvent(new Event('input'));
}
setStatus('생각 중…');
setTimeout(() => (window._appSendFn || handleSendStop)(), 30);
break;
}
case 'tts_queue':
if (msg.aheadCount > 0) setStatus(`대기 중… (앞에 ${msg.aheadCount}명)`);
break;
case 'tts_stream_start':
currentTtsId = msg.id;
state = 'assistant_speaking';
setStatus('말하는 중…');
break;
case 'tts_end':
ttsBusy = false;
currentTtsId = null;
pumpTtsQueue();
if (ttsQueue.length === 0) {
state = 'listening';
setStatus('듣는 중…');
}
break;
case 'error':
console.warn('[voice-call] engine error:', msg.message);
// The failed request may have been the one holding ttsBusy — without
// resetting it here, pumpTtsQueue() early-returns forever and the
// call goes silent for the rest of the session.
if (!msg.id || msg.id === currentTtsId) {
ttsBusy = false;
currentTtsId = null;
pumpTtsQueue();
if (ttsQueue.length === 0) {
state = 'listening';
setStatus('듣는 중…');
}
}
break;
}
}
function pumpTtsQueue() {
if (ttsBusy || ttsQueue.length === 0) return;
if (!ws || ws.readyState !== WebSocket.OPEN) return;
const text = ttsQueue.shift();
ttsBusy = true;
const id = 'tts-' + Date.now() + '-' + Math.random().toString(36).slice(2, 7);
currentTtsId = id;
ws.send(JSON.stringify({ type: 'tts_start', text, id }));
}
// Strips markdown syntax, tables, URLs and emoji so raw LLM output
// (**bold**, bullet markers, code blocks, | table | pipes |, links, 🦞
// etc.) doesn't get force-pronounced by XTTS under language='ko' — those
// symbols were producing garbled non-Korean-sounding artifacts mid-sentence
// (tables were the worst offender: pipes + dash separator rows + raw URLs).
// Sino-Korean digit-string reading, e.g. 27 -> "이십칠", 0 -> "영". Handles
// 0-9999 (plenty for measurement-style numbers — temperatures, percentages,
// speeds); larger numbers get an extra 만/억 grouping pass but aren't the
// target use case. Mirrors app.js's _sinoKoreanInt.
function _sinoKoreanInt(n) {
if (n === 0) return '영';
const digits = ['', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
const smallUnits = ['', '십', '백', '천'];
const bigUnits = ['', '만', '억', '조'];
function fourDigit(num) {
if (num === 0) return '';
let s = '';
const ds = String(num).padStart(4, '0').split('').map(Number);
for (let i = 0; i < 4; i++) {
const d = ds[i], unit = smallUnits[3 - i];
if (d === 0) continue;
s += (d === 1 && unit !== '') ? unit : (digits[d] + unit);
}
return s;
}
const groups = [];
let rem = n;
while (rem > 0) { groups.push(rem % 10000); rem = Math.floor(rem / 10000); }
let result = '';
for (let g = groups.length - 1; g >= 0; g--) {
if (groups[g] === 0) continue;
result += fourDigit(groups[g]) + bigUnits[g];
}
return result || '영';
}
// "27.7" -> "이십칠점칠" (decimal digits read one at a time, standard
// Korean convention), "-3.2" -> "마이너스 삼점이".
function _numToHangulWord(numStr) {
const neg = numStr[0] === '-';
if (neg) numStr = numStr.slice(1);
const parts = numStr.split('.');
let word = _sinoKoreanInt(parseInt(parts[0], 10) || 0);
if (parts[1]) {
const d = ['영', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
word += '점' + parts[1].split('').map((c) => d[+c]).join('');
}
return (neg ? '마이너스 ' : '') + word;
}
// Korean uses native-Korean numerals (하나/둘/셋...), not Sino-Korean
// (일/이/삼...), specifically for clock hours and hour-durations — "9시"
// is "아홉 시", never "구시" (reported 2026-07-10). Everything else (분,
// 초, 도, %, dates, etc.) correctly stays Sino-Korean via _numToHangulWord.
const _NATIVE_HOUR = { 1: '한', 2: '두', 3: '세', 4: '네', 5: '다섯', 6: '여섯', 7: '일곱', 8: '여덟', 9: '아홉', 10: '열', 11: '열한', 12: '열두' };
function _hourReplacer(m, numStr, suffix) {
const n = parseInt(numStr, 10);
return _NATIVE_HOUR[n] ? _NATIVE_HOUR[n] + suffix : m;
}
function sanitizeForSpeech(text) {
return text
.replace(/```[\s\S]*?```/g, ' ')
.replace(/`([^`]+)`/g, '$1')
.replace(/!\[[^\]]*\]\([^)]*\)/g, ' ')
.replace(/\[([^\]]+)\]\([^)]*\)/g, '$1')
// markdown table separator rows, e.g. "|---|:--:|---|"
.replace(/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?\s*$/gm, ' ')
// remaining table rows: "| a | b |" -> "a, b" so cells read as a list
.replace(/^\s*\|(.+)\|\s*$/gm, (_m, row) => row.split('|').map((c) => c.trim()).filter(Boolean).join(', '))
.replace(/https?:\/\/\S+/g, ' ')
// OmniVoice reads a bare "°C"/"°F" as "그램"(grams) instead of degrees
// — the digits (even with a decimal point) come through fine once the
// symbol itself is spelled out in Korean (verified via STT round-trip,
// 2026-07-10). Order matters: °C/°F before the bare ° fallback.
.replace(/(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도')
.replace(/(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도')
.replace(/(\d+(?:\.\d+)?)\s*°/g, '$1도')
// Spell out common units so a bare symbol doesn't get read wrong
// (measured: numerals+symbols like "%"/"m/s" have a real error rate,
// spelling everything in Hangul is dramatically more reliable, 2026-07-10).
.replace(/(\d+(?:\.\d+)?)\s*%/g, '$1퍼센트')
.replace(/(\d+(?:\.\d+)?)\s*km\/h/gi, '$1킬로미터 매 시')
.replace(/(\d+(?:\.\d+)?)\s*m\/s/gi, '$1미터 매 초')
.replace(/(\d+(?:\.\d+)?)\s*mm/gi, '$1밀리미터')
.replace(/(\d+(?:\.\d+)?)\s*km/gi, '$1킬로미터')
// Wind-direction abbreviations in parens (e.g. "2.6 m/s (SSW)") aren't
// Korean words, so OmniVoice mangles them — spell out the compass point
// instead. Longest abbreviations first so "SSW" doesn't partial-match as
// "S" (reported 2026-07-10). Must run before the generic paren->period
// conversion below, while the parens are still intact to anchor on.
.replace(/\((NNE|ENE|ESE|SSE|SSW|WSW|WNW|NNW|NE|SE|SW|NW|N|S|E|W)\)/g, (_m, dir) => {
const d = { N: '북', NNE: '북북동', NE: '북동', ENE: '동북동', E: '동', ESE: '동남동', SE: '남동', SSE: '남남동', S: '남', SSW: '남남서', SW: '남서', WSW: '서남서', W: '서', WNW: '서북서', NW: '북서', NNW: '북북서' };
return d[dir] ? ' ' + d[dir] + ' ' : _m;
})
// OmniVoice chokes on a bare "(" — especially Hangul-adjacent, e.g.
// "영웅(Heroic)" — and stops generating audio entirely instead of just
// mispronouncing it. Swap for a period instead of just erasing: still
// no literal "(" reaches the model, but the parenthetical aside gets a
// clearer pause than a comma did (measured: comma ~+0.02-0.08s over a
// bare space, period ~2-3x that, 2026-07-10).
.replace(/[((]\s*/g, '. ')
.replace(/\s*[))]\s*/g, '. ')
.replace(/\.\s*\./g, '.')
.replace(/^\s{0,3}#{1,6}\s+/gm, '')
.replace(/^\s*[-*+]\s+/gm, '')
.replace(/^\s*>\s?/gm, '')
.replace(/\*\*([^*]+)\*\*/g, '$1')
.replace(/\*([^*]+)\*/g, '$1')
.replace(/__([^_]+)__/g, '$1')
.replace(/_([^_]+)_/g, '$1')
// Paragraph/line breaks were silently vanishing into the final \s+ ->
// ' ' collapse below, so a blank line between two unrelated facts (e.g.
// wind speed, then a separate rain/comfort sentence) got read with zero
// pause at all — worse than the old paren-as-space bug. Must run after
// all the ^...$/gm passes above (they still need real newlines).
.replace(/(?<![.!?,:;])\n{2,}/g, '. ')
.replace(/(?<![.!?,:;])\n/g, ', ')
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}\u{FE0F}]/gu, '')
// Hour numbers use native-Korean counting, not Sino-Korean — must run
// before the generic number pass below so "9시" doesn't become "구시".
.replace(/(\d{1,2})(\s*시간?)/g, _hourReplacer)
// Final pass: every remaining numeral becomes spelled-out Hangul — the
// single biggest reliability win measured today (raw numerals: frequent
// digit swaps/drops; spelled-out: ~7/8 clean in repeated STT round-trip
// testing, 2026-07-10).
.replace(/-?\d+(?:\.\d+)?/g, (m) => _numToHangulWord(m))
.replace(/\s+/g, ' ')
.trim();
}
function enqueueSentence(text) {
text = sanitizeForSpeech((text || '').trim());
if (!text) return;
ttsQueue.push(text);
pumpTtsQueue();
}
function flushPendingTts() {
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
}
// Accumulates sentence fragments until there's enough text to be safe to
// stream to XTTS, then enqueues the merged chunk. A line break (bullet
// item, paragraph line) is an intentional structural separation, though —
// never merge it with a neighbor, same reasoning as app.js's text-button
// fix. Without this, a short line like "습도: 86%" (under MIN_TTS_CHARS)
// silently absorbed the next bullet line and the pause between them
// vanished (reported 2026-07-10).
function bufferSentence(sentence, isLineBreak) {
if (isLineBreak) {
flushPendingTts();
enqueueSentence(sentence);
return;
}
pendingTts = pendingTts ? pendingTts + ' ' + sentence : sentence;
if (pendingTts.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_TTS_CHARS) {
flushPendingTts();
}
}
function splitCompleteSentences(text) {
const sentences = [];
let start = 0;
for (let i = 0; i < text.length; i++) {
// A "." between two digits is a decimal point (e.g. "27.7"), not a
// sentence end — without this guard, streamed numbers get sliced in
// half mid-decimal into two separate TTS calls (found 2026-07-10 while
// debugging why weather readouts sounded choppy).
if (text[i] === '.' && /\d/.test(text[i - 1] || '') && /\d/.test(text[i + 1] || '')) continue;
if (/[.!?\n]/.test(text[i])) {
const boundaryChar = text[i];
let j = i + 1;
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
let piece = text.slice(start, j).trim();
// bufferSentence() rejoins pieces with a plain space, not a real
// "\n" — so by the time sanitizeForSpeech runs, its "^...$/gm"
// bullet-marker regex only sees ONE line (the whole joined string)
// and misses every bullet after the first. Strip it per-piece here,
// while a real line boundary still exists to anchor on (2026-07-10).
piece = piece.replace(/^[-*+]\s+/, '');
// A "\n" boundary carries no punctuation of its own, so the .trim()
// above silently erases the pause it implied — sanitizeForSpeech's
// \n handling never gets a chance to see it since the newline is
// gone by the time this piece reaches it. Put an explicit period
// back so list items / paragraph breaks still read as a pause
// instead of running straight into the next line (2026-07-10).
if (piece && boundaryChar === '\n' && !/[.!?,:;]$/.test(piece)) {
piece += '.';
}
if (piece) sentences.push({ text: piece, isLineBreak: boundaryChar === '\n' });
start = j;
i = j - 1;
}
}
return { sentences, consumedLength: start };
}
// Hooked from app.js when partialContent is replaced wholesale rather than
// appended to (e.g. a multi-step tool turn's finalAnswer) — spokenUpTo was an
// offset into the old string and must be dropped before the new text arrives.
window._voiceCallResetSpoken = function () {
spokenUpTo = 0;
};
// Hooked from app.js's SSE token handler — only acts while a call is active.
window._voiceCallOnToken = function (fullPartialContent) {
if (!active) return;
const tail = fullPartialContent.slice(spokenUpTo);
const { sentences, consumedLength } = splitCompleteSentences(tail);
if (consumedLength > 0) {
spokenUpTo += consumedLength;
for (const s of sentences) bufferSentence(s.text, s.isLineBreak);
}
};
// Hooked from app.js when the SSE stream for a turn finishes.
window._voiceCallOnTurnDone = function (finalText) {
if (!active) return;
const remaining = (finalText || '').slice(spokenUpTo).trim();
if (remaining) bufferSentence(remaining);
flushPendingTts();
spokenUpTo = 0;
};
})();