- 숫자를 완전한 한글 표기로 변환 (사이노-한국어 기본, 시각은 순우리말) - °C/°F/%/m/s/mm/km 등 단위를 한글로 스펠아웃 - 괄호 앞뒤 pause 강화, 방위 약어(SSW 등) 한글 변환 - 마크다운 표/리스트를 줄바꿈 전에 분할 후 정제하도록 파이프라인 재설계 - 구조적 개행(리스트/표 셀) 유래 조각은 병합 금지, 대화체 문장만 병합 - 소수점(27.7) 오탐 문장경계 버그 수정 - num_step 16, speed 1.15로 튜닝 (품질 손실 없이 생성속도 개선) - 대기열 위치 표시 (tts_queue 메시지) Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
497 lines
21 KiB
JavaScript
497 lines
21 KiB
JavaScript
// ── Voice: 실시간 연속 대화 모드 ───────────────────────────────────────────
|
||
// AudioWorklet 캡처(16kHz PCM16) → /ws/voice-realtime → GPU STT/TTS 스트리밍.
|
||
// VAD(에너지 기반)로 발화 시작/끝을 로컬에서 판단해 stt_start/stop을 보내고,
|
||
// LLM 응답 SSE 토큰에서 완성되는 문장 단위로 잘라 TTS를 파이프라이닝한다.
|
||
// barge-in: 재생 중 사용자가 다시 말하면 즉시 재생을 끊고 새 인식을 시작한다.
|
||
|
||
(function () {
|
||
const RMS_SPEECH_ON = 0.020;
|
||
const RMS_SPEECH_OFF = 0.012;
|
||
const SPEECH_CONFIRM_CHUNKS = 2; // ~200ms of sustained energy to confirm speech start
|
||
const SILENCE_HANGOVER_MS = 700; // silence before we consider the utterance finished
|
||
const CHUNK_MS = 100;
|
||
const PREROLL_CHUNKS = 3; // ~300ms of lead-in kept before speech is confirmed
|
||
// XTTS's streaming inference_stream() can crash (CUDA device-side assert,
|
||
// poisons the whole GPU context) on very short standalone text — merge short
|
||
// sentence fragments together before sending them off as one TTS request.
|
||
const MIN_TTS_CHARS = 15;
|
||
|
||
let ws = null;
|
||
let captureCtx = null, playCtx = null;
|
||
let micStream = null;
|
||
let captureNode = null, playerNode = null, gainNode = null;
|
||
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
|
||
let speechChunks = 0;
|
||
let silenceMs = 0;
|
||
let preRoll = [];
|
||
let ttsQueue = [];
|
||
let ttsBusy = false;
|
||
let currentTtsId = null;
|
||
let spokenUpTo = 0;
|
||
let pendingTts = '';
|
||
let active = false;
|
||
let wakeLock = null;
|
||
|
||
// The Wake Lock is auto-released whenever the tab loses visibility (screen off,
|
||
// app-switch) — without it, mobile browsers throttle/suspend the background tab
|
||
// and the call silently drops. Re-acquire it once the tab is visible again.
|
||
async function acquireWakeLock() {
|
||
if (!('wakeLock' in navigator)) return;
|
||
try {
|
||
wakeLock = await navigator.wakeLock.request('screen');
|
||
wakeLock.addEventListener('release', () => { wakeLock = null; });
|
||
} catch (e) {
|
||
console.warn('[voice-call] wakeLock request failed:', e.message);
|
||
}
|
||
}
|
||
document.addEventListener('visibilitychange', () => {
|
||
if (active && wakeLock === null && document.visibilityState === 'visible') acquireWakeLock();
|
||
});
|
||
|
||
function setStatus(text) {
|
||
const el = document.getElementById('voice-call-status');
|
||
if (el) el.textContent = text;
|
||
}
|
||
function setCaption(text) {
|
||
const el = document.getElementById('voice-call-caption');
|
||
if (el) el.textContent = text || '';
|
||
}
|
||
|
||
window.toggleVoiceCall = function () {
|
||
if (active) voiceCallHangup();
|
||
else startVoiceCall();
|
||
};
|
||
|
||
window.voiceCallSetVolume = function (v) {
|
||
const vol = parseFloat(v);
|
||
if (!Number.isFinite(vol)) return;
|
||
localStorage.setItem('voiceCallVolume', String(vol));
|
||
if (gainNode) gainNode.gain.value = vol;
|
||
};
|
||
|
||
window.voiceCallHangup = function () {
|
||
active = false;
|
||
state = 'idle';
|
||
try { ws && ws.close(); } catch {}
|
||
ws = null;
|
||
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
|
||
try { captureCtx && captureCtx.close(); } catch {}
|
||
try { playCtx && playCtx.close(); } catch {}
|
||
captureCtx = playCtx = micStream = captureNode = playerNode = gainNode = null;
|
||
try { wakeLock && wakeLock.release(); } catch {}
|
||
wakeLock = null;
|
||
ttsQueue = []; ttsBusy = false; currentTtsId = null;
|
||
const bar = document.getElementById('voice-call-bar');
|
||
if (bar) bar.style.display = 'none';
|
||
const btn = document.getElementById('voice-call-btn');
|
||
if (btn) { btn.textContent = '📞'; btn.classList.remove('recording'); }
|
||
};
|
||
|
||
async function startVoiceCall() {
|
||
const btn = document.getElementById('voice-call-btn');
|
||
const bar = document.getElementById('voice-call-bar');
|
||
try {
|
||
micStream = await navigator.mediaDevices.getUserMedia({ audio: { channelCount: 1, echoCancellation: true, noiseSuppression: true } });
|
||
} catch (e) {
|
||
alert('마이크 권한이 필요합니다: ' + e.message);
|
||
return;
|
||
}
|
||
|
||
active = true;
|
||
acquireWakeLock();
|
||
if (btn) { btn.textContent = '📵'; btn.classList.add('recording'); }
|
||
if (bar) bar.style.display = 'flex';
|
||
setStatus('통화 연결 중…');
|
||
setCaption('');
|
||
state = 'listening';
|
||
speechChunks = 0; silenceMs = 0; preRoll = [];
|
||
spokenUpTo = 0; pendingTts = ''; ttsQueue = []; ttsBusy = false; currentTtsId = null;
|
||
|
||
captureCtx = new (window.AudioContext || window.webkitAudioContext)();
|
||
await captureCtx.audioWorklet.addModule('voice-call-worklets.js');
|
||
const source = captureCtx.createMediaStreamSource(micStream);
|
||
captureNode = new AudioWorkletNode(captureCtx, 'mic-capture-processor');
|
||
captureNode.port.onmessage = handleMicChunk;
|
||
source.connect(captureNode);
|
||
|
||
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
|
||
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
|
||
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
|
||
// A GainNode we control from the in-app slider — on some mobile browsers
|
||
// the hardware volume rocker maps to the call/mic audio session while a
|
||
// getUserMedia stream is open, not to this Web Audio output, so the phone's
|
||
// own volume slider doesn't reliably affect playback here.
|
||
gainNode = playCtx.createGain();
|
||
const savedVolume = parseFloat(localStorage.getItem('voiceCallVolume') || '1');
|
||
gainNode.gain.value = savedVolume;
|
||
const volumeSlider = document.getElementById('voice-call-volume');
|
||
if (volumeSlider) volumeSlider.value = String(savedVolume);
|
||
playerNode.connect(gainNode).connect(playCtx.destination);
|
||
|
||
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
|
||
const token = getAuthToken();
|
||
const qs = token ? `?token=${encodeURIComponent(token)}` : '';
|
||
ws = new WebSocket(`${proto}://${API.replace(/^https?:\/\//, '')}/ws/voice-realtime${qs}`);
|
||
ws.binaryType = 'arraybuffer';
|
||
ws.onopen = () => setStatus('듣는 중…');
|
||
ws.onmessage = handleWsMessage;
|
||
ws.onerror = () => setStatus('연결 오류');
|
||
ws.onclose = () => { if (active) voiceCallHangup(); };
|
||
}
|
||
|
||
function handleMicChunk(e) {
|
||
if (!active) return;
|
||
const { pcm, rms } = e.data;
|
||
|
||
if (state === 'listening' || state === 'assistant_speaking') {
|
||
preRoll.push(pcm);
|
||
if (preRoll.length > PREROLL_CHUNKS) preRoll.shift();
|
||
if (rms > RMS_SPEECH_ON) {
|
||
speechChunks++;
|
||
if (speechChunks >= SPEECH_CONFIRM_CHUNKS) onSpeechStart();
|
||
} else {
|
||
speechChunks = 0;
|
||
}
|
||
} else if (state === 'user_speaking') {
|
||
if (ws && ws.readyState === WebSocket.OPEN) ws.send(pcm.buffer);
|
||
if (rms < RMS_SPEECH_OFF) {
|
||
silenceMs += CHUNK_MS;
|
||
if (silenceMs >= SILENCE_HANGOVER_MS) onSpeechEnd();
|
||
} else {
|
||
silenceMs = 0;
|
||
}
|
||
}
|
||
}
|
||
|
||
function onSpeechStart() {
|
||
if (state === 'assistant_speaking') {
|
||
// barge-in: cut playback immediately and cancel whatever is still generating
|
||
ttsQueue = [];
|
||
if (playerNode) playerNode.port.postMessage({ type: 'clear' });
|
||
if (ws && ws.readyState === WebSocket.OPEN && currentTtsId) {
|
||
ws.send(JSON.stringify({ type: 'tts_cancel', id: currentTtsId }));
|
||
}
|
||
ttsBusy = false;
|
||
currentTtsId = null;
|
||
}
|
||
state = 'user_speaking';
|
||
silenceMs = 0;
|
||
setStatus('듣는 중…');
|
||
if (ws && ws.readyState === WebSocket.OPEN) {
|
||
ws.send(JSON.stringify({ type: 'stt_start', language: 'ko' }));
|
||
for (const chunk of preRoll) ws.send(chunk.buffer);
|
||
}
|
||
preRoll = [];
|
||
}
|
||
|
||
function onSpeechEnd() {
|
||
state = 'processing';
|
||
speechChunks = 0;
|
||
setStatus('생각 중…');
|
||
if (ws && ws.readyState === WebSocket.OPEN) {
|
||
ws.send(JSON.stringify({ type: 'stt_stop', id: 'turn-' + Date.now() }));
|
||
}
|
||
}
|
||
|
||
function handleWsMessage(e) {
|
||
if (e.data instanceof ArrayBuffer) {
|
||
if (playerNode) playerNode.port.postMessage({ type: 'push', pcm: new Int16Array(e.data) }, [e.data]);
|
||
return;
|
||
}
|
||
let msg;
|
||
try { msg = JSON.parse(e.data); } catch { return; }
|
||
|
||
switch (msg.type) {
|
||
case 'stt_partial':
|
||
setCaption(msg.text || '');
|
||
break;
|
||
|
||
case 'stt_final': {
|
||
setCaption('');
|
||
const text = (msg.text || '').trim();
|
||
if (!text) { state = 'listening'; setStatus('듣는 중…'); break; }
|
||
spokenUpTo = 0; pendingTts = '';
|
||
const input = document.getElementById('chat-input');
|
||
if (input) {
|
||
input.value = text;
|
||
input.dispatchEvent(new Event('input'));
|
||
}
|
||
setStatus('생각 중…');
|
||
setTimeout(() => (window._appSendFn || handleSendStop)(), 30);
|
||
break;
|
||
}
|
||
|
||
case 'tts_queue':
|
||
if (msg.aheadCount > 0) setStatus(`대기 중… (앞에 ${msg.aheadCount}명)`);
|
||
break;
|
||
|
||
case 'tts_stream_start':
|
||
currentTtsId = msg.id;
|
||
state = 'assistant_speaking';
|
||
setStatus('말하는 중…');
|
||
break;
|
||
|
||
case 'tts_end':
|
||
ttsBusy = false;
|
||
currentTtsId = null;
|
||
pumpTtsQueue();
|
||
if (ttsQueue.length === 0) {
|
||
state = 'listening';
|
||
setStatus('듣는 중…');
|
||
}
|
||
break;
|
||
|
||
case 'error':
|
||
console.warn('[voice-call] engine error:', msg.message);
|
||
// The failed request may have been the one holding ttsBusy — without
|
||
// resetting it here, pumpTtsQueue() early-returns forever and the
|
||
// call goes silent for the rest of the session.
|
||
if (!msg.id || msg.id === currentTtsId) {
|
||
ttsBusy = false;
|
||
currentTtsId = null;
|
||
pumpTtsQueue();
|
||
if (ttsQueue.length === 0) {
|
||
state = 'listening';
|
||
setStatus('듣는 중…');
|
||
}
|
||
}
|
||
break;
|
||
}
|
||
}
|
||
|
||
function pumpTtsQueue() {
|
||
if (ttsBusy || ttsQueue.length === 0) return;
|
||
if (!ws || ws.readyState !== WebSocket.OPEN) return;
|
||
const text = ttsQueue.shift();
|
||
ttsBusy = true;
|
||
const id = 'tts-' + Date.now() + '-' + Math.random().toString(36).slice(2, 7);
|
||
currentTtsId = id;
|
||
ws.send(JSON.stringify({ type: 'tts_start', text, id }));
|
||
}
|
||
|
||
// Strips markdown syntax, tables, URLs and emoji so raw LLM output
|
||
// (**bold**, bullet markers, code blocks, | table | pipes |, links, 🦞
|
||
// etc.) doesn't get force-pronounced by XTTS under language='ko' — those
|
||
// symbols were producing garbled non-Korean-sounding artifacts mid-sentence
|
||
// (tables were the worst offender: pipes + dash separator rows + raw URLs).
|
||
// Sino-Korean digit-string reading, e.g. 27 -> "이십칠", 0 -> "영". Handles
|
||
// 0-9999 (plenty for measurement-style numbers — temperatures, percentages,
|
||
// speeds); larger numbers get an extra 만/억 grouping pass but aren't the
|
||
// target use case. Mirrors app.js's _sinoKoreanInt.
|
||
function _sinoKoreanInt(n) {
|
||
if (n === 0) return '영';
|
||
const digits = ['', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
|
||
const smallUnits = ['', '십', '백', '천'];
|
||
const bigUnits = ['', '만', '억', '조'];
|
||
function fourDigit(num) {
|
||
if (num === 0) return '';
|
||
let s = '';
|
||
const ds = String(num).padStart(4, '0').split('').map(Number);
|
||
for (let i = 0; i < 4; i++) {
|
||
const d = ds[i], unit = smallUnits[3 - i];
|
||
if (d === 0) continue;
|
||
s += (d === 1 && unit !== '') ? unit : (digits[d] + unit);
|
||
}
|
||
return s;
|
||
}
|
||
const groups = [];
|
||
let rem = n;
|
||
while (rem > 0) { groups.push(rem % 10000); rem = Math.floor(rem / 10000); }
|
||
let result = '';
|
||
for (let g = groups.length - 1; g >= 0; g--) {
|
||
if (groups[g] === 0) continue;
|
||
result += fourDigit(groups[g]) + bigUnits[g];
|
||
}
|
||
return result || '영';
|
||
}
|
||
|
||
// "27.7" -> "이십칠점칠" (decimal digits read one at a time, standard
|
||
// Korean convention), "-3.2" -> "마이너스 삼점이".
|
||
function _numToHangulWord(numStr) {
|
||
const neg = numStr[0] === '-';
|
||
if (neg) numStr = numStr.slice(1);
|
||
const parts = numStr.split('.');
|
||
let word = _sinoKoreanInt(parseInt(parts[0], 10) || 0);
|
||
if (parts[1]) {
|
||
const d = ['영', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
|
||
word += '점' + parts[1].split('').map((c) => d[+c]).join('');
|
||
}
|
||
return (neg ? '마이너스 ' : '') + word;
|
||
}
|
||
|
||
// Korean uses native-Korean numerals (하나/둘/셋...), not Sino-Korean
|
||
// (일/이/삼...), specifically for clock hours and hour-durations — "9시"
|
||
// is "아홉 시", never "구시" (reported 2026-07-10). Everything else (분,
|
||
// 초, 도, %, dates, etc.) correctly stays Sino-Korean via _numToHangulWord.
|
||
const _NATIVE_HOUR = { 1: '한', 2: '두', 3: '세', 4: '네', 5: '다섯', 6: '여섯', 7: '일곱', 8: '여덟', 9: '아홉', 10: '열', 11: '열한', 12: '열두' };
|
||
function _hourReplacer(m, numStr, suffix) {
|
||
const n = parseInt(numStr, 10);
|
||
return _NATIVE_HOUR[n] ? _NATIVE_HOUR[n] + suffix : m;
|
||
}
|
||
|
||
function sanitizeForSpeech(text) {
|
||
return text
|
||
.replace(/```[\s\S]*?```/g, ' ')
|
||
.replace(/`([^`]+)`/g, '$1')
|
||
.replace(/!\[[^\]]*\]\([^)]*\)/g, ' ')
|
||
.replace(/\[([^\]]+)\]\([^)]*\)/g, '$1')
|
||
// markdown table separator rows, e.g. "|---|:--:|---|"
|
||
.replace(/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?\s*$/gm, ' ')
|
||
// remaining table rows: "| a | b |" -> "a, b" so cells read as a list
|
||
.replace(/^\s*\|(.+)\|\s*$/gm, (_m, row) => row.split('|').map((c) => c.trim()).filter(Boolean).join(', '))
|
||
.replace(/https?:\/\/\S+/g, ' ')
|
||
// OmniVoice reads a bare "°C"/"°F" as "그램"(grams) instead of degrees
|
||
// — the digits (even with a decimal point) come through fine once the
|
||
// symbol itself is spelled out in Korean (verified via STT round-trip,
|
||
// 2026-07-10). Order matters: °C/°F before the bare ° fallback.
|
||
.replace(/(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도')
|
||
.replace(/(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도')
|
||
.replace(/(\d+(?:\.\d+)?)\s*°/g, '$1도')
|
||
// Spell out common units so a bare symbol doesn't get read wrong
|
||
// (measured: numerals+symbols like "%"/"m/s" have a real error rate,
|
||
// spelling everything in Hangul is dramatically more reliable, 2026-07-10).
|
||
.replace(/(\d+(?:\.\d+)?)\s*%/g, '$1퍼센트')
|
||
.replace(/(\d+(?:\.\d+)?)\s*km\/h/gi, '$1킬로미터 매 시')
|
||
.replace(/(\d+(?:\.\d+)?)\s*m\/s/gi, '$1미터 매 초')
|
||
.replace(/(\d+(?:\.\d+)?)\s*mm/gi, '$1밀리미터')
|
||
.replace(/(\d+(?:\.\d+)?)\s*km/gi, '$1킬로미터')
|
||
// Wind-direction abbreviations in parens (e.g. "2.6 m/s (SSW)") aren't
|
||
// Korean words, so OmniVoice mangles them — spell out the compass point
|
||
// instead. Longest abbreviations first so "SSW" doesn't partial-match as
|
||
// "S" (reported 2026-07-10). Must run before the generic paren->period
|
||
// conversion below, while the parens are still intact to anchor on.
|
||
.replace(/\((NNE|ENE|ESE|SSE|SSW|WSW|WNW|NNW|NE|SE|SW|NW|N|S|E|W)\)/g, (_m, dir) => {
|
||
const d = { N: '북', NNE: '북북동', NE: '북동', ENE: '동북동', E: '동', ESE: '동남동', SE: '남동', SSE: '남남동', S: '남', SSW: '남남서', SW: '남서', WSW: '서남서', W: '서', WNW: '서북서', NW: '북서', NNW: '북북서' };
|
||
return d[dir] ? ' ' + d[dir] + ' ' : _m;
|
||
})
|
||
// OmniVoice chokes on a bare "(" — especially Hangul-adjacent, e.g.
|
||
// "영웅(Heroic)" — and stops generating audio entirely instead of just
|
||
// mispronouncing it. Swap for a period instead of just erasing: still
|
||
// no literal "(" reaches the model, but the parenthetical aside gets a
|
||
// clearer pause than a comma did (measured: comma ~+0.02-0.08s over a
|
||
// bare space, period ~2-3x that, 2026-07-10).
|
||
.replace(/[((]\s*/g, '. ')
|
||
.replace(/\s*[))]\s*/g, '. ')
|
||
.replace(/\.\s*\./g, '.')
|
||
.replace(/^\s{0,3}#{1,6}\s+/gm, '')
|
||
.replace(/^\s*[-*+]\s+/gm, '')
|
||
.replace(/^\s*>\s?/gm, '')
|
||
.replace(/\*\*([^*]+)\*\*/g, '$1')
|
||
.replace(/\*([^*]+)\*/g, '$1')
|
||
.replace(/__([^_]+)__/g, '$1')
|
||
.replace(/_([^_]+)_/g, '$1')
|
||
// Paragraph/line breaks were silently vanishing into the final \s+ ->
|
||
// ' ' collapse below, so a blank line between two unrelated facts (e.g.
|
||
// wind speed, then a separate rain/comfort sentence) got read with zero
|
||
// pause at all — worse than the old paren-as-space bug. Must run after
|
||
// all the ^...$/gm passes above (they still need real newlines).
|
||
.replace(/(?<![.!?,:;])\n{2,}/g, '. ')
|
||
.replace(/(?<![.!?,:;])\n/g, ', ')
|
||
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}\u{FE0F}]/gu, '')
|
||
// Hour numbers use native-Korean counting, not Sino-Korean — must run
|
||
// before the generic number pass below so "9시" doesn't become "구시".
|
||
.replace(/(\d{1,2})(\s*시간?)/g, _hourReplacer)
|
||
// Final pass: every remaining numeral becomes spelled-out Hangul — the
|
||
// single biggest reliability win measured today (raw numerals: frequent
|
||
// digit swaps/drops; spelled-out: ~7/8 clean in repeated STT round-trip
|
||
// testing, 2026-07-10).
|
||
.replace(/-?\d+(?:\.\d+)?/g, (m) => _numToHangulWord(m))
|
||
.replace(/\s+/g, ' ')
|
||
.trim();
|
||
}
|
||
|
||
function enqueueSentence(text) {
|
||
text = sanitizeForSpeech((text || '').trim());
|
||
if (!text) return;
|
||
ttsQueue.push(text);
|
||
pumpTtsQueue();
|
||
}
|
||
|
||
function flushPendingTts() {
|
||
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
|
||
}
|
||
|
||
// Accumulates sentence fragments until there's enough text to be safe to
|
||
// stream to XTTS, then enqueues the merged chunk. A line break (bullet
|
||
// item, paragraph line) is an intentional structural separation, though —
|
||
// never merge it with a neighbor, same reasoning as app.js's text-button
|
||
// fix. Without this, a short line like "습도: 86%" (under MIN_TTS_CHARS)
|
||
// silently absorbed the next bullet line and the pause between them
|
||
// vanished (reported 2026-07-10).
|
||
function bufferSentence(sentence, isLineBreak) {
|
||
if (isLineBreak) {
|
||
flushPendingTts();
|
||
enqueueSentence(sentence);
|
||
return;
|
||
}
|
||
pendingTts = pendingTts ? pendingTts + ' ' + sentence : sentence;
|
||
if (pendingTts.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_TTS_CHARS) {
|
||
flushPendingTts();
|
||
}
|
||
}
|
||
|
||
function splitCompleteSentences(text) {
|
||
const sentences = [];
|
||
let start = 0;
|
||
for (let i = 0; i < text.length; i++) {
|
||
// A "." between two digits is a decimal point (e.g. "27.7"), not a
|
||
// sentence end — without this guard, streamed numbers get sliced in
|
||
// half mid-decimal into two separate TTS calls (found 2026-07-10 while
|
||
// debugging why weather readouts sounded choppy).
|
||
if (text[i] === '.' && /\d/.test(text[i - 1] || '') && /\d/.test(text[i + 1] || '')) continue;
|
||
if (/[.!?\n]/.test(text[i])) {
|
||
const boundaryChar = text[i];
|
||
let j = i + 1;
|
||
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
|
||
let piece = text.slice(start, j).trim();
|
||
// bufferSentence() rejoins pieces with a plain space, not a real
|
||
// "\n" — so by the time sanitizeForSpeech runs, its "^...$/gm"
|
||
// bullet-marker regex only sees ONE line (the whole joined string)
|
||
// and misses every bullet after the first. Strip it per-piece here,
|
||
// while a real line boundary still exists to anchor on (2026-07-10).
|
||
piece = piece.replace(/^[-*+]\s+/, '');
|
||
// A "\n" boundary carries no punctuation of its own, so the .trim()
|
||
// above silently erases the pause it implied — sanitizeForSpeech's
|
||
// \n handling never gets a chance to see it since the newline is
|
||
// gone by the time this piece reaches it. Put an explicit period
|
||
// back so list items / paragraph breaks still read as a pause
|
||
// instead of running straight into the next line (2026-07-10).
|
||
if (piece && boundaryChar === '\n' && !/[.!?,:;]$/.test(piece)) {
|
||
piece += '.';
|
||
}
|
||
if (piece) sentences.push({ text: piece, isLineBreak: boundaryChar === '\n' });
|
||
start = j;
|
||
i = j - 1;
|
||
}
|
||
}
|
||
return { sentences, consumedLength: start };
|
||
}
|
||
|
||
// Hooked from app.js when partialContent is replaced wholesale rather than
|
||
// appended to (e.g. a multi-step tool turn's finalAnswer) — spokenUpTo was an
|
||
// offset into the old string and must be dropped before the new text arrives.
|
||
window._voiceCallResetSpoken = function () {
|
||
spokenUpTo = 0;
|
||
};
|
||
|
||
// Hooked from app.js's SSE token handler — only acts while a call is active.
|
||
window._voiceCallOnToken = function (fullPartialContent) {
|
||
if (!active) return;
|
||
const tail = fullPartialContent.slice(spokenUpTo);
|
||
const { sentences, consumedLength } = splitCompleteSentences(tail);
|
||
if (consumedLength > 0) {
|
||
spokenUpTo += consumedLength;
|
||
for (const s of sentences) bufferSentence(s.text, s.isLineBreak);
|
||
}
|
||
};
|
||
|
||
// Hooked from app.js when the SSE stream for a turn finishes.
|
||
window._voiceCallOnTurnDone = function (finalText) {
|
||
if (!active) return;
|
||
const remaining = (finalText || '').slice(spokenUpTo).trim();
|
||
if (remaining) bufferSentence(remaining);
|
||
flushPendingTts();
|
||
spokenUpTo = 0;
|
||
};
|
||
})();
|