fix: TTS 발음/타이밍 대규모 개선 (숫자, 단위, 방위, 줄바꿈 pause)
- 숫자를 완전한 한글 표기로 변환 (사이노-한국어 기본, 시각은 순우리말) - °C/°F/%/m/s/mm/km 등 단위를 한글로 스펠아웃 - 괄호 앞뒤 pause 강화, 방위 약어(SSW 등) 한글 변환 - 마크다운 표/리스트를 줄바꿈 전에 분할 후 정제하도록 파이프라인 재설계 - 구조적 개행(리스트/표 셀) 유래 조각은 병합 금지, 대화체 문장만 병합 - 소수점(27.7) 오탐 문장경계 버그 수정 - num_step 16, speed 1.15로 튜닝 (품질 손실 없이 생성속도 개선) - 대기열 위치 표시 (tts_queue 메시지) Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -69,6 +69,25 @@ def crash_and_restart(where: str, e: Exception):
|
||||
os._exit(1)
|
||||
|
||||
|
||||
# Every TTS request (call streaming + the single-shot text button) serializes
|
||||
# on the same GPU (measured: one request already saturates it, see
|
||||
# project_tts_edge memory), so this counter doubles as an honest queue
|
||||
# position — no separate scheduler needed.
|
||||
_gpu_pending_lock = threading.Lock()
|
||||
_gpu_pending = 0
|
||||
|
||||
def gpu_queue_enter() -> int:
|
||||
global _gpu_pending
|
||||
with _gpu_pending_lock:
|
||||
_gpu_pending += 1
|
||||
return _gpu_pending
|
||||
|
||||
def gpu_queue_exit():
|
||||
global _gpu_pending
|
||||
with _gpu_pending_lock:
|
||||
_gpu_pending -= 1
|
||||
|
||||
|
||||
def load_models(args):
|
||||
log.info('Loading faster-whisper model=%s device=%s ...', args.stt_model, args.device)
|
||||
from faster_whisper import WhisperModel
|
||||
@@ -137,6 +156,8 @@ class Engine:
|
||||
def synthesize_full(self, text: str, speed=None, num_step=None):
|
||||
if num_step is None:
|
||||
num_step = 16
|
||||
if speed is None:
|
||||
speed = 1.15
|
||||
audio = self.tts_model.generate(
|
||||
text=text[:4000],
|
||||
language='Korean',
|
||||
@@ -221,13 +242,29 @@ async def handle_tts_start(ws, engine: Engine, state: ConnectionState, msg: dict
|
||||
loop.call_soon_threadsafe(queue.put_nowait, ('error', str(e)))
|
||||
if is_unrecoverable_cuda_error(e):
|
||||
crash_and_restart('handle_tts_start', e)
|
||||
finally:
|
||||
gpu_queue_exit()
|
||||
|
||||
position = gpu_queue_enter()
|
||||
if position > 1:
|
||||
# Someone else's synthesis is already running on the GPU (it's
|
||||
# effectively serialized — see project_tts_edge memory); let the
|
||||
# client show a queue indicator instead of silently hanging.
|
||||
await ws.send(json.dumps({'type': 'tts_queue', 'aheadCount': position - 1, 'id': req_id}))
|
||||
threading.Thread(target=produce, daemon=True).start()
|
||||
await ws.send(json.dumps({'type': 'tts_stream_start', 'sampleRate': engine.tts_sample_rate, 'id': req_id}))
|
||||
try:
|
||||
stream_started = False
|
||||
while True:
|
||||
kind, payload = await queue.get()
|
||||
if kind == 'chunk':
|
||||
if not stream_started:
|
||||
# synthesize_stream yields the whole clip as one chunk
|
||||
# (see its docstring) once GPU generation actually
|
||||
# finishes, so this is the right moment to tell the
|
||||
# client "now speaking" — sending it eagerly up front
|
||||
# would stomp the tts_queue status above.
|
||||
await ws.send(json.dumps({'type': 'tts_stream_start', 'sampleRate': engine.tts_sample_rate, 'id': req_id}))
|
||||
stream_started = True
|
||||
await ws.send(payload)
|
||||
elif kind == 'error':
|
||||
await ws.send(json.dumps({'type': 'error', 'message': payload, 'id': req_id}))
|
||||
@@ -268,7 +305,11 @@ async def handle_connection(ws, engine: Engine):
|
||||
speed = msg.get('speed')
|
||||
num_step = msg.get('num_step')
|
||||
loop = asyncio.get_event_loop()
|
||||
pcm, sample_rate = await loop.run_in_executor(None, engine.synthesize_full, text, speed, num_step)
|
||||
gpu_queue_enter()
|
||||
try:
|
||||
pcm, sample_rate = await loop.run_in_executor(None, engine.synthesize_full, text, speed, num_step)
|
||||
finally:
|
||||
gpu_queue_exit()
|
||||
await ws.send(json.dumps({
|
||||
'type': 'tts_result',
|
||||
'audioBase64': base64.b64encode(pcm).decode('ascii'),
|
||||
|
||||
+259
-27
@@ -5697,6 +5697,7 @@ var _ttsEnabled = localStorage.getItem('ttsEnabled') !== 'false'; // default on
|
||||
var _ttsAudio = null;
|
||||
var _ttsBtnActive = null;
|
||||
var _ttsAbort = null;
|
||||
var _ttsSeqResolve = null; // resolves the currently-awaited piece's completion, so _ttsStop() can unblock the playback loop
|
||||
var _appSendFn = null;
|
||||
|
||||
function initAppVoice(sendFn) {
|
||||
@@ -5707,9 +5708,139 @@ function initAppVoice(sendFn) {
|
||||
function _ttsStop() {
|
||||
if (_ttsAbort) { _ttsAbort.abort(); _ttsAbort = null; }
|
||||
if (_ttsAudio) { _ttsAudio.pause(); _ttsAudio = null; }
|
||||
if (_ttsSeqResolve) { var r = _ttsSeqResolve; _ttsSeqResolve = null; r(); }
|
||||
if (_ttsBtnActive) { _ttsBtnActive.textContent = '🔊'; _ttsBtnActive.classList.remove('playing'); _ttsBtnActive = null; }
|
||||
}
|
||||
|
||||
// Splits raw text (before per-piece cleaning) into sentence/line-sized
|
||||
// pieces synthesized and played back-to-back, instead of one big combined
|
||||
// call. A single long combined call to OmniVoice measurably drops/garbles
|
||||
// content (verified via STT round-trip, 2026-07-10, see project_tts_edge
|
||||
// memory) — short calls are far more reliable. Splits on "\n" too, not just
|
||||
// .!?, so bullet-list facts each become their own call rather than getting
|
||||
// joined into one comma-separated blob (a comma pause alone wasn't enough —
|
||||
// still failed reliably in testing). Tiny fragments get merged (mirrors
|
||||
// voice-call.js's MIN_TTS_CHARS=15) so we don't fire a round-trip per clause.
|
||||
// Splits raw text into pieces synthesized and played back-to-back, one pass:
|
||||
// short sentence/line fragments accumulate into a buffer and flush once they
|
||||
// cross MIN_CHARS (mirrors voice-call.js's MIN_TTS_CHARS=15, but lower — that
|
||||
// threshold dates back to XTTS's streaming inference crashing on very short
|
||||
// standalone text, which OmniVoice doesn't share; tested down to ~6 chars).
|
||||
// Markdown table cells are the exception: each one is pushed as its own
|
||||
// never-merged piece — tested 8/8 clean vs. frequent drops when cells get
|
||||
// joined into one comma-list or merged across a row boundary (2026-07-10,
|
||||
// e.g. "81%" vanishing, or "습도 바람 서울" nonsensically mixing header/data
|
||||
// cells together). A separator row ("|---|---|") is dropped, it has no
|
||||
// spoken content. Leading bullet markers get stripped per-piece too — once
|
||||
// merged with plain spaces (not a real "\n"), _cleanForSpeech's ^...$/gm
|
||||
// regexes only see the whole joined string as one line.
|
||||
function _splitForSpeech(text) {
|
||||
var out = [];
|
||||
var buf = '';
|
||||
var MIN_CHARS = 6;
|
||||
function flush() { if (buf) { out.push(buf); buf = ''; } }
|
||||
function addPiece(s, isLineBreak) {
|
||||
s = s.trim();
|
||||
if (!s) return;
|
||||
if (/^\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?$/.test(s)) return; // separator row
|
||||
var rowMatch = /^\|(.+)\|$/.exec(s);
|
||||
if (rowMatch) {
|
||||
flush(); // don't let a pending sentence fragment bleed into table cells
|
||||
// Numbers are where the model actually loses reliability (measured
|
||||
// repeatedly, project_tts_edge memory) — isolating every single cell
|
||||
// fixed that but made a whole table read out painfully slowly (one
|
||||
// network round-trip per cell). Words alone rarely garble, so merge
|
||||
// consecutive non-numeric cells (e.g. city + weather condition) and
|
||||
// only isolate cells that contain a digit — cuts call count roughly in
|
||||
// half on a typical weather table while keeping the error-prone cells
|
||||
// separate (reported 2026-07-10).
|
||||
var cellBuf = '';
|
||||
rowMatch[1].split('|').map(function(c) { return c.trim(); }).filter(Boolean).forEach(function(cell) {
|
||||
if (/\d/.test(cell)) {
|
||||
if (cellBuf) { out.push(cellBuf); cellBuf = ''; }
|
||||
out.push(cell);
|
||||
} else {
|
||||
cellBuf = cellBuf ? cellBuf + ', ' + cell : cell;
|
||||
}
|
||||
});
|
||||
if (cellBuf) out.push(cellBuf);
|
||||
return;
|
||||
}
|
||||
var bulletStripped = s.replace(/^[-*+]\s+/, '');
|
||||
if (!bulletStripped) return;
|
||||
if (isLineBreak) {
|
||||
// A line break (bullet item, paragraph line) is an intentional
|
||||
// structural separation, same reasoning as table cells above — never
|
||||
// merge it with a neighbor. Without this, a short line like "습도:
|
||||
// 86%" (under MIN_CHARS) silently absorbed the next bullet line and
|
||||
// the pause between them vanished (reported 2026-07-10). Plain prose
|
||||
// sentences (.!? boundaries, not a line break) still merge below —
|
||||
// that's normal conversational flow, not an itemized fact list.
|
||||
flush();
|
||||
out.push(bulletStripped);
|
||||
return;
|
||||
}
|
||||
buf = buf ? buf + ' ' + bulletStripped : bulletStripped;
|
||||
if (buf.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_CHARS) flush();
|
||||
}
|
||||
|
||||
var start = 0;
|
||||
for (var i = 0; i < text.length; i++) {
|
||||
if (text[i] === '.' && /\d/.test(text[i - 1] || '') && /\d/.test(text[i + 1] || '')) continue; // decimal point, not a boundary
|
||||
if (/[.!?\n]/.test(text[i])) {
|
||||
var j = i + 1;
|
||||
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
|
||||
addPiece(text.slice(start, j), text[i] === '\n');
|
||||
start = j;
|
||||
i = j - 1;
|
||||
}
|
||||
}
|
||||
addPiece(text.slice(start), false);
|
||||
flush();
|
||||
return out.length ? out : [text];
|
||||
}
|
||||
|
||||
async function _ttsPlaySequence(btn, pieces) {
|
||||
for (var idx = 0; idx < pieces.length; idx++) {
|
||||
if (_ttsBtnActive !== btn) return; // stopped/superseded before this piece started
|
||||
var cleaned = _cleanForSpeech(pieces[idx]);
|
||||
if (!cleaned) continue;
|
||||
var ac = new AbortController();
|
||||
_ttsAbort = ac;
|
||||
var blob;
|
||||
try {
|
||||
var r = await fetch('/api/voice/tts', {
|
||||
method: 'POST',
|
||||
headers: Object.assign({ 'Content-Type': 'application/json' }, authHeaders()),
|
||||
credentials: 'include',
|
||||
body: JSON.stringify({ text: cleaned.slice(0, 2000) }),
|
||||
signal: ac.signal,
|
||||
});
|
||||
if (!r.ok) throw new Error('TTS 오류 ' + r.status);
|
||||
blob = await r.blob();
|
||||
} catch (e) {
|
||||
if (e.name === 'AbortError' || _ttsBtnActive !== btn) return;
|
||||
_ttsStop();
|
||||
alert('TTS 오류: ' + e.message);
|
||||
return;
|
||||
}
|
||||
if (_ttsBtnActive !== btn) return; // superseded while this piece was fetching
|
||||
var url = URL.createObjectURL(blob);
|
||||
var audio = new Audio(url);
|
||||
_ttsAudio = audio;
|
||||
btn.textContent = '⏹';
|
||||
var done = new Promise(function(resolve) {
|
||||
_ttsSeqResolve = resolve;
|
||||
audio.onended = function() { URL.revokeObjectURL(url); _ttsSeqResolve = null; resolve(); };
|
||||
audio.onerror = function() { URL.revokeObjectURL(url); _ttsSeqResolve = null; resolve(); };
|
||||
});
|
||||
audio.play();
|
||||
await done;
|
||||
if (_ttsBtnActive !== btn) return; // stopped mid-playback
|
||||
}
|
||||
if (_ttsBtnActive === btn) _ttsStop(); // whole sequence finished naturally
|
||||
}
|
||||
|
||||
// Shared core for playTTSText/playTTSMsg. Guards every async callback with a
|
||||
// "is this still the active button?" check so a slow/stale response from a
|
||||
// superseded click can never start overlapping playback (the echo bug).
|
||||
@@ -5721,34 +5852,136 @@ function _ttsPlay(btn, text) {
|
||||
|
||||
btn.textContent = '⌛'; btn.classList.add('playing');
|
||||
_ttsBtnActive = btn;
|
||||
var ac = new AbortController();
|
||||
_ttsAbort = ac;
|
||||
fetch('/api/voice/tts', {
|
||||
method: 'POST',
|
||||
headers: Object.assign({ 'Content-Type': 'application/json' }, authHeaders()),
|
||||
credentials: 'include',
|
||||
body: JSON.stringify({ text: text.slice(0, 2000) }),
|
||||
signal: ac.signal,
|
||||
})
|
||||
.then(function(r) { if (!r.ok) throw new Error('TTS 오류 ' + r.status); return r.blob(); })
|
||||
.then(function(blob) {
|
||||
if (_ttsBtnActive !== btn) return; // superseded by a newer click
|
||||
var url = URL.createObjectURL(blob);
|
||||
_ttsAudio = new Audio(url);
|
||||
_ttsAudio.onended = function() { URL.revokeObjectURL(url); if (_ttsBtnActive === btn) _ttsStop(); };
|
||||
_ttsAudio.onerror = function() { URL.revokeObjectURL(url); if (_ttsBtnActive === btn) _ttsStop(); };
|
||||
btn.textContent = '⏹'; _ttsAudio.play();
|
||||
})
|
||||
.catch(function(e) {
|
||||
if (e.name === 'AbortError' || _ttsBtnActive !== btn) return;
|
||||
_ttsStop();
|
||||
alert('TTS 오류: ' + e.message);
|
||||
});
|
||||
_ttsPlaySequence(btn, _splitForSpeech(text));
|
||||
}
|
||||
|
||||
// Sino-Korean digit-string reading, e.g. 27 -> "이십칠", 0 -> "영". Handles
|
||||
// 0-9999 (plenty for the measurement-style numbers TTS text actually
|
||||
// contains — temperatures, percentages, speeds); larger numbers get an
|
||||
// extra 만/억 grouping pass but aren't the target use case.
|
||||
function _sinoKoreanInt(n) {
|
||||
if (n === 0) return '영';
|
||||
var digits = ['', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
|
||||
var smallUnits = ['', '십', '백', '천'];
|
||||
var bigUnits = ['', '만', '억', '조'];
|
||||
function fourDigit(num) {
|
||||
if (num === 0) return '';
|
||||
var s = '';
|
||||
var ds = String(num).padStart(4, '0').split('').map(Number);
|
||||
for (var i = 0; i < 4; i++) {
|
||||
var d = ds[i], unit = smallUnits[3 - i];
|
||||
if (d === 0) continue;
|
||||
s += (d === 1 && unit !== '') ? unit : (digits[d] + unit);
|
||||
}
|
||||
return s;
|
||||
}
|
||||
var groups = [];
|
||||
var rem = n;
|
||||
while (rem > 0) { groups.push(rem % 10000); rem = Math.floor(rem / 10000); }
|
||||
var result = '';
|
||||
for (var g = groups.length - 1; g >= 0; g--) {
|
||||
if (groups[g] === 0) continue;
|
||||
result += fourDigit(groups[g]) + bigUnits[g];
|
||||
}
|
||||
return result || '영';
|
||||
}
|
||||
|
||||
// "27.7" -> "이십칠점칠" (decimal digits read one at a time, standard
|
||||
// Korean convention), "-3.2" -> "마이너스 삼점이".
|
||||
function _numToHangulWord(numStr) {
|
||||
var neg = numStr[0] === '-';
|
||||
if (neg) numStr = numStr.slice(1);
|
||||
var parts = numStr.split('.');
|
||||
var word = _sinoKoreanInt(parseInt(parts[0], 10) || 0);
|
||||
if (parts[1]) {
|
||||
var d = ['영', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
|
||||
word += '점' + parts[1].split('').map(function(c) { return d[+c]; }).join('');
|
||||
}
|
||||
return (neg ? '마이너스 ' : '') + word;
|
||||
}
|
||||
|
||||
// Korean uses native-Korean numerals (하나/둘/셋...), not Sino-Korean
|
||||
// (일/이/삼...), specifically for clock hours and hour-durations — "9시" is
|
||||
// "아홉 시", never "구시" (reported 2026-07-10). Everything else (분, 초,
|
||||
// 도, %, dates, etc.) correctly stays Sino-Korean via _numToHangulWord.
|
||||
var _NATIVE_HOUR = { 1: '한', 2: '두', 3: '세', 4: '네', 5: '다섯', 6: '여섯', 7: '일곱', 8: '여덟', 9: '아홉', 10: '열', 11: '열한', 12: '열두' };
|
||||
function _hourReplacer(m, numStr, suffix) {
|
||||
var n = parseInt(numStr, 10);
|
||||
return _NATIVE_HOUR[n] ? _NATIVE_HOUR[n] + suffix : m;
|
||||
}
|
||||
|
||||
// Strips markdown, emoji and bare parens before handing text to the GPU
|
||||
// voice engine (OmniVoice). Mirrors voice-call.js's sanitizeForSpeech: emoji
|
||||
// get mispronounced as garbled syllables, a lone "(" makes OmniVoice
|
||||
// stop generating audio entirely instead of just mispronouncing it, and a
|
||||
// bare "°C"/"°F" gets read as "그램"(grams) instead of degrees.
|
||||
function _cleanForSpeech(text) {
|
||||
return (text || '')
|
||||
.replace(/!\[.*?\]\(.*?\)/g, '')
|
||||
.replace(/\[([^\]]+)\]\([^)]+\)/g, '$1')
|
||||
.replace(/```[\s\S]*?```/g, '')
|
||||
// Table handling was also never ported from voice-call.js — a leftover
|
||||
// "|" right before a cell's content (e.g. "| 81% |") is yet another
|
||||
// unknown-symbol case that makes OmniVoice choke and drop content
|
||||
// (reported 2026-07-10, e.g. "81%" vanishing entirely from a table row).
|
||||
.replace(/^\s*\|?\s*:?-{2,}:?\s*(\|\s*:?-{2,}:?\s*)*\|?\s*$/gm, ' ')
|
||||
.replace(/^\s*\|(.+)\|\s*$/gm, function(_m, row) { return row.split('|').map(function(c) { return c.trim(); }).filter(Boolean).join(', '); })
|
||||
// Bullet markers were never stripped here (unlike voice-call.js's
|
||||
// sanitizeForSpeech, which already had this) — a leftover "- " right
|
||||
// after the newline-to-comma pause below reads as a rushed, breathless
|
||||
// run-on (reported 2026-07-10). Must run before \n gets consumed.
|
||||
.replace(/^\s*[-*+]\s+/gm, '')
|
||||
.replace(/[#*`_~>]/g, ' ')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*°/g, '$1도')
|
||||
// Spell out common units so a bare symbol doesn't get read wrong
|
||||
// (measured: numerals+symbols like "%"/"m/s" have a real error rate,
|
||||
// spelling everything in Hangul is dramatically more reliable, 2026-07-10).
|
||||
.replace(/(\d+(?:\.\d+)?)\s*%/g, '$1퍼센트')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*km\/h/gi, '$1킬로미터 매 시')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*m\/s/gi, '$1미터 매 초')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*mm/gi, '$1밀리미터')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*km/gi, '$1킬로미터')
|
||||
// Wind-direction abbreviations in parens (e.g. "2.6 m/s (SSW)") aren't
|
||||
// Korean words, so OmniVoice mangles them — spell out the compass point
|
||||
// instead. Longest abbreviations first so "SSW" doesn't partial-match as
|
||||
// "S" (reported 2026-07-10). Must run before the generic paren->period
|
||||
// conversion below, while the parens are still intact to anchor on.
|
||||
.replace(/\((NNE|ENE|ESE|SSE|SSW|WSW|WNW|NNW|NE|SE|SW|NW|N|S|E|W)\)/g, function(_m, dir) {
|
||||
var d = { N: '북', NNE: '북북동', NE: '북동', ENE: '동북동', E: '동', ESE: '동남동', SE: '남동', SSE: '남남동', S: '남', SSW: '남남서', SW: '남서', WSW: '서남서', W: '서', WNW: '서북서', NW: '북서', NNW: '북북서' };
|
||||
return d[dir] ? ' ' + d[dir] + ' ' : _m;
|
||||
})
|
||||
// Swap parens for a period rather than just erasing them — still no bare
|
||||
// "(" reaches the model (the crash trigger), but a period gives the
|
||||
// parenthetical aside a clearer pause than a comma did (measured: comma
|
||||
// ~+0.02-0.08s over a bare space, period ~2-3x that, 2026-07-10).
|
||||
.replace(/[((]\s*/g, '. ')
|
||||
.replace(/\s*[))]\s*/g, '. ')
|
||||
.replace(/\.\s*\./g, '.')
|
||||
// Paragraph/line breaks were silently vanishing into the final \s+ ->
|
||||
// ' ' collapse below, so a blank line between two unrelated facts (e.g.
|
||||
// wind speed, then a separate rain/comfort sentence) got read with zero
|
||||
// pause at all — worse than the old paren-as-space bug. Give them a
|
||||
// pause proportional to how strong a break they represent.
|
||||
.replace(/(?<![.!?,:;])\n{2,}/g, '. ')
|
||||
.replace(/(?<![.!?,:;])\n/g, ', ')
|
||||
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}\u{FE0F}]/gu, '')
|
||||
// Hour numbers use native-Korean counting, not Sino-Korean — must run
|
||||
// before the generic number pass below so "9시" doesn't become "구시".
|
||||
.replace(/(\d{1,2})(\s*시간?)/g, _hourReplacer)
|
||||
// Final pass: every remaining numeral (including ones the unit
|
||||
// conversions above just introduced digits next to, e.g. "27.7도")
|
||||
// becomes spelled-out Hangul — this is the single biggest reliability
|
||||
// win measured today (raw numerals: frequent digit swaps/drops;
|
||||
// spelled-out: ~7/8 clean in repeated STT round-trip testing).
|
||||
.replace(/-?\d+(?:\.\d+)?/g, function(m) { return _numToHangulWord(m); })
|
||||
.replace(/\s+/g, ' ')
|
||||
.trim();
|
||||
}
|
||||
|
||||
function playTTSText(btn, text) {
|
||||
text = (text||'').replace(/```[\s\S]*?```/g,'').replace(/[#*`_~>]/g,' ').replace(/\s+/g,' ').trim();
|
||||
_ttsPlay(btn, text);
|
||||
_ttsPlay(btn, text || '');
|
||||
}
|
||||
|
||||
function updateTTSMode() {
|
||||
@@ -5770,8 +6003,7 @@ function initTTSToggle() {
|
||||
function playTTSMsg(btn, idx) {
|
||||
var msg = chatHistory[idx];
|
||||
if (!msg) return;
|
||||
var text = (msg.content || '').replace(/!\[.*?\]\(.*?\)/g, '').replace(/\[([^\]]+)\]\([^)]+\)/g, '$1').replace(/```[\s\S]*?```/g, '').replace(/[#*`_~>]/g, ' ').replace(/\s+/g, ' ').trim();
|
||||
_ttsPlay(btn, text);
|
||||
_ttsPlay(btn, msg.content || '');
|
||||
}
|
||||
|
||||
function saveLuckyModal() {
|
||||
|
||||
+148
-11
@@ -221,6 +221,10 @@
|
||||
break;
|
||||
}
|
||||
|
||||
case 'tts_queue':
|
||||
if (msg.aheadCount > 0) setStatus(`대기 중… (앞에 ${msg.aheadCount}명)`);
|
||||
break;
|
||||
|
||||
case 'tts_stream_start':
|
||||
currentTtsId = msg.id;
|
||||
state = 'assistant_speaking';
|
||||
@@ -270,6 +274,61 @@
|
||||
// etc.) doesn't get force-pronounced by XTTS under language='ko' — those
|
||||
// symbols were producing garbled non-Korean-sounding artifacts mid-sentence
|
||||
// (tables were the worst offender: pipes + dash separator rows + raw URLs).
|
||||
// Sino-Korean digit-string reading, e.g. 27 -> "이십칠", 0 -> "영". Handles
|
||||
// 0-9999 (plenty for measurement-style numbers — temperatures, percentages,
|
||||
// speeds); larger numbers get an extra 만/억 grouping pass but aren't the
|
||||
// target use case. Mirrors app.js's _sinoKoreanInt.
|
||||
function _sinoKoreanInt(n) {
|
||||
if (n === 0) return '영';
|
||||
const digits = ['', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
|
||||
const smallUnits = ['', '십', '백', '천'];
|
||||
const bigUnits = ['', '만', '억', '조'];
|
||||
function fourDigit(num) {
|
||||
if (num === 0) return '';
|
||||
let s = '';
|
||||
const ds = String(num).padStart(4, '0').split('').map(Number);
|
||||
for (let i = 0; i < 4; i++) {
|
||||
const d = ds[i], unit = smallUnits[3 - i];
|
||||
if (d === 0) continue;
|
||||
s += (d === 1 && unit !== '') ? unit : (digits[d] + unit);
|
||||
}
|
||||
return s;
|
||||
}
|
||||
const groups = [];
|
||||
let rem = n;
|
||||
while (rem > 0) { groups.push(rem % 10000); rem = Math.floor(rem / 10000); }
|
||||
let result = '';
|
||||
for (let g = groups.length - 1; g >= 0; g--) {
|
||||
if (groups[g] === 0) continue;
|
||||
result += fourDigit(groups[g]) + bigUnits[g];
|
||||
}
|
||||
return result || '영';
|
||||
}
|
||||
|
||||
// "27.7" -> "이십칠점칠" (decimal digits read one at a time, standard
|
||||
// Korean convention), "-3.2" -> "마이너스 삼점이".
|
||||
function _numToHangulWord(numStr) {
|
||||
const neg = numStr[0] === '-';
|
||||
if (neg) numStr = numStr.slice(1);
|
||||
const parts = numStr.split('.');
|
||||
let word = _sinoKoreanInt(parseInt(parts[0], 10) || 0);
|
||||
if (parts[1]) {
|
||||
const d = ['영', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
|
||||
word += '점' + parts[1].split('').map((c) => d[+c]).join('');
|
||||
}
|
||||
return (neg ? '마이너스 ' : '') + word;
|
||||
}
|
||||
|
||||
// Korean uses native-Korean numerals (하나/둘/셋...), not Sino-Korean
|
||||
// (일/이/삼...), specifically for clock hours and hour-durations — "9시"
|
||||
// is "아홉 시", never "구시" (reported 2026-07-10). Everything else (분,
|
||||
// 초, 도, %, dates, etc.) correctly stays Sino-Korean via _numToHangulWord.
|
||||
const _NATIVE_HOUR = { 1: '한', 2: '두', 3: '세', 4: '네', 5: '다섯', 6: '여섯', 7: '일곱', 8: '여덟', 9: '아홉', 10: '열', 11: '열한', 12: '열두' };
|
||||
function _hourReplacer(m, numStr, suffix) {
|
||||
const n = parseInt(numStr, 10);
|
||||
return _NATIVE_HOUR[n] ? _NATIVE_HOUR[n] + suffix : m;
|
||||
}
|
||||
|
||||
function sanitizeForSpeech(text) {
|
||||
return text
|
||||
.replace(/```[\s\S]*?```/g, ' ')
|
||||
@@ -281,10 +340,39 @@
|
||||
// remaining table rows: "| a | b |" -> "a, b" so cells read as a list
|
||||
.replace(/^\s*\|(.+)\|\s*$/gm, (_m, row) => row.split('|').map((c) => c.trim()).filter(Boolean).join(', '))
|
||||
.replace(/https?:\/\/\S+/g, ' ')
|
||||
// OmniVoice reads a bare "°C"/"°F" as "그램"(grams) instead of degrees
|
||||
// — the digits (even with a decimal point) come through fine once the
|
||||
// symbol itself is spelled out in Korean (verified via STT round-trip,
|
||||
// 2026-07-10). Order matters: °C/°F before the bare ° fallback.
|
||||
.replace(/(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*°/g, '$1도')
|
||||
// Spell out common units so a bare symbol doesn't get read wrong
|
||||
// (measured: numerals+symbols like "%"/"m/s" have a real error rate,
|
||||
// spelling everything in Hangul is dramatically more reliable, 2026-07-10).
|
||||
.replace(/(\d+(?:\.\d+)?)\s*%/g, '$1퍼센트')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*km\/h/gi, '$1킬로미터 매 시')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*m\/s/gi, '$1미터 매 초')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*mm/gi, '$1밀리미터')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*km/gi, '$1킬로미터')
|
||||
// Wind-direction abbreviations in parens (e.g. "2.6 m/s (SSW)") aren't
|
||||
// Korean words, so OmniVoice mangles them — spell out the compass point
|
||||
// instead. Longest abbreviations first so "SSW" doesn't partial-match as
|
||||
// "S" (reported 2026-07-10). Must run before the generic paren->period
|
||||
// conversion below, while the parens are still intact to anchor on.
|
||||
.replace(/\((NNE|ENE|ESE|SSE|SSW|WSW|WNW|NNW|NE|SE|SW|NW|N|S|E|W)\)/g, (_m, dir) => {
|
||||
const d = { N: '북', NNE: '북북동', NE: '북동', ENE: '동북동', E: '동', ESE: '동남동', SE: '남동', SSE: '남남동', S: '남', SSW: '남남서', SW: '남서', WSW: '서남서', W: '서', WNW: '서북서', NW: '북서', NNW: '북북서' };
|
||||
return d[dir] ? ' ' + d[dir] + ' ' : _m;
|
||||
})
|
||||
// OmniVoice chokes on a bare "(" — especially Hangul-adjacent, e.g.
|
||||
// "영웅(Heroic)" — and stops generating audio entirely instead of just
|
||||
// mispronouncing it. Drop the parens but keep their contents.
|
||||
.replace(/[()()]/g, ' ')
|
||||
// mispronouncing it. Swap for a period instead of just erasing: still
|
||||
// no literal "(" reaches the model, but the parenthetical aside gets a
|
||||
// clearer pause than a comma did (measured: comma ~+0.02-0.08s over a
|
||||
// bare space, period ~2-3x that, 2026-07-10).
|
||||
.replace(/[((]\s*/g, '. ')
|
||||
.replace(/\s*[))]\s*/g, '. ')
|
||||
.replace(/\.\s*\./g, '.')
|
||||
.replace(/^\s{0,3}#{1,6}\s+/gm, '')
|
||||
.replace(/^\s*[-*+]\s+/gm, '')
|
||||
.replace(/^\s*>\s?/gm, '')
|
||||
@@ -292,7 +380,22 @@
|
||||
.replace(/\*([^*]+)\*/g, '$1')
|
||||
.replace(/__([^_]+)__/g, '$1')
|
||||
.replace(/_([^_]+)_/g, '$1')
|
||||
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}]/gu, '')
|
||||
// Paragraph/line breaks were silently vanishing into the final \s+ ->
|
||||
// ' ' collapse below, so a blank line between two unrelated facts (e.g.
|
||||
// wind speed, then a separate rain/comfort sentence) got read with zero
|
||||
// pause at all — worse than the old paren-as-space bug. Must run after
|
||||
// all the ^...$/gm passes above (they still need real newlines).
|
||||
.replace(/(?<![.!?,:;])\n{2,}/g, '. ')
|
||||
.replace(/(?<![.!?,:;])\n/g, ', ')
|
||||
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}\u{FE0F}]/gu, '')
|
||||
// Hour numbers use native-Korean counting, not Sino-Korean — must run
|
||||
// before the generic number pass below so "9시" doesn't become "구시".
|
||||
.replace(/(\d{1,2})(\s*시간?)/g, _hourReplacer)
|
||||
// Final pass: every remaining numeral becomes spelled-out Hangul — the
|
||||
// single biggest reliability win measured today (raw numerals: frequent
|
||||
// digit swaps/drops; spelled-out: ~7/8 clean in repeated STT round-trip
|
||||
// testing, 2026-07-10).
|
||||
.replace(/-?\d+(?:\.\d+)?/g, (m) => _numToHangulWord(m))
|
||||
.replace(/\s+/g, ' ')
|
||||
.trim();
|
||||
}
|
||||
@@ -304,13 +407,26 @@
|
||||
pumpTtsQueue();
|
||||
}
|
||||
|
||||
function flushPendingTts() {
|
||||
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
|
||||
}
|
||||
|
||||
// Accumulates sentence fragments until there's enough text to be safe to
|
||||
// stream to XTTS, then enqueues the merged chunk.
|
||||
function bufferSentence(sentence) {
|
||||
// stream to XTTS, then enqueues the merged chunk. A line break (bullet
|
||||
// item, paragraph line) is an intentional structural separation, though —
|
||||
// never merge it with a neighbor, same reasoning as app.js's text-button
|
||||
// fix. Without this, a short line like "습도: 86%" (under MIN_TTS_CHARS)
|
||||
// silently absorbed the next bullet line and the pause between them
|
||||
// vanished (reported 2026-07-10).
|
||||
function bufferSentence(sentence, isLineBreak) {
|
||||
if (isLineBreak) {
|
||||
flushPendingTts();
|
||||
enqueueSentence(sentence);
|
||||
return;
|
||||
}
|
||||
pendingTts = pendingTts ? pendingTts + ' ' + sentence : sentence;
|
||||
if (pendingTts.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_TTS_CHARS) {
|
||||
enqueueSentence(pendingTts);
|
||||
pendingTts = '';
|
||||
flushPendingTts();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -318,11 +434,32 @@
|
||||
const sentences = [];
|
||||
let start = 0;
|
||||
for (let i = 0; i < text.length; i++) {
|
||||
// A "." between two digits is a decimal point (e.g. "27.7"), not a
|
||||
// sentence end — without this guard, streamed numbers get sliced in
|
||||
// half mid-decimal into two separate TTS calls (found 2026-07-10 while
|
||||
// debugging why weather readouts sounded choppy).
|
||||
if (text[i] === '.' && /\d/.test(text[i - 1] || '') && /\d/.test(text[i + 1] || '')) continue;
|
||||
if (/[.!?\n]/.test(text[i])) {
|
||||
const boundaryChar = text[i];
|
||||
let j = i + 1;
|
||||
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
|
||||
const piece = text.slice(start, j).trim();
|
||||
if (piece) sentences.push(piece);
|
||||
let piece = text.slice(start, j).trim();
|
||||
// bufferSentence() rejoins pieces with a plain space, not a real
|
||||
// "\n" — so by the time sanitizeForSpeech runs, its "^...$/gm"
|
||||
// bullet-marker regex only sees ONE line (the whole joined string)
|
||||
// and misses every bullet after the first. Strip it per-piece here,
|
||||
// while a real line boundary still exists to anchor on (2026-07-10).
|
||||
piece = piece.replace(/^[-*+]\s+/, '');
|
||||
// A "\n" boundary carries no punctuation of its own, so the .trim()
|
||||
// above silently erases the pause it implied — sanitizeForSpeech's
|
||||
// \n handling never gets a chance to see it since the newline is
|
||||
// gone by the time this piece reaches it. Put an explicit period
|
||||
// back so list items / paragraph breaks still read as a pause
|
||||
// instead of running straight into the next line (2026-07-10).
|
||||
if (piece && boundaryChar === '\n' && !/[.!?,:;]$/.test(piece)) {
|
||||
piece += '.';
|
||||
}
|
||||
if (piece) sentences.push({ text: piece, isLineBreak: boundaryChar === '\n' });
|
||||
start = j;
|
||||
i = j - 1;
|
||||
}
|
||||
@@ -344,7 +481,7 @@
|
||||
const { sentences, consumedLength } = splitCompleteSentences(tail);
|
||||
if (consumedLength > 0) {
|
||||
spokenUpTo += consumedLength;
|
||||
for (const s of sentences) bufferSentence(s);
|
||||
for (const s of sentences) bufferSentence(s.text, s.isLineBreak);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -353,7 +490,7 @@
|
||||
if (!active) return;
|
||||
const remaining = (finalText || '').slice(spokenUpTo).trim();
|
||||
if (remaining) bufferSentence(remaining);
|
||||
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
|
||||
flushPendingTts();
|
||||
spokenUpTo = 0;
|
||||
};
|
||||
})();
|
||||
|
||||
Reference in New Issue
Block a user