fix: TTS 발음/타이밍 대규모 개선 (숫자, 단위, 방위, 줄바꿈 pause)
- 숫자를 완전한 한글 표기로 변환 (사이노-한국어 기본, 시각은 순우리말) - °C/°F/%/m/s/mm/km 등 단위를 한글로 스펠아웃 - 괄호 앞뒤 pause 강화, 방위 약어(SSW 등) 한글 변환 - 마크다운 표/리스트를 줄바꿈 전에 분할 후 정제하도록 파이프라인 재설계 - 구조적 개행(리스트/표 셀) 유래 조각은 병합 금지, 대화체 문장만 병합 - 소수점(27.7) 오탐 문장경계 버그 수정 - num_step 16, speed 1.15로 튜닝 (품질 손실 없이 생성속도 개선) - 대기열 위치 표시 (tts_queue 메시지) Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
+148
-11
@@ -221,6 +221,10 @@
|
||||
break;
|
||||
}
|
||||
|
||||
case 'tts_queue':
|
||||
if (msg.aheadCount > 0) setStatus(`대기 중… (앞에 ${msg.aheadCount}명)`);
|
||||
break;
|
||||
|
||||
case 'tts_stream_start':
|
||||
currentTtsId = msg.id;
|
||||
state = 'assistant_speaking';
|
||||
@@ -270,6 +274,61 @@
|
||||
// etc.) doesn't get force-pronounced by XTTS under language='ko' — those
|
||||
// symbols were producing garbled non-Korean-sounding artifacts mid-sentence
|
||||
// (tables were the worst offender: pipes + dash separator rows + raw URLs).
|
||||
// Sino-Korean digit-string reading, e.g. 27 -> "이십칠", 0 -> "영". Handles
|
||||
// 0-9999 (plenty for measurement-style numbers — temperatures, percentages,
|
||||
// speeds); larger numbers get an extra 만/억 grouping pass but aren't the
|
||||
// target use case. Mirrors app.js's _sinoKoreanInt.
|
||||
function _sinoKoreanInt(n) {
|
||||
if (n === 0) return '영';
|
||||
const digits = ['', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
|
||||
const smallUnits = ['', '십', '백', '천'];
|
||||
const bigUnits = ['', '만', '억', '조'];
|
||||
function fourDigit(num) {
|
||||
if (num === 0) return '';
|
||||
let s = '';
|
||||
const ds = String(num).padStart(4, '0').split('').map(Number);
|
||||
for (let i = 0; i < 4; i++) {
|
||||
const d = ds[i], unit = smallUnits[3 - i];
|
||||
if (d === 0) continue;
|
||||
s += (d === 1 && unit !== '') ? unit : (digits[d] + unit);
|
||||
}
|
||||
return s;
|
||||
}
|
||||
const groups = [];
|
||||
let rem = n;
|
||||
while (rem > 0) { groups.push(rem % 10000); rem = Math.floor(rem / 10000); }
|
||||
let result = '';
|
||||
for (let g = groups.length - 1; g >= 0; g--) {
|
||||
if (groups[g] === 0) continue;
|
||||
result += fourDigit(groups[g]) + bigUnits[g];
|
||||
}
|
||||
return result || '영';
|
||||
}
|
||||
|
||||
// "27.7" -> "이십칠점칠" (decimal digits read one at a time, standard
|
||||
// Korean convention), "-3.2" -> "마이너스 삼점이".
|
||||
function _numToHangulWord(numStr) {
|
||||
const neg = numStr[0] === '-';
|
||||
if (neg) numStr = numStr.slice(1);
|
||||
const parts = numStr.split('.');
|
||||
let word = _sinoKoreanInt(parseInt(parts[0], 10) || 0);
|
||||
if (parts[1]) {
|
||||
const d = ['영', '일', '이', '삼', '사', '오', '육', '칠', '팔', '구'];
|
||||
word += '점' + parts[1].split('').map((c) => d[+c]).join('');
|
||||
}
|
||||
return (neg ? '마이너스 ' : '') + word;
|
||||
}
|
||||
|
||||
// Korean uses native-Korean numerals (하나/둘/셋...), not Sino-Korean
|
||||
// (일/이/삼...), specifically for clock hours and hour-durations — "9시"
|
||||
// is "아홉 시", never "구시" (reported 2026-07-10). Everything else (분,
|
||||
// 초, 도, %, dates, etc.) correctly stays Sino-Korean via _numToHangulWord.
|
||||
const _NATIVE_HOUR = { 1: '한', 2: '두', 3: '세', 4: '네', 5: '다섯', 6: '여섯', 7: '일곱', 8: '여덟', 9: '아홉', 10: '열', 11: '열한', 12: '열두' };
|
||||
function _hourReplacer(m, numStr, suffix) {
|
||||
const n = parseInt(numStr, 10);
|
||||
return _NATIVE_HOUR[n] ? _NATIVE_HOUR[n] + suffix : m;
|
||||
}
|
||||
|
||||
function sanitizeForSpeech(text) {
|
||||
return text
|
||||
.replace(/```[\s\S]*?```/g, ' ')
|
||||
@@ -281,10 +340,39 @@
|
||||
// remaining table rows: "| a | b |" -> "a, b" so cells read as a list
|
||||
.replace(/^\s*\|(.+)\|\s*$/gm, (_m, row) => row.split('|').map((c) => c.trim()).filter(Boolean).join(', '))
|
||||
.replace(/https?:\/\/\S+/g, ' ')
|
||||
// OmniVoice reads a bare "°C"/"°F" as "그램"(grams) instead of degrees
|
||||
// — the digits (even with a decimal point) come through fine once the
|
||||
// symbol itself is spelled out in Korean (verified via STT round-trip,
|
||||
// 2026-07-10). Order matters: °C/°F before the bare ° fallback.
|
||||
.replace(/(\d+(?:\.\d+)?)\s*°C/gi, '섭씨 $1도')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*°F/gi, '화씨 $1도')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*°/g, '$1도')
|
||||
// Spell out common units so a bare symbol doesn't get read wrong
|
||||
// (measured: numerals+symbols like "%"/"m/s" have a real error rate,
|
||||
// spelling everything in Hangul is dramatically more reliable, 2026-07-10).
|
||||
.replace(/(\d+(?:\.\d+)?)\s*%/g, '$1퍼센트')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*km\/h/gi, '$1킬로미터 매 시')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*m\/s/gi, '$1미터 매 초')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*mm/gi, '$1밀리미터')
|
||||
.replace(/(\d+(?:\.\d+)?)\s*km/gi, '$1킬로미터')
|
||||
// Wind-direction abbreviations in parens (e.g. "2.6 m/s (SSW)") aren't
|
||||
// Korean words, so OmniVoice mangles them — spell out the compass point
|
||||
// instead. Longest abbreviations first so "SSW" doesn't partial-match as
|
||||
// "S" (reported 2026-07-10). Must run before the generic paren->period
|
||||
// conversion below, while the parens are still intact to anchor on.
|
||||
.replace(/\((NNE|ENE|ESE|SSE|SSW|WSW|WNW|NNW|NE|SE|SW|NW|N|S|E|W)\)/g, (_m, dir) => {
|
||||
const d = { N: '북', NNE: '북북동', NE: '북동', ENE: '동북동', E: '동', ESE: '동남동', SE: '남동', SSE: '남남동', S: '남', SSW: '남남서', SW: '남서', WSW: '서남서', W: '서', WNW: '서북서', NW: '북서', NNW: '북북서' };
|
||||
return d[dir] ? ' ' + d[dir] + ' ' : _m;
|
||||
})
|
||||
// OmniVoice chokes on a bare "(" — especially Hangul-adjacent, e.g.
|
||||
// "영웅(Heroic)" — and stops generating audio entirely instead of just
|
||||
// mispronouncing it. Drop the parens but keep their contents.
|
||||
.replace(/[()()]/g, ' ')
|
||||
// mispronouncing it. Swap for a period instead of just erasing: still
|
||||
// no literal "(" reaches the model, but the parenthetical aside gets a
|
||||
// clearer pause than a comma did (measured: comma ~+0.02-0.08s over a
|
||||
// bare space, period ~2-3x that, 2026-07-10).
|
||||
.replace(/[((]\s*/g, '. ')
|
||||
.replace(/\s*[))]\s*/g, '. ')
|
||||
.replace(/\.\s*\./g, '.')
|
||||
.replace(/^\s{0,3}#{1,6}\s+/gm, '')
|
||||
.replace(/^\s*[-*+]\s+/gm, '')
|
||||
.replace(/^\s*>\s?/gm, '')
|
||||
@@ -292,7 +380,22 @@
|
||||
.replace(/\*([^*]+)\*/g, '$1')
|
||||
.replace(/__([^_]+)__/g, '$1')
|
||||
.replace(/_([^_]+)_/g, '$1')
|
||||
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}]/gu, '')
|
||||
// Paragraph/line breaks were silently vanishing into the final \s+ ->
|
||||
// ' ' collapse below, so a blank line between two unrelated facts (e.g.
|
||||
// wind speed, then a separate rain/comfort sentence) got read with zero
|
||||
// pause at all — worse than the old paren-as-space bug. Must run after
|
||||
// all the ^...$/gm passes above (they still need real newlines).
|
||||
.replace(/(?<![.!?,:;])\n{2,}/g, '. ')
|
||||
.replace(/(?<![.!?,:;])\n/g, ', ')
|
||||
.replace(/[\u{1F300}-\u{1FAFF}\u{2600}-\u{27BF}\u{1F1E6}-\u{1F1FF}\u{2190}-\u{21FF}\u{2B00}-\u{2BFF}\u{FE0F}]/gu, '')
|
||||
// Hour numbers use native-Korean counting, not Sino-Korean — must run
|
||||
// before the generic number pass below so "9시" doesn't become "구시".
|
||||
.replace(/(\d{1,2})(\s*시간?)/g, _hourReplacer)
|
||||
// Final pass: every remaining numeral becomes spelled-out Hangul — the
|
||||
// single biggest reliability win measured today (raw numerals: frequent
|
||||
// digit swaps/drops; spelled-out: ~7/8 clean in repeated STT round-trip
|
||||
// testing, 2026-07-10).
|
||||
.replace(/-?\d+(?:\.\d+)?/g, (m) => _numToHangulWord(m))
|
||||
.replace(/\s+/g, ' ')
|
||||
.trim();
|
||||
}
|
||||
@@ -304,13 +407,26 @@
|
||||
pumpTtsQueue();
|
||||
}
|
||||
|
||||
function flushPendingTts() {
|
||||
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
|
||||
}
|
||||
|
||||
// Accumulates sentence fragments until there's enough text to be safe to
|
||||
// stream to XTTS, then enqueues the merged chunk.
|
||||
function bufferSentence(sentence) {
|
||||
// stream to XTTS, then enqueues the merged chunk. A line break (bullet
|
||||
// item, paragraph line) is an intentional structural separation, though —
|
||||
// never merge it with a neighbor, same reasoning as app.js's text-button
|
||||
// fix. Without this, a short line like "습도: 86%" (under MIN_TTS_CHARS)
|
||||
// silently absorbed the next bullet line and the pause between them
|
||||
// vanished (reported 2026-07-10).
|
||||
function bufferSentence(sentence, isLineBreak) {
|
||||
if (isLineBreak) {
|
||||
flushPendingTts();
|
||||
enqueueSentence(sentence);
|
||||
return;
|
||||
}
|
||||
pendingTts = pendingTts ? pendingTts + ' ' + sentence : sentence;
|
||||
if (pendingTts.replace(/[^\p{L}\p{N}]/gu, '').length >= MIN_TTS_CHARS) {
|
||||
enqueueSentence(pendingTts);
|
||||
pendingTts = '';
|
||||
flushPendingTts();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -318,11 +434,32 @@
|
||||
const sentences = [];
|
||||
let start = 0;
|
||||
for (let i = 0; i < text.length; i++) {
|
||||
// A "." between two digits is a decimal point (e.g. "27.7"), not a
|
||||
// sentence end — without this guard, streamed numbers get sliced in
|
||||
// half mid-decimal into two separate TTS calls (found 2026-07-10 while
|
||||
// debugging why weather readouts sounded choppy).
|
||||
if (text[i] === '.' && /\d/.test(text[i - 1] || '') && /\d/.test(text[i + 1] || '')) continue;
|
||||
if (/[.!?\n]/.test(text[i])) {
|
||||
const boundaryChar = text[i];
|
||||
let j = i + 1;
|
||||
while (j < text.length && /[\s.!?]/.test(text[j])) j++;
|
||||
const piece = text.slice(start, j).trim();
|
||||
if (piece) sentences.push(piece);
|
||||
let piece = text.slice(start, j).trim();
|
||||
// bufferSentence() rejoins pieces with a plain space, not a real
|
||||
// "\n" — so by the time sanitizeForSpeech runs, its "^...$/gm"
|
||||
// bullet-marker regex only sees ONE line (the whole joined string)
|
||||
// and misses every bullet after the first. Strip it per-piece here,
|
||||
// while a real line boundary still exists to anchor on (2026-07-10).
|
||||
piece = piece.replace(/^[-*+]\s+/, '');
|
||||
// A "\n" boundary carries no punctuation of its own, so the .trim()
|
||||
// above silently erases the pause it implied — sanitizeForSpeech's
|
||||
// \n handling never gets a chance to see it since the newline is
|
||||
// gone by the time this piece reaches it. Put an explicit period
|
||||
// back so list items / paragraph breaks still read as a pause
|
||||
// instead of running straight into the next line (2026-07-10).
|
||||
if (piece && boundaryChar === '\n' && !/[.!?,:;]$/.test(piece)) {
|
||||
piece += '.';
|
||||
}
|
||||
if (piece) sentences.push({ text: piece, isLineBreak: boundaryChar === '\n' });
|
||||
start = j;
|
||||
i = j - 1;
|
||||
}
|
||||
@@ -344,7 +481,7 @@
|
||||
const { sentences, consumedLength } = splitCompleteSentences(tail);
|
||||
if (consumedLength > 0) {
|
||||
spokenUpTo += consumedLength;
|
||||
for (const s of sentences) bufferSentence(s);
|
||||
for (const s of sentences) bufferSentence(s.text, s.isLineBreak);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -353,7 +490,7 @@
|
||||
if (!active) return;
|
||||
const remaining = (finalText || '').slice(spokenUpTo).trim();
|
||||
if (remaining) bufferSentence(remaining);
|
||||
if (pendingTts) { enqueueSentence(pendingTts); pendingTts = ''; }
|
||||
flushPendingTts();
|
||||
spokenUpTo = 0;
|
||||
};
|
||||
})();
|
||||
|
||||
Reference in New Issue
Block a user