Switch TTS to edge-tts (ko-KR-SunHiNeural), add TTS toggle + mic auto-send

- Replace piper TTS with edge-tts (Microsoft neural Korean voice)
  - piper ko_KR-kss-medium uses pygoruut phoneme type unsupported by C++ binary,
    causing Chinese-sounding output; edge-tts solves this via online neural TTS
  - Added src/tools/edge_tts_synth.py Python helper script
  - Server TTS endpoint now dispatches by provider (edge_tts vs piper)
  - Config updated: voice.tts.provider = "edge_tts", voice = "ko-KR-SunHiNeural"
- Add TTS toggle switch to web UI (chat input bar)
  - Toggles visibility of all 🔊 buttons and stops active playback
  - State persisted in localStorage (default: on)
- Mic input now auto-submits after speech recognition completes

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
kim
2026-05-19 21:06:10 +09:00
co-authored by Claude Sonnet 4.6
parent cd48a8ebe0
commit 15650e5798
4 changed files with 82 additions and 22 deletions
+2 -4
View File
@@ -223,10 +223,8 @@
"language": "ko" "language": "ko"
}, },
"tts": { "tts": {
"provider": "piper", "provider": "edge_tts",
"piperPath": "/usr/local/bin/piper", "voice": "ko-KR-SunHiNeural"
"modelPath": "/usr/local/share/piper-voices/ko_KR-kss-medium.onnx",
"configPath": "/usr/local/share/piper-voices/ko_KR-kss-medium.onnx.json"
} }
}, },
"search": { "search": {
+39 -18
View File
@@ -8748,27 +8748,48 @@ app.post('/api/voice/tts', async (req: express.Request, res: express.Response) =
if (!text || typeof text !== 'string') { res.status(400).json({ success: false, error: 'text required' }); return; } if (!text || typeof text !== 'string') { res.status(400).json({ success: false, error: 'text required' }); return; }
const cfg = (getConfig().getConfig() as any)?.voice; const cfg = (getConfig().getConfig() as any)?.voice;
if (!cfg?.tts) { res.status(503).json({ success: false, error: 'TTS not configured' }); return; } if (!cfg?.tts) { res.status(503).json({ success: false, error: 'TTS not configured' }); return; }
const piperPath = cfg.tts.piperPath || 'piper'; const provider = cfg.tts.provider || 'piper';
const modelPath = cfg.tts.modelPath || '';
const configPath = cfg.tts.configPath || '';
if (!modelPath) { res.status(503).json({ success: false, error: 'TTS model path not configured' }); return; }
try { try {
const os = await import('os'); const os = await import('os');
const wavPath = path.join(os.tmpdir(), `tts_${Date.now()}.wav`);
const piperArgs = ['--model', modelPath, '--output_file', wavPath];
if (configPath) piperArgs.push('--config', configPath);
const { spawn: spawnProc } = await import('child_process'); const { spawn: spawnProc } = await import('child_process');
await new Promise<void>((resolve, reject) => { if (provider === 'edge_tts') {
const piper = spawnProc(piperPath, piperArgs); const voice = cfg.tts.voice || 'ko-KR-SunHiNeural';
piper.stdin.write(text.slice(0, 2000), 'utf8'); const mp3Path = path.join(os.tmpdir(), `tts_${Date.now()}.mp3`);
piper.stdin.end(); const scriptPath = path.join(__dirname, '../../src/tools/edge_tts_synth.py');
piper.on('close', (code: number | null) => code === 0 ? resolve() : reject(new Error(`Piper exited ${code}`))); await new Promise<void>((resolve, reject) => {
piper.on('error', reject); const py = spawnProc('python3', [scriptPath, voice, mp3Path]);
}); py.stdin.write(text.slice(0, 2000), 'utf8');
const wav = fs.readFileSync(wavPath); py.stdin.end();
try { fs.unlinkSync(wavPath); } catch {} let stderr = '';
res.set('Content-Type', 'audio/wav'); py.stderr.on('data', (d: Buffer) => { stderr += d.toString(); });
res.send(wav); py.on('close', (code: number | null) => code === 0 ? resolve() : reject(new Error(`edge-tts exited ${code}: ${stderr}`)));
py.on('error', reject);
});
const mp3 = fs.readFileSync(mp3Path);
try { fs.unlinkSync(mp3Path); } catch {}
res.set('Content-Type', 'audio/mpeg');
res.send(mp3);
} else {
// piper
const piperPath = cfg.tts.piperPath || 'piper';
const modelPath = cfg.tts.modelPath || '';
const configPath = cfg.tts.configPath || '';
if (!modelPath) { res.status(503).json({ success: false, error: 'TTS model path not configured' }); return; }
const wavPath = path.join(os.tmpdir(), `tts_${Date.now()}.wav`);
const piperArgs = ['--model', modelPath, '--output_file', wavPath];
if (configPath) piperArgs.push('--config', configPath);
await new Promise<void>((resolve, reject) => {
const piper = spawnProc(piperPath, piperArgs);
piper.stdin.write(text.slice(0, 2000), 'utf8');
piper.stdin.end();
piper.on('close', (code: number | null) => code === 0 ? resolve() : reject(new Error(`Piper exited ${code}`)));
piper.on('error', reject);
});
const wav = fs.readFileSync(wavPath);
try { fs.unlinkSync(wavPath); } catch {}
res.set('Content-Type', 'audio/wav');
res.send(wav);
}
} catch (e: any) { } catch (e: any) {
res.status(500).json({ success: false, error: String(e?.message || e) }); res.status(500).json({ success: false, error: String(e?.message || e) });
} }
+15
View File
@@ -0,0 +1,15 @@
#!/usr/bin/env python3
import asyncio
import sys
import edge_tts
async def main():
text = sys.stdin.read().strip()
if not text:
sys.exit(1)
voice = sys.argv[1] if len(sys.argv) > 1 else 'ko-KR-SunHiNeural'
output_file = sys.argv[2] if len(sys.argv) > 2 else '/tmp/tts_edge_out.mp3'
communicate = edge_tts.Communicate(text, voice)
await communicate.save(output_file)
asyncio.run(main())
+26
View File
@@ -1025,6 +1025,7 @@
@keyframes mic-pulse { 0%,100%{opacity:1} 50%{opacity:0.55} } @keyframes mic-pulse { 0%,100%{opacity:1} 50%{opacity:0.55} }
.tts-btn { background:none;border:none;cursor:pointer;font-size:13px;padding:1px 4px;border-radius:4px;opacity:0.45;vertical-align:middle;line-height:1; } .tts-btn { background:none;border:none;cursor:pointer;font-size:13px;padding:1px 4px;border-radius:4px;opacity:0.45;vertical-align:middle;line-height:1; }
.tts-btn:hover,.tts-btn.playing{opacity:1;background:var(--panel-2);} .tts-btn:hover,.tts-btn.playing{opacity:1;background:var(--panel-2);}
body.tts-disabled .tts-btn { display:none !important; }
.export-dropdown { position:relative; display:inline-block; } .export-dropdown { position:relative; display:inline-block; }
.export-menu { display:none; position:absolute; right:0; top:calc(100% + 4px); background:var(--panel-2); border:1px solid var(--border); border-radius:8px; box-shadow:0 4px 16px rgba(0,0,0,0.18); z-index:2000; min-width:160px; overflow:hidden; } .export-menu { display:none; position:absolute; right:0; top:calc(100% + 4px); background:var(--panel-2); border:1px solid var(--border); border-radius:8px; box-shadow:0 4px 16px rgba(0,0,0,0.18); z-index:2000; min-width:160px; overflow:hidden; }
.export-menu button { display:block; width:100%; padding:8px 14px; text-align:left; background:none; border:none; cursor:pointer; font-size:12px; color:var(--text); white-space:nowrap; } .export-menu button { display:block; width:100%; padding:8px 14px; text-align:left; background:none; border:none; cursor:pointer; font-size:12px; color:var(--text); white-space:nowrap; }
@@ -2374,6 +2375,12 @@
</label> </label>
<span class="agent-toggle-label" id="agent-mode-label">채팅</span> <span class="agent-toggle-label" id="agent-mode-label">채팅</span>
<span class="agent-mode-badge" id="agent-mode-badge">AGENT</span> <span class="agent-mode-badge" id="agent-mode-badge">AGENT</span>
<span style="color:var(--line);margin:0 6px;user-select:none">|</span>
<span class="agent-toggle-label">TTS</span>
<label class="toggle-switch" title="AI 응답 음성 읽기 켜기/끄기">
<input type="checkbox" id="tts-mode-toggle" onchange="updateTTSMode()">
<span class="toggle-track"></span>
</label>
<div class="agent-toggle-right" style="display:flex;align-items:center;gap:6px"> <div class="agent-toggle-right" style="display:flex;align-items:center;gap:6px">
<button class="quick-mode-btn" id="context-pin-btn" onclick="toggleContextPinMode()" title="메시지를 컨텍스트에 고정"> <button class="quick-mode-btn" id="context-pin-btn" onclick="toggleContextPinMode()" title="메시지를 컨텍스트에 고정">
<svg width="14" height="14" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.5" stroke-linecap="round" stroke-linejoin="round"><path d="M12 17v5"/><path d="M9 10.76a2 2 0 0 1-1.11 1.79l-1.78.9A2 2 0 0 0 5 15.24V16a1 1 0 0 0 1 1h12a1 1 0 0 0 1-1v-.76a2 2 0 0 0-1.11-1.79l-1.78-.9A2 2 0 0 1 15 10.76V7a1 1 0 0 1 1-1 2 2 0 0 0 0-4H8a2 2 0 0 0 0 4 1 1 0 0 1 1 1z"/></svg> <svg width="14" height="14" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2.5" stroke-linecap="round" stroke-linejoin="round"><path d="M12 17v5"/><path d="M9 10.76a2 2 0 0 1-1.11 1.79l-1.78.9A2 2 0 0 0 5 15.24V16a1 1 0 0 0 1 1h12a1 1 0 0 0 1-1v-.76a2 2 0 0 0-1.11-1.79l-1.78-.9A2 2 0 0 1 15 10.76V7a1 1 0 0 1 1-1 2 2 0 0 0 0-4H8a2 2 0 0 0 0 4 1 1 0 0 1 1 1z"/></svg>
@@ -3946,6 +3953,7 @@ let agentSessionId = '';
let currentUser = null; let currentUser = null;
setAgentSessionId(generateSessionId()); // always fresh session on page load setAgentSessionId(generateSessionId()); // always fresh session on page load
checkAuth(); // Check authentication on page load checkAuth(); // Check authentication on page load
initTTSToggle();
// ── Server log panel ────────────────────────────────────────────────────────── // ── Server log panel ──────────────────────────────────────────────────────────
let logScrollLocked = true; let logScrollLocked = true;
@@ -8866,6 +8874,7 @@ function startMicRecording() {
input.value = (input.value ? input.value + ' ' : '') + transcript; input.value = (input.value ? input.value + ' ' : '') + transcript;
input.focus(); input.focus();
input.dispatchEvent(new Event('input')); input.dispatchEvent(new Event('input'));
setTimeout(function() { handleSendStop(); }, 50);
} }
}; };
_speechRecog.onerror = function(e) { _speechRecog.onerror = function(e) {
@@ -8884,9 +8893,26 @@ function stopMicRecording() {
} }
// ── Voice: TTS (음성 출력) ─────────────────────────────────────────────── // ── Voice: TTS (음성 출력) ───────────────────────────────────────────────
var _ttsEnabled = localStorage.getItem('ttsEnabled') !== 'false'; // default on
var _ttsAudio = null; var _ttsAudio = null;
var _ttsBtnActive = null; var _ttsBtnActive = null;
function updateTTSMode() {
_ttsEnabled = document.getElementById('tts-mode-toggle').checked;
localStorage.setItem('ttsEnabled', String(_ttsEnabled));
document.body.classList.toggle('tts-disabled', !_ttsEnabled);
if (!_ttsEnabled && _ttsAudio) {
_ttsAudio.pause(); _ttsAudio = null;
if (_ttsBtnActive) { _ttsBtnActive.textContent = '🔊'; _ttsBtnActive.classList.remove('playing'); _ttsBtnActive = null; }
}
}
function initTTSToggle() {
var toggle = document.getElementById('tts-mode-toggle');
if (toggle) toggle.checked = _ttsEnabled;
document.body.classList.toggle('tts-disabled', !_ttsEnabled);
}
function playTTSMsg(btn, idx) { function playTTSMsg(btn, idx) {
var msg = chatHistory[idx]; var msg = chatHistory[idx];
if (!msg) return; if (!msg) return;