diff --git a/.smallclaw/config.json b/.smallclaw/config.json index 08f0cb9..27f3aac 100644 --- a/.smallclaw/config.json +++ b/.smallclaw/config.json @@ -223,10 +223,8 @@ "language": "ko" }, "tts": { - "provider": "piper", - "piperPath": "/usr/local/bin/piper", - "modelPath": "/usr/local/share/piper-voices/ko_KR-kss-medium.onnx", - "configPath": "/usr/local/share/piper-voices/ko_KR-kss-medium.onnx.json" + "provider": "edge_tts", + "voice": "ko-KR-SunHiNeural" } }, "search": { diff --git a/src/gateway/server-v2.ts b/src/gateway/server-v2.ts index 9812110..538d9c8 100644 --- a/src/gateway/server-v2.ts +++ b/src/gateway/server-v2.ts @@ -8748,27 +8748,48 @@ app.post('/api/voice/tts', async (req: express.Request, res: express.Response) = if (!text || typeof text !== 'string') { res.status(400).json({ success: false, error: 'text required' }); return; } const cfg = (getConfig().getConfig() as any)?.voice; if (!cfg?.tts) { res.status(503).json({ success: false, error: 'TTS not configured' }); return; } - const piperPath = cfg.tts.piperPath || 'piper'; - const modelPath = cfg.tts.modelPath || ''; - const configPath = cfg.tts.configPath || ''; - if (!modelPath) { res.status(503).json({ success: false, error: 'TTS model path not configured' }); return; } + const provider = cfg.tts.provider || 'piper'; try { const os = await import('os'); - const wavPath = path.join(os.tmpdir(), `tts_${Date.now()}.wav`); - const piperArgs = ['--model', modelPath, '--output_file', wavPath]; - if (configPath) piperArgs.push('--config', configPath); const { spawn: spawnProc } = await import('child_process'); - await new Promise((resolve, reject) => { - const piper = spawnProc(piperPath, piperArgs); - piper.stdin.write(text.slice(0, 2000), 'utf8'); - piper.stdin.end(); - piper.on('close', (code: number | null) => code === 0 ? resolve() : reject(new Error(`Piper exited ${code}`))); - piper.on('error', reject); - }); - const wav = fs.readFileSync(wavPath); - try { fs.unlinkSync(wavPath); } catch {} - res.set('Content-Type', 'audio/wav'); - res.send(wav); + if (provider === 'edge_tts') { + const voice = cfg.tts.voice || 'ko-KR-SunHiNeural'; + const mp3Path = path.join(os.tmpdir(), `tts_${Date.now()}.mp3`); + const scriptPath = path.join(__dirname, '../../src/tools/edge_tts_synth.py'); + await new Promise((resolve, reject) => { + const py = spawnProc('python3', [scriptPath, voice, mp3Path]); + py.stdin.write(text.slice(0, 2000), 'utf8'); + py.stdin.end(); + let stderr = ''; + py.stderr.on('data', (d: Buffer) => { stderr += d.toString(); }); + py.on('close', (code: number | null) => code === 0 ? resolve() : reject(new Error(`edge-tts exited ${code}: ${stderr}`))); + py.on('error', reject); + }); + const mp3 = fs.readFileSync(mp3Path); + try { fs.unlinkSync(mp3Path); } catch {} + res.set('Content-Type', 'audio/mpeg'); + res.send(mp3); + } else { + // piper + const piperPath = cfg.tts.piperPath || 'piper'; + const modelPath = cfg.tts.modelPath || ''; + const configPath = cfg.tts.configPath || ''; + if (!modelPath) { res.status(503).json({ success: false, error: 'TTS model path not configured' }); return; } + const wavPath = path.join(os.tmpdir(), `tts_${Date.now()}.wav`); + const piperArgs = ['--model', modelPath, '--output_file', wavPath]; + if (configPath) piperArgs.push('--config', configPath); + await new Promise((resolve, reject) => { + const piper = spawnProc(piperPath, piperArgs); + piper.stdin.write(text.slice(0, 2000), 'utf8'); + piper.stdin.end(); + piper.on('close', (code: number | null) => code === 0 ? resolve() : reject(new Error(`Piper exited ${code}`))); + piper.on('error', reject); + }); + const wav = fs.readFileSync(wavPath); + try { fs.unlinkSync(wavPath); } catch {} + res.set('Content-Type', 'audio/wav'); + res.send(wav); + } } catch (e: any) { res.status(500).json({ success: false, error: String(e?.message || e) }); } diff --git a/src/tools/edge_tts_synth.py b/src/tools/edge_tts_synth.py new file mode 100644 index 0000000..68d4421 --- /dev/null +++ b/src/tools/edge_tts_synth.py @@ -0,0 +1,15 @@ +#!/usr/bin/env python3 +import asyncio +import sys +import edge_tts + +async def main(): + text = sys.stdin.read().strip() + if not text: + sys.exit(1) + voice = sys.argv[1] if len(sys.argv) > 1 else 'ko-KR-SunHiNeural' + output_file = sys.argv[2] if len(sys.argv) > 2 else '/tmp/tts_edge_out.mp3' + communicate = edge_tts.Communicate(text, voice) + await communicate.save(output_file) + +asyncio.run(main()) diff --git a/web-ui/index.html b/web-ui/index.html index fda94f9..1669030 100644 --- a/web-ui/index.html +++ b/web-ui/index.html @@ -1025,6 +1025,7 @@ @keyframes mic-pulse { 0%,100%{opacity:1} 50%{opacity:0.55} } .tts-btn { background:none;border:none;cursor:pointer;font-size:13px;padding:1px 4px;border-radius:4px;opacity:0.45;vertical-align:middle;line-height:1; } .tts-btn:hover,.tts-btn.playing{opacity:1;background:var(--panel-2);} + body.tts-disabled .tts-btn { display:none !important; } .export-dropdown { position:relative; display:inline-block; } .export-menu { display:none; position:absolute; right:0; top:calc(100% + 4px); background:var(--panel-2); border:1px solid var(--border); border-radius:8px; box-shadow:0 4px 16px rgba(0,0,0,0.18); z-index:2000; min-width:160px; overflow:hidden; } .export-menu button { display:block; width:100%; padding:8px 14px; text-align:left; background:none; border:none; cursor:pointer; font-size:12px; color:var(--text); white-space:nowrap; } @@ -2374,6 +2375,12 @@ 채팅 AGENT + | + TTS +