diff --git a/src/tools/voice_engine.py b/src/tools/voice_engine.py
index 9bd1b12..b2a8863 100644
--- a/src/tools/voice_engine.py
+++ b/src/tools/voice_engine.py
@@ -119,7 +119,9 @@ class Engine:
f.write(audio_bytes)
path = f.name
try:
- segments, info = self.stt_model.transcribe(path, language=language or None, beam_size=5, vad_filter=True)
+ segments, info = self.stt_model.transcribe(
+ path, language=language or None, beam_size=5, vad_filter=True, condition_on_previous_text=False,
+ )
text = ''.join(seg.text for seg in segments).strip()
return text, (info.language if info else language)
finally:
@@ -140,7 +142,9 @@ class Engine:
# ── Streaming STT (partial decode of an accumulating PCM buffer) ──────
def transcribe_pcm(self, pcm: bytes, language: str):
audio = pcm16_bytes_to_float(pcm)
- segments, info = self.stt_model.transcribe(audio, language=language or None, beam_size=1, vad_filter=True)
+ segments, info = self.stt_model.transcribe(
+ audio, language=language or None, beam_size=1, vad_filter=True, condition_on_previous_text=False,
+ )
text = ''.join(seg.text for seg in segments).strip()
return text, (info.language if info else language)
diff --git a/web-ui/app.js b/web-ui/app.js
index 3730f4e..4793b76 100644
--- a/web-ui/app.js
+++ b/web-ui/app.js
@@ -1561,6 +1561,7 @@ async function sendChat(queuedMessage = null) {
const token = event.text || event.content || event.delta || '';
if (token) {
partialContent += token;
+ if (window._voiceCallOnToken) window._voiceCallOnToken(partialContent);
}
break;
}
@@ -1813,6 +1814,11 @@ async function sendChat(queuedMessage = null) {
if (s.finalAnswer) {
addProcessEntry('final', s.finalAnswer);
partialContent = s.finalAnswer; // track for stop
+ // partialContent just got replaced wholesale (not appended to), so
+ // any voice-call spokenUpTo offset computed against the old string
+ // is now meaningless — reset before feeding the new text through.
+ if (window._voiceCallResetSpoken) window._voiceCallResetSpoken();
+ if (window._voiceCallOnToken) window._voiceCallOnToken(partialContent);
}
break;
}
@@ -1903,6 +1909,8 @@ async function sendChat(queuedMessage = null) {
}
}
+ if (window._voiceCallOnTurnDone) window._voiceCallOnTurnDone(finalReply || partialContent);
+
if (finalReply) {
// Append file cards for any attachments sent via the 'files' SSE event
if (turnFileLinks.length > 0) {
diff --git a/web-ui/index.html b/web-ui/index.html
index f04d990..6bba984 100644
--- a/web-ui/index.html
+++ b/web-ui/index.html
@@ -387,10 +387,18 @@
+
+ 통화 연결 중…
+
+ 🔈
+
+
+
+
+
+
fail2ban 차단 IP
+
+
+
+
@@ -1914,6 +1931,7 @@
+
diff --git a/web-ui/voice-call.js b/web-ui/voice-call.js
index 6be83aa..75a56fb 100644
--- a/web-ui/voice-call.js
+++ b/web-ui/voice-call.js
@@ -19,7 +19,7 @@
let ws = null;
let captureCtx = null, playCtx = null;
let micStream = null;
- let captureNode = null, playerNode = null;
+ let captureNode = null, playerNode = null, gainNode = null;
let state = 'idle'; // idle | listening | user_speaking | processing | assistant_speaking
let speechChunks = 0;
let silenceMs = 0;
@@ -62,6 +62,13 @@
else startVoiceCall();
};
+ window.voiceCallSetVolume = function (v) {
+ const vol = parseFloat(v);
+ if (!Number.isFinite(vol)) return;
+ localStorage.setItem('voiceCallVolume', String(vol));
+ if (gainNode) gainNode.gain.value = vol;
+ };
+
window.voiceCallHangup = function () {
active = false;
state = 'idle';
@@ -70,7 +77,7 @@
try { micStream && micStream.getTracks().forEach(t => t.stop()); } catch {}
try { captureCtx && captureCtx.close(); } catch {}
try { playCtx && playCtx.close(); } catch {}
- captureCtx = playCtx = micStream = captureNode = playerNode = null;
+ captureCtx = playCtx = micStream = captureNode = playerNode = gainNode = null;
try { wakeLock && wakeLock.release(); } catch {}
wakeLock = null;
ttsQueue = []; ttsBusy = false; currentTtsId = null;
@@ -110,7 +117,16 @@
playCtx = new (window.AudioContext || window.webkitAudioContext)({ sampleRate: 24000 });
await playCtx.audioWorklet.addModule('voice-call-worklets.js');
playerNode = new AudioWorkletNode(playCtx, 'pcm-player-processor');
- playerNode.connect(playCtx.destination);
+ // A GainNode we control from the in-app slider — on some mobile browsers
+ // the hardware volume rocker maps to the call/mic audio session while a
+ // getUserMedia stream is open, not to this Web Audio output, so the phone's
+ // own volume slider doesn't reliably affect playback here.
+ gainNode = playCtx.createGain();
+ const savedVolume = parseFloat(localStorage.getItem('voiceCallVolume') || '1');
+ gainNode.gain.value = savedVolume;
+ const volumeSlider = document.getElementById('voice-call-volume');
+ if (volumeSlider) volumeSlider.value = String(savedVolume);
+ playerNode.connect(gainNode).connect(playCtx.destination);
const proto = location.protocol === 'https:' ? 'wss' : 'ws';
const token = getAuthToken();
@@ -298,6 +314,13 @@
return { sentences, consumedLength: start };
}
+ // Hooked from app.js when partialContent is replaced wholesale rather than
+ // appended to (e.g. a multi-step tool turn's finalAnswer) — spokenUpTo was an
+ // offset into the old string and must be dropped before the new text arrives.
+ window._voiceCallResetSpoken = function () {
+ spokenUpTo = 0;
+ };
+
// Hooked from app.js's SSE token handler — only acts while a call is active.
window._voiceCallOnToken = function (fullPartialContent) {
if (!active) return;