Merge #4016 (re-arm browser voice mode when speechSynthesis drops, #3983) onto master

This commit is contained in:
nesquena-hermes
2026-06-13 07:37:31 +00:00
2 changed files with 140 additions and 2 deletions

View File

@@ -853,8 +853,48 @@ window._micPendingSend=window._micPendingSend||false;
// a different session's last assistant reply if the user navigated away // a different session's last assistant reply if the user navigated away
// between send and stream completion. (Opus pre-release advisor.) // between send and stream completion. (Opus pre-release advisor.)
let _voiceModeThinkingSid=null; let _voiceModeThinkingSid=null;
let _browserTtsKeepAlive=null;
let _browserTtsWatchdog=null;
let _browserTtsSuppressNextErrorRearm=false;
const SILENCE_MS=1800; // auto-send after 1.8s silence const SILENCE_MS=1800; // auto-send after 1.8s silence
function _clearBrowserTtsRecovery(){
if(_browserTtsKeepAlive){
clearInterval(_browserTtsKeepAlive);
_browserTtsKeepAlive=null;
}
if(_browserTtsWatchdog){
clearTimeout(_browserTtsWatchdog);
_browserTtsWatchdog=null;
}
}
function _armBrowserTtsRecovery(clean, rate){
_clearBrowserTtsRecovery();
_browserTtsSuppressNextErrorRearm=false;
const safeRate=(Number.isFinite(rate)&&rate>0)?rate:1;
// Chromium can drop utter.onend on later turns, so force a recovery path.
const watchdogMs=Math.max(4000,Math.round((String(clean||'').length/(12*safeRate))*1000)+10000);
_browserTtsWatchdog=setTimeout(()=>{
if(!_voiceModeActive||_voiceModeState!=='speaking') return;
_browserTtsSuppressNextErrorRearm=true;
try{ speechSynthesis.cancel(); }catch(_){}
_clearBrowserTtsRecovery();
_startListening();
},watchdogMs);
_browserTtsKeepAlive=setInterval(()=>{
if(!_voiceModeActive||_voiceModeState!=='speaking'){
_clearBrowserTtsRecovery();
return;
}
if(!speechSynthesis.speaking) return;
try{
speechSynthesis.pause();
speechSynthesis.resume();
}catch(_){}
},10000);
}
function _setState(state){ function _setState(state){
_voiceModeState=state; _voiceModeState=state;
indicator.className='voice-mode-indicator '+state; indicator.className='voice-mode-indicator '+state;
@@ -867,6 +907,7 @@ window._micPendingSend=window._micPendingSend||false;
function _startListening(){ function _startListening(){
if(!_voiceModeActive) return; if(!_voiceModeActive) return;
_clearBrowserTtsRecovery();
_setState('listening'); _setState('listening');
_recognition=new SpeechRecognition(); _recognition=new SpeechRecognition();
@@ -1057,14 +1098,27 @@ window._micPendingSend=window._micPendingSend||false;
if(!isNaN(savedPitch)) utter.pitch=Math.min(2,Math.max(0,savedPitch)); if(!isNaN(savedPitch)) utter.pitch=Math.min(2,Math.max(0,savedPitch));
utter.onend=()=>{ utter.onend=()=>{
_browserTtsSuppressNextErrorRearm=false;
_clearBrowserTtsRecovery();
// After speaking, go back to listening // After speaking, go back to listening
if(_voiceModeActive) setTimeout(()=>_startListening(),500); if(_voiceModeActive&&_voiceModeState==='speaking') setTimeout(()=>_startListening(),500);
}; };
utter.onerror=()=>{ utter.onerror=()=>{
_clearBrowserTtsRecovery();
if(_browserTtsSuppressNextErrorRearm){
_browserTtsSuppressNextErrorRearm=false;
return;
}
if(_voiceModeActive) setTimeout(()=>_startListening(),1000); if(_voiceModeActive) setTimeout(()=>_startListening(),1000);
}; };
speechSynthesis.speak(utter); _armBrowserTtsRecovery(clean, utter.rate);
try{
speechSynthesis.speak(utter);
}catch(_){
_clearBrowserTtsRecovery();
if(_voiceModeActive) setTimeout(()=>_startListening(),1000);
}
} }
// Hook into response completion — observe when the agent finishes // Hook into response completion — observe when the agent finishes
@@ -1121,10 +1175,12 @@ window._micPendingSend=window._micPendingSend||false;
_voiceModeActive=false; _voiceModeActive=false;
_voiceModeState='idle'; _voiceModeState='idle';
_voiceModeThinkingSid=null; _voiceModeThinkingSid=null;
_browserTtsSuppressNextErrorRearm=false;
modeBtn.classList.remove('active'); modeBtn.classList.remove('active');
_setButtonTooltip(modeBtn, t('voice_mode_toggle')); _setButtonTooltip(modeBtn, t('voice_mode_toggle'));
bar.style.display='none'; bar.style.display='none';
clearTimeout(_silenceTimer); clearTimeout(_silenceTimer);
_clearBrowserTtsRecovery();
try{ if(_recognition) _recognition.abort(); }catch(_){} try{ if(_recognition) _recognition.abort(); }catch(_){}
_recognition=null; _recognition=null;
if(typeof stopTTS==='function') stopTTS(); if(typeof stopTTS==='function') stopTTS();

View File

@@ -0,0 +1,82 @@
from pathlib import Path
import re
REPO = Path(__file__).resolve().parents[1]
def _extract_function(src: str, name: str) -> str:
anchor = f"function {name}("
start = src.find(anchor)
assert start != -1, f"{name}() must exist"
body_start = src.find("{", start)
assert body_start != -1, f"{name}() must have a body"
depth = 1
idx = body_start + 1
while depth and idx < len(src):
if src[idx] == "{":
depth += 1
elif src[idx] == "}":
depth -= 1
idx += 1
assert depth == 0, f"{name}() body must balance braces"
return src[start:idx]
def test_boot_js_declares_browser_tts_recovery_helpers():
src = (REPO / "static" / "boot.js").read_text(encoding="utf-8")
assert "let _browserTtsKeepAlive=null;" in src
assert "let _browserTtsWatchdog=null;" in src
assert "let _browserTtsSuppressNextErrorRearm=false;" in src
assert "function _clearBrowserTtsRecovery()" in src
assert "function _armBrowserTtsRecovery(clean, rate)" in src
def test_browser_tts_watchdog_rearms_listening_if_onend_drops():
src = (REPO / "static" / "boot.js").read_text(encoding="utf-8")
arm_body = _extract_function(src, "_armBrowserTtsRecovery")
assert "_browserTtsWatchdog=setTimeout" in arm_body
assert "_voiceModeState!=='speaking'" in arm_body
assert "_browserTtsSuppressNextErrorRearm=true;" in arm_body
assert "speechSynthesis.cancel()" in arm_body
assert "_startListening();" in arm_body
assert "_browserTtsKeepAlive=setInterval" in arm_body
assert "speechSynthesis.pause();" in arm_body
assert "speechSynthesis.resume();" in arm_body
def test_browser_tts_callbacks_and_deactivate_clear_recovery_handles():
src = (REPO / "static" / "boot.js").read_text(encoding="utf-8")
speak_body = _extract_function(src, "_speakResponse")
assert "const utter=new SpeechSynthesisUtterance(clean);" in speak_body
assert "utter.onend=()=>{" in speak_body
assert "utter.onerror=()=>{" in speak_body
assert speak_body.count("_clearBrowserTtsRecovery();") >= 2, (
"Both browser TTS completion callbacks must clear watchdog/keep-alive handles."
)
assert "_browserTtsSuppressNextErrorRearm=false;" in speak_body
assert "_voiceModeActive&&_voiceModeState==='speaking'" in speak_body
assert "if(_browserTtsSuppressNextErrorRearm){" in speak_body
assert "_armBrowserTtsRecovery(clean, utter.rate);" in speak_body
deactivate_body = _extract_function(src, "_deactivate")
assert "_clearBrowserTtsRecovery();" in deactivate_body, (
"_deactivate() must clear browser TTS watchdog/keep-alive handles."
)
assert "_browserTtsSuppressNextErrorRearm=false;" in deactivate_body
def test_edge_audio_branch_stays_separate():
src = (REPO / "static" / "boot.js").read_text(encoding="utf-8")
edge_match = re.search(
r'if\(engine==="edge"\)\{(.*?)\n\s+return;\n\s+\}',
src,
re.DOTALL,
)
assert edge_match, "Edge audio branch must exist"
edge_body = edge_match.group(1)
assert "const audio = new Audio(url);" in edge_body
assert "audio.onended = () => {" in edge_body
assert "_armBrowserTtsRecovery" not in edge_body, (
"The browser speechSynthesis workaround must not be injected into the Edge audio branch."
)