From d7e0821f0cf064acd458b1994620fd106249a129 Mon Sep 17 00:00:00 2001 From: NetDevAutomate Date: Tue, 29 Sep 2026 09:51:01 +0100 Subject: [PATCH 1/5] test(web): RED -- server voice is fetched but never heard Reported: the header badge says "Kokoro (server)", nothing is ever heard, and OpenVox appears to see nothing. Reproduced in Chromium against the real local OpenVox by clicking the voice toggle: /api/tts/health, /warm and /speak all answered 200 (speak with audio/wav), then the browser refused the audio: Loading media from 'blob:http://127.0.0.1:.../...' violates the following Content Security Policy directive: "default-src 'self'". Note that 'media-src' was not explicitly set, so 'default-src' is used as a fallback. play() rejected with NotSupportedError and the engine swallowed it. The same page with the policy bypassed played to the end. The server tier plays the host's WAV through URL.createObjectURL (since adf74933, 23 Aug); the policy (`default-src 'self'`, no media-src) arrived on 2 Sep in 1f10f059, and 'self' does not match a blob: URL. Why nothing caught it: the only browser test of this path needs a running OpenVox, so it skips in CI (which has none), and its "played" check counted play() calls rather than playback. RED, each for the stated reason: - the CSP has no media-src allowing blob:, in both modes (2 unit tests); - with the host's answers faked at the network edge, so CI runs it, the "Voice enabled" audio is refused by the page's own policy; - a play() the page refuses produces no notice, so the learner cannot tell a refused audio from a dead server; - a host that answers health in 4s is treated as absent. Measured here: OpenVox's /v1/models takes 1.2-2.7s, so 5 of 8 /api/tts/health calls ran past the page's 2.5s probe budget, and 1 page load in 5 fell back to system voices with a healthy host; - the real-OpenVox browser test now asserts the audio played (it fails on the same policy refusal). Guard that passes now and must keep passing: blob: appears in no directive but media-src (2 unit tests). --- .../studyloop/tests/e2e/test_server_tts.py | 218 ++++++++++++++++-- packages/studyloop/tests/test_web_app.py | 48 ++++ 2 files changed, 251 insertions(+), 15 deletions(-) diff --git a/packages/studyloop/tests/e2e/test_server_tts.py b/packages/studyloop/tests/e2e/test_server_tts.py index e997f43eb..90a3986b5 100644 --- a/packages/studyloop/tests/e2e/test_server_tts.py +++ b/packages/studyloop/tests/e2e/test_server_tts.py @@ -17,13 +17,20 @@ from __future__ import annotations +import io import sys +import time +import wave +from contextlib import suppress from pathlib import Path from typing import TYPE_CHECKING import pytest pytest.importorskip("requests") +pytest.importorskip("playwright") + +from playwright.sync_api import Error as PlaywrightError _tests_dir = str(Path(__file__).resolve().parent.parent) if _tests_dir not in sys.path: @@ -32,7 +39,7 @@ from e2e._env import ConsoleWatch, launch_env, shutdown # noqa: E402 if TYPE_CHECKING: - from playwright.sync_api import Browser + from playwright.sync_api import Browser, Page pytestmark = [pytest.mark.e2e] @@ -158,6 +165,189 @@ def test_warm_is_idempotent(env) -> None: assert response.json()["warmed"] is True +# --------------------------------------------------------------------------- +# Playback probes — did the learner HEAR it, not just did the engine ask? +# --------------------------------------------------------------------------- + +#: Installed before any app script: records how every media element's play() +#: ended and every CSP violation, so a test can tell "the engine asked for audio" +#: apart from "the audio played". +_PLAYBACK_PROBE = """ +window.__plays = []; +window.__cspViolations = []; +document.addEventListener('securitypolicyviolation', (e) => { + window.__cspViolations.push(`${e.effectiveDirective} ${e.blockedURI}`); +}); +const __play = HTMLMediaElement.prototype.play; +HTMLMediaElement.prototype.play = function () { + const record = { outcome: 'pending', playing: false }; + window.__plays.push(record); + this.addEventListener('playing', () => { record.playing = true; }); + const promise = __play.call(this); + promise.then( + () => { record.outcome = 'resolved'; }, + (err) => { record.outcome = `${err.name}: ${err.message}`; }, + ); + return promise; +}; +""" + +_PLAYS_SETTLED = ( + "() => window.__plays.length > 0 && window.__plays.every((p) => p.outcome !== 'pending')" +) + + +def _enable_voice(page: Page, base_url: str, *, timeout_ms: int) -> dict: + """Click the header voice toggle, as the learner does, and report how its + "Voice enabled" utterance ended. + + A real click rather than page.evaluate: it is the learner's path, and it + gives the page the user activation a browser requires before audio plays. + """ + page.goto(f"{base_url}/", wait_until="domcontentloaded", timeout=30000) + page.wait_for_function( + "() => !!window.Alpine && !!window.Alpine.store('settings')", timeout=15000 + ) + page.click("button[title^='Toggle voice']") + page.wait_for_function( + "() => !!window.ttsEngine && window.ttsEngine.tier !== null", timeout=timeout_ms + ) + # Only the server tier plays through a media element; on any other tier + # there is nothing to wait for, and the caller's tier assertion says why. + if page.evaluate("() => window.ttsEngine.tier") == "server-openvox": + page.wait_for_function(_PLAYS_SETTLED, timeout=timeout_ms) + return page.evaluate( + "() => ({ plays: window.__plays, csp: window.__cspViolations," + " tier: window.ttsEngine.tier })" + ) + + +#: The host's answers, faked at the network edge so the next two tests need no +#: OpenVox and therefore run in CI. Only /api/tts/* is faked: the page, its real +#: Content-Security-Policy header and the real tts-engine.js come from the server. +_FAKE_HEALTH = { + "available": True, + "model": "kokoro", + "voice_count": 1, + "detail": "", + "voices": [{"id": "bf_emma", "language": "en-gb", "british": True}], +} + + +def _silent_wav(seconds: float = 0.2, rate: int = 24000) -> bytes: + """A real, decodable WAV in the shape OpenVox returns: 24 kHz, 16-bit, mono.""" + buffer = io.BytesIO() + with wave.open(buffer, "wb") as out: + out.setnchannels(1) + out.setsampwidth(2) + out.setframerate(rate) + out.writeframes(b"\x00\x00" * int(seconds * rate)) + return buffer.getvalue() + + +def _fake_host(page: Page, *, audio: bytes, health_delay_s: float = 0.0) -> None: + def health(route) -> None: + if health_delay_s: + time.sleep(health_delay_s) # the page's fetch waits; nothing else runs meanwhile + # The page may have given up and aborted the fetch while we slept. + with suppress(PlaywrightError): + route.fulfill(json=_FAKE_HEALTH) + + page.route("**/api/tts/health", health) + page.route("**/api/tts/warm", lambda route: route.fulfill(json={"warmed": True})) + page.route( + "**/api/tts/speak", + lambda route: route.fulfill(status=200, content_type="audio/wav", body=audio), + ) + + +def test_host_audio_plays_under_the_pages_own_policy(browser: Browser, env) -> None: + """The host's audio must PLAY in the page, not merely be fetched. + + Regression: `default-src 'self'` (2 Sep) with no media-src refused the blob: + URL every server-tier utterance plays through. /api/tts/speak answered 200 + audio/wav, play() rejected, and the learner heard nothing while the badge + said "Kokoro (server)". The real-engine test below caught that only while + OpenVox was running, and so never in CI, which has none. This one fakes the + host's answers instead, so CI runs it. + """ + ctx = browser.new_context(viewport={"width": 1440, "height": 900}) + page = ctx.new_page() + page.add_init_script(_PLAYBACK_PROBE) + _fake_host(page, audio=_silent_wav()) + watch = ConsoleWatch(page) + try: + state = _enable_voice(page, env.base_url, timeout_ms=20000) + + assert state["tier"] == "server-openvox" + assert state["csp"] == [], f"the page's own policy refused: {state['csp']}" + assert [p["outcome"] for p in state["plays"]] == ["resolved"], state["plays"] + assert state["plays"][0]["playing"], "play() resolved but playback never started" + watch.assert_clean("playing the host's audio") + finally: + ctx.close() + + +def test_audio_the_page_cannot_play_is_reported_not_silent(browser: Browser, env) -> None: + """When the host answers but the browser will not play it, the learner is told. + + The engine resolved a refused play() silently, which is why the regression + above looked like a dead speech server: the badge stayed green, OpenVox had + answered every request, and the only trace was one console line. Bytes no + browser can decode stand in for any refusal: a policy, a codec, a corrupt + response. + """ + ctx = browser.new_context(viewport={"width": 1440, "height": 900}) + page = ctx.new_page() + page.add_init_script(_PLAYBACK_PROBE) + _fake_host(page, audio=b"this is not audio " * 64) + try: + state = _enable_voice(page, env.base_url, timeout_ms=20000) + assert state["plays"][0]["outcome"] != "resolved", "undecodable bytes played?" + + toast = _toast_text(page, timeout_ms=5000) + assert "would not play" in toast, f"the refusal was silent (toast: {toast!r})" + # A failure notice has to last long enough to read; the store's 2s + # default is gone before a sentence is finished. + page.wait_for_timeout(3000) + assert page.evaluate("() => window.Alpine.store('toast').visible"), ( + "the failure notice vanished before it could be read" + ) + finally: + ctx.close() + + +def test_a_slow_but_alive_host_still_gets_the_server_tier(browser: Browser, env) -> None: + """A host that answers in a few seconds is present, not absent. + + Measured on the developer's Mac (29 Sep): OpenVox's own /v1/models takes 1.2 + to 2.7s, so /api/tts/health does too, and the page gave it 2.5s. Five of eight + health calls ran over, and one page load in five fell back to system voices + with a healthy host -- a badge that flips between loads for no visible reason. + Four seconds is over the old budget and well inside a sane one. + """ + ctx = browser.new_context(viewport={"width": 1440, "height": 900}) + page = ctx.new_page() + page.add_init_script(_PLAYBACK_PROBE) + _fake_host(page, audio=_silent_wav(), health_delay_s=4.0) + try: + state = _enable_voice(page, env.base_url, timeout_ms=20000) + assert state["tier"] == "server-openvox", ( + f"a host answering in 4s was treated as absent (tier {state['tier']!r})" + ) + finally: + ctx.close() + + +def _toast_text(page: Page, *, timeout_ms: int) -> str: + """The toast's message once one appears, or '' if none does in time.""" + from playwright.sync_api import TimeoutError as PlaywrightTimeoutError + + with suppress(PlaywrightTimeoutError): + page.wait_for_function("() => !!window.Alpine.store('toast').message", timeout=timeout_ms) + return page.evaluate("() => window.Alpine.store('toast').message || ''") + + # --------------------------------------------------------------------------- # Browser leg — the reason this path exists # --------------------------------------------------------------------------- @@ -172,30 +362,28 @@ def test_browser_uses_the_server_tier_without_touching_webgpu(browser: Browser, needs none of them. A tablet on `--lan` is served over plain HTTP, so it is not a secure context and cannot have the first path -- if this assertion ever fails, the tablet has silently lost its voice. + + It asserts that the audio PLAYED. It used to count play() calls, which kept + passing while the page's own policy refused every one (see + test_host_audio_plays_under_the_pages_own_policy below). """ if not _health(env)["available"]: pytest.skip("no OpenVox on this host") ctx = browser.new_context(viewport={"width": 1440, "height": 900}) page = ctx.new_page() + page.add_init_script(_PLAYBACK_PROBE) watch = ConsoleWatch(page) try: - page.goto(f"{env.base_url}/", wait_until="domcontentloaded", timeout=30000) - page.wait_for_timeout(1500) - page.evaluate( - """() => { - window.__played = 0; - const orig = Audio.prototype.play; - Audio.prototype.play = function () { window.__played++; return orig.call(this); }; - }""" - ) - page.evaluate("async () => { await window.ttsEngine.init(); }") + # A cold model costs ~51s on its first utterance, so allow for one. + state = _enable_voice(page, env.base_url, timeout_ms=120000) - assert page.evaluate("() => window.ttsEngine.tier") == "server-openvox" + assert state["tier"] == "server-openvox" assert page.evaluate("() => window.ttsEngine.listVoices().length") > 0 - - page.evaluate("async () => { await window.ttsEngine.speak('Hello from the host.'); }") - assert page.evaluate("() => window.__played") >= 1, "no audio element ever played" + assert state["csp"] == [], f"the page's own policy refused: {state['csp']}" + assert [p["outcome"] for p in state["plays"]] == ["resolved"], ( + f"the host's audio never played: {state['plays']}" + ) assert page.evaluate("() => window.ttsEngine._audioCtx ? 'created' : 'never'") == "never", ( "the server tier should never build an AudioContext" ) diff --git a/packages/studyloop/tests/test_web_app.py b/packages/studyloop/tests/test_web_app.py index 9fc432a94..068b6104e 100644 --- a/packages/studyloop/tests/test_web_app.py +++ b/packages/studyloop/tests/test_web_app.py @@ -196,6 +196,54 @@ def test_data_exception_present_when_dev_mode_is_on(self) -> None: assert "data:" in connect_src +class TestMediaSrcAllowsServerSpeech: + """The server voice tier plays through a blob: URL, so media-src must allow + blob: -- in both modes, and in no other directive. + + tts-engine.js ``_speakServer`` plays the host's WAV with + ``new Audio(URL.createObjectURL(blob))``. With no media-src, media falls + back to ``default-src 'self'``, and ``'self'`` does not match a blob: URL. + Reproduced in Chromium against the real host engine on 2026-09-29: + /api/tts/speak answered 200 audio/wav, the browser refused the load + ("Loading media from 'blob:...' violates the following Content Security + Policy directive: "default-src 'self'""), play() rejected with + NotSupportedError, and the learner heard nothing while the badge said + "Kokoro (server)". The same page with the policy bypassed played. + """ + + @staticmethod + def _directives(dev_mode: bool) -> dict[str, list[str]]: + csp = TestClient(create_app(dev_mode=dev_mode)).get("/").headers["content-security-policy"] + directives: dict[str, list[str]] = {} + for part in csp.split(";"): + tokens = part.split() + if tokens: + directives[tokens[0]] = tokens[1:] + return directives + + @pytest.mark.parametrize("dev_mode", [False, True]) + def test_media_src_allows_blob_urls(self, dev_mode: bool) -> None: + directives = self._directives(dev_mode) + assert "media-src" in directives, ( + "no media-src: media falls back to default-src 'self', which refuses " + "the blob: URL every server-tier utterance plays through" + ) + assert "'self'" in directives["media-src"] + assert "blob:" in directives["media-src"] + + @pytest.mark.parametrize("dev_mode", [False, True]) + def test_blob_is_allowed_for_media_only(self, dev_mode: bool) -> None: + """Guard: the relaxation must not spread. blob: in script-src would let + a script run code from a blob it built; media is the one consumer.""" + directives = self._directives(dev_mode) + widened = sorted( + name + for name, sources in directives.items() + if "blob:" in sources and name != "media-src" + ) + assert not widened, f"blob: must stay scoped to media-src; also found in {widened}" + + class TestSecurityHeadersSurviveExceptions: """R-13d: SecurityHeadersMiddleware was a BaseHTTPMiddleware, whose response path is bypassed when a route raises -- the resulting 500 From 748353874fdc2b552cbd7d8e8a7709e9b484bfe8 Mon Sep 17 00:00:00 2001 From: NetDevAutomate Date: Tue, 29 Sep 2026 09:52:52 +0100 Subject: [PATCH 2/5] fix(web): let the server voice tier play -- media-src 'self' blob: The server tier plays the host's WAV through `new Audio(URL.createObjectURL(blob))`. Since 2 Sep (1f10f059) the page's policy has been `default-src 'self'` with no media-src, and 'self' does not match a blob: URL, so the browser refused every utterance: /api/tts/speak answered 200 audio/wav, play() rejected with NotSupportedError, and the learner heard nothing while the badge said "Kokoro (server)". media-src now allows 'self' and blob:, in both modes. The exception is media-only (a unit guard fails if blob: appears in any other directive): a blob: URL can only be minted by this origin's own script, and script-src stays 'self' 'unsafe-eval', so no new script source is admitted. Chosen over a data: URL (broader: any markup can carry one) and over streaming the audio from a GET URL (the element could no longer read a 503 and fall back to system voices). Verified in Chromium: the faked-host playback test and the four CSP unit tests pass; the whole header suite (test_web_app.py, 32) passes. The real-OpenVox browser test resolved the system-voice tier on this run -- the 2.5s health budget, fixed separately below, not the policy. --- packages/studyloop/src/studyloop/web/app.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/packages/studyloop/src/studyloop/web/app.py b/packages/studyloop/src/studyloop/web/app.py index 208355123..74dec68cf 100644 --- a/packages/studyloop/src/studyloop/web/app.py +++ b/packages/studyloop/src/studyloop/web/app.py @@ -107,6 +107,16 @@ async def _lifespan(app: FastAPI) -> AsyncIterator[None]: # passes `--dev` gets no `data:` relaxation at all, since ghostty-web is # the sole consumer (found in the M1 council, A2: the first cut sent this # unconditionally, in every mode, for a --dev-only need). +# - media-src 'self' blob:: the server voice tier plays the host's WAV +# through `new Audio(URL.createObjectURL(blob))` (tts-engine.js +# `_speakServer`), and `'self'` does not match a blob: URL. With no +# media-src, media fell back to `default-src 'self'` and the browser +# refused every utterance -- /api/tts/speak answered 200 audio/wav, play() +# rejected, the badge said "Kokoro (server)" and nothing was heard, from +# 2 Sep (when default-src arrived) until 29 Sep. Reproduced in Chromium +# against the real host engine; the same page with the policy bypassed +# played. A blob: URL can only be minted by this origin's own script, and +# the exception is media-only: script-src stays `'self' 'unsafe-eval'`. # # R-13c adds the four directives a CSP audit checks for by name rather than # trusting default-src to cover them, each verified free of cost here: @@ -131,6 +141,7 @@ def _build_csp(dev_mode: bool) -> str: return ( "default-src 'self'; script-src 'self' 'unsafe-eval'; " f"style-src 'self' 'unsafe-inline'; {connect_src}; " + "media-src 'self' blob:; " "object-src 'none'; base-uri 'self'; frame-ancestors 'none'; " "form-action 'self'" ) From 878e826aeaf02ea18731107f4e82376f3e059878 Mon Sep 17 00:00:00 2001 From: NetDevAutomate Date: Tue, 29 Sep 2026 09:55:51 +0100 Subject: [PATCH 3/5] fix(web): say so when the voice audio will not play _speakServer resolved a refused play() and an element error in silence, so the policy regression above showed as a dead speech server: the badge stayed "Kokoro (server)", OpenVox had answered every request, and the only trace was one console line. The learner, reasonably, went looking at OpenVox. A refused or undecodable utterance now raises the settings store's existing `tts:engine-notice` toast -- a listener nothing in the engine had emitted since the in-browser neural tier was removed. One notice per utterance (the error event and the rejected play() arrive together), and none for audio that stop() or a newer utterance cut short, which rejects play() with AbortError. NotAllowedError and NotSupportedError get a plain-language hint; anything else is named as it is. Engine notices now stay up 8s rather than the toast's 2s default, which is gone before the sentence is read. Unchanged: no tier switch on a refusal (falling back to system voices is a separate decision, and they need the same user activation), and the fetch-failure path still only logs. Verified: the refusal test passes (undecodable bytes, faked host); the engine e2e file (13) and the JS unit suite (167) pass. --- .../src/studyloop/web/static/components.js | 4 ++- .../src/studyloop/web/static/tts-engine.js | 32 +++++++++++++++++-- 2 files changed, 33 insertions(+), 3 deletions(-) diff --git a/packages/studyloop/src/studyloop/web/static/components.js b/packages/studyloop/src/studyloop/web/static/components.js index e7762621d..5ff4f21b6 100644 --- a/packages/studyloop/src/studyloop/web/static/components.js +++ b/packages/studyloop/src/studyloop/web/static/components.js @@ -465,7 +465,9 @@ document.addEventListener("alpine:init", () => { window.addEventListener('tts:engine-notice', (e) => { const d = (e && e.detail) || {}; this.ttsNotice = d.message || ''; - if (this.ttsNotice) Alpine.store('toast').show(this.ttsNotice); + // 8s, not the toast's 2s default: a failure notice is a sentence the + // learner has to read, and 2s is gone before it is finished. + if (this.ttsNotice) Alpine.store('toast').show(this.ttsNotice, 8000); }); // Adopt whatever the engine already resolved before this store mounted — diff --git a/packages/studyloop/src/studyloop/web/static/tts-engine.js b/packages/studyloop/src/studyloop/web/static/tts-engine.js index 54f793d4e..025c220ec 100644 --- a/packages/studyloop/src/studyloop/web/static/tts-engine.js +++ b/packages/studyloop/src/studyloop/web/static/tts-engine.js @@ -53,6 +53,12 @@ const SERVER_TTS_SPEAK = '/api/tts/speak'; const SERVER_TTS_WARM = '/api/tts/warm'; // A host that is not answering must not hold the voice system in 'warming'. const SERVER_PROBE_TIMEOUT_MS = 2500; +/* What the two common play() refusals mean, for the learner rather than a + * developer. Anything else is reported by its own name. */ +const UNPLAYABLE_HINTS = { + NotAllowedError: 'the browser blocks sound until you click or tap the page', + NotSupportedError: 'the browser refused the audio', +}; /* The voice used until the learner picks one, and until the host's catalogue is * known. * @@ -300,9 +306,24 @@ class TTSEngine { this._serverAudio = audio; try { await new Promise((resolve) => { + // A refused or undecodable audio used to resolve here in silence. The + // badge stayed green, the host had answered every request, and the + // learner went looking at the speech server -- which was fine: the page's + // own policy was refusing the audio. Say so instead. One notice per + // utterance (the error event and the rejected play() arrive together), + // and none for audio that stop() or a newer utterance cut short. + let reported = false; + const refused = (reason) => { + if (!reported && !this._stopped && gen === this._generation) { + reported = true; + const hint = UNPLAYABLE_HINTS[reason] || reason; + this._notice(`The voice would not play (${hint}), so nothing was spoken.`); + } + resolve(); + }; audio.onended = resolve; - audio.onerror = resolve; - audio.play().catch(() => resolve()); + audio.onerror = () => refused(audio.error ? `media error ${audio.error.code}` : 'media error'); + audio.play().catch((err) => refused((err && err.name) || 'play() refused')); }); } finally { URL.revokeObjectURL(url); @@ -310,6 +331,13 @@ class TTSEngine { } } + /* Tell the learner something went wrong inside the engine. The settings store + turns tts:engine-notice into a toast; a console line alone is read by nobody. */ + _notice(message) { + console.warn('[tts-engine]', message); + window.dispatchEvent(new CustomEvent('tts:engine-notice', { detail: { message } })); + } + _initWebSpeech(reason = 'no-server-speech', detail = '') { if (!('speechSynthesis' in window)) { console.warn('[tts-engine] Web Speech API not available, using silent mode'); From e659f8cf9527dd92fbe404eee87adf3bbe930d05 Mon Sep 17 00:00:00 2001 From: NetDevAutomate Date: Tue, 29 Sep 2026 09:59:40 +0100 Subject: [PATCH 4/5] fix(web): give the voice health probe a budget a healthy host fits in The page gave /api/tts/health 2.5s before treating the host as absent. OpenVox's own /v1/models takes 1.2-2.7s on this Mac, and the health route waits on it. Measured on 29 Sep against the real OpenVox: 5 of 8 health calls ran past 2.5s, and 1 page load in 5 resolved to system voices with a healthy host -- a badge that changes between reloads for no visible reason. The real-OpenVox browser test landed on system voices in one run for the same reason. The budget is now 8s. It still ends a hung host's probe (the point of the bound: an init that never resolves looks broken), and leaves room for a tablet's LAN hop. A speech server that is not running on this machine refuses the connection at once (health against a closed local port: 13ms), so the budget changes nothing in the common "no server" case. Verified: after the change, 10 of 10 page loads resolved the server tier against the same OpenVox (health 2.4-2.6s on that run, 6 of 8 over the old budget); the server-TTS e2e file passed 11/11 twice with OpenVox running, and the engine e2e file 13/13. --- packages/studyloop/src/studyloop/web/static/tts-engine.js | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/packages/studyloop/src/studyloop/web/static/tts-engine.js b/packages/studyloop/src/studyloop/web/static/tts-engine.js index 025c220ec..ae504b9c7 100644 --- a/packages/studyloop/src/studyloop/web/static/tts-engine.js +++ b/packages/studyloop/src/studyloop/web/static/tts-engine.js @@ -51,8 +51,12 @@ const SERVER_TTS_HEALTH = '/api/tts/health'; const SERVER_TTS_SPEAK = '/api/tts/speak'; const SERVER_TTS_WARM = '/api/tts/warm'; -// A host that is not answering must not hold the voice system in 'warming'. -const SERVER_PROBE_TIMEOUT_MS = 2500; +// A host that is not answering must not hold the voice system in 'warming' -- +// but a healthy one has to fit. OpenVox's own /v1/models takes 1.2-2.7s on the +// developer's Mac (measured 29 Sep), so /api/tts/health does too; a 2.5s budget +// missed 5 of 8 calls and sent 1 page load in 5 to system voices with the host +// up. 8s still ends a hung host's probe and leaves room for a tablet's LAN hop. +const SERVER_PROBE_TIMEOUT_MS = 8000; /* What the two common play() refusals mean, for the learner rather than a * developer. Anything else is reported by its own name. */ const UNPLAYABLE_HINTS = { From b9cc072966fde1cf6209d19e40e540e3c60de27f Mon Sep 17 00:00:00 2001 From: NetDevAutomate Date: Tue, 29 Sep 2026 10:00:48 +0100 Subject: [PATCH 5/5] docs: the web voice fixes, and the playback notice CHANGELOG [Unreleased] Fixed: the media-src policy regression (2 Sep to now), the silent playback refusal, and the 2.5s health budget. The voice guide gains a Playback notice bullet under Controls: the badge names the tier, not whether this device could play a given utterance. Verified: mkdocs build --strict (docs extra) and the docs tests (524 passed, 3 skipped). --- CHANGELOG.md | 15 +++++++++++++++ docs/voice-output.md | 1 + 2 files changed, 16 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 996debb09..ad618dc4f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -83,6 +83,21 @@ experience may change before `1.0.0`. ### Fixed +- Web voice plays again. From 2 September the page's security policy + (`default-src 'self'`, no `media-src`) refused the `blob:` URL every + server-voice utterance plays through, so the badge said **Kokoro (server)**, + the speech server answered every request, and nothing was heard. The policy + now allows `media-src 'self' blob:`, for media only. A new browser test fakes + the speech server's answers, so CI checks that the audio actually plays; the + only earlier test needed a running speech server and so never ran in CI. +- When the speech server sends audio that the browser will not play, a notice + now says so and stays up for 8 seconds. The refusal used to be silent, which + made a browser-side problem look like a dead speech server. +- A slow but healthy speech server keeps the server voice. The page gave + `/api/tts/health` 2.5 seconds, and OpenVox's own model list takes 1.2 to + 2.7 seconds, so about one page load in five fell back to system voices with + the server up. The budget is now 8 seconds; a server that is not running + still fails at once. - The web header is one line of controls on the brand's centre line. **Voice**, **Theme**, **Font** and **Size** each have a visible label, set above the control and out of the layout, so every control is the same 32px diff --git a/docs/voice-output.md b/docs/voice-output.md index 91ce41f94..79fa8a411 100644 --- a/docs/voice-output.md +++ b/docs/voice-output.md @@ -311,6 +311,7 @@ An iPad (both Brave and Chrome) and an Android tablet all speak through the serv - **Read once** — tap the speaker icon on a card, or press `T`. Reads the current content once. This works whether or not the header voice toggle is on. - **Voice toggle** — the header speaker button. It enables the app's own spoken announcements (Pomodoro transitions and confirmations such as "voice enabled") and reveals the voice selector and engine badge. Persists to `localStorage` under `voice`. **Click-only — no key is bound to it, and it does not read cards to you automatically.** - **Engine badge** — names the tier actually speaking (`server-openvox`, `web-speech`, `silent`), shown whenever voice is on rather than only on failure. A badge that appears only when something breaks teaches nobody what working looks like. +- **Playback notice** — if the server sends audio but the browser will not play it, a notice says so (for example, that the browser blocks sound until you click the page) instead of staying silent. The badge names the tier, not whether this device could play a particular utterance. - **Stop** — a stop button appears in the header while audio is playing; click it (or it clears automatically when playback ends) to interrupt mid-utterance. It halts server-audio playback, not just Web Speech API output. !!! warning "There is no auto-voice, and `V` is unbound"