diff --git a/CHANGELOG.md b/CHANGELOG.md index 996debb0..ad618dc4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -83,6 +83,21 @@ experience may change before `1.0.0`. ### Fixed +- Web voice plays again. From 2 September the page's security policy + (`default-src 'self'`, no `media-src`) refused the `blob:` URL every + server-voice utterance plays through, so the badge said **Kokoro (server)**, + the speech server answered every request, and nothing was heard. The policy + now allows `media-src 'self' blob:`, for media only. A new browser test fakes + the speech server's answers, so CI checks that the audio actually plays; the + only earlier test needed a running speech server and so never ran in CI. +- When the speech server sends audio that the browser will not play, a notice + now says so and stays up for 8 seconds. The refusal used to be silent, which + made a browser-side problem look like a dead speech server. +- A slow but healthy speech server keeps the server voice. The page gave + `/api/tts/health` 2.5 seconds, and OpenVox's own model list takes 1.2 to + 2.7 seconds, so about one page load in five fell back to system voices with + the server up. The budget is now 8 seconds; a server that is not running + still fails at once. - The web header is one line of controls on the brand's centre line. **Voice**, **Theme**, **Font** and **Size** each have a visible label, set above the control and out of the layout, so every control is the same 32px diff --git a/docs/voice-output.md b/docs/voice-output.md index 91ce41f9..79fa8a41 100644 --- a/docs/voice-output.md +++ b/docs/voice-output.md @@ -311,6 +311,7 @@ An iPad (both Brave and Chrome) and an Android tablet all speak through the serv - **Read once** — tap the speaker icon on a card, or press `T`. Reads the current content once. This works whether or not the header voice toggle is on. - **Voice toggle** — the header speaker button. It enables the app's own spoken announcements (Pomodoro transitions and confirmations such as "voice enabled") and reveals the voice selector and engine badge. Persists to `localStorage` under `voice`. **Click-only — no key is bound to it, and it does not read cards to you automatically.** - **Engine badge** — names the tier actually speaking (`server-openvox`, `web-speech`, `silent`), shown whenever voice is on rather than only on failure. A badge that appears only when something breaks teaches nobody what working looks like. +- **Playback notice** — if the server sends audio but the browser will not play it, a notice says so (for example, that the browser blocks sound until you click the page) instead of staying silent. The badge names the tier, not whether this device could play a particular utterance. - **Stop** — a stop button appears in the header while audio is playing; click it (or it clears automatically when playback ends) to interrupt mid-utterance. It halts server-audio playback, not just Web Speech API output. !!! warning "There is no auto-voice, and `V` is unbound" diff --git a/packages/studyloop/src/studyloop/web/app.py b/packages/studyloop/src/studyloop/web/app.py index 20835512..74dec68c 100644 --- a/packages/studyloop/src/studyloop/web/app.py +++ b/packages/studyloop/src/studyloop/web/app.py @@ -107,6 +107,16 @@ async def _lifespan(app: FastAPI) -> AsyncIterator[None]: # passes `--dev` gets no `data:` relaxation at all, since ghostty-web is # the sole consumer (found in the M1 council, A2: the first cut sent this # unconditionally, in every mode, for a --dev-only need). +# - media-src 'self' blob:: the server voice tier plays the host's WAV +# through `new Audio(URL.createObjectURL(blob))` (tts-engine.js +# `_speakServer`), and `'self'` does not match a blob: URL. With no +# media-src, media fell back to `default-src 'self'` and the browser +# refused every utterance -- /api/tts/speak answered 200 audio/wav, play() +# rejected, the badge said "Kokoro (server)" and nothing was heard, from +# 2 Sep (when default-src arrived) until 29 Sep. Reproduced in Chromium +# against the real host engine; the same page with the policy bypassed +# played. A blob: URL can only be minted by this origin's own script, and +# the exception is media-only: script-src stays `'self' 'unsafe-eval'`. # # R-13c adds the four directives a CSP audit checks for by name rather than # trusting default-src to cover them, each verified free of cost here: @@ -131,6 +141,7 @@ def _build_csp(dev_mode: bool) -> str: return ( "default-src 'self'; script-src 'self' 'unsafe-eval'; " f"style-src 'self' 'unsafe-inline'; {connect_src}; " + "media-src 'self' blob:; " "object-src 'none'; base-uri 'self'; frame-ancestors 'none'; " "form-action 'self'" ) diff --git a/packages/studyloop/src/studyloop/web/static/components.js b/packages/studyloop/src/studyloop/web/static/components.js index e7762621..5ff4f21b 100644 --- a/packages/studyloop/src/studyloop/web/static/components.js +++ b/packages/studyloop/src/studyloop/web/static/components.js @@ -465,7 +465,9 @@ document.addEventListener("alpine:init", () => { window.addEventListener('tts:engine-notice', (e) => { const d = (e && e.detail) || {}; this.ttsNotice = d.message || ''; - if (this.ttsNotice) Alpine.store('toast').show(this.ttsNotice); + // 8s, not the toast's 2s default: a failure notice is a sentence the + // learner has to read, and 2s is gone before it is finished. + if (this.ttsNotice) Alpine.store('toast').show(this.ttsNotice, 8000); }); // Adopt whatever the engine already resolved before this store mounted — diff --git a/packages/studyloop/src/studyloop/web/static/tts-engine.js b/packages/studyloop/src/studyloop/web/static/tts-engine.js index 54f793d4..ae504b9c 100644 --- a/packages/studyloop/src/studyloop/web/static/tts-engine.js +++ b/packages/studyloop/src/studyloop/web/static/tts-engine.js @@ -51,8 +51,18 @@ const SERVER_TTS_HEALTH = '/api/tts/health'; const SERVER_TTS_SPEAK = '/api/tts/speak'; const SERVER_TTS_WARM = '/api/tts/warm'; -// A host that is not answering must not hold the voice system in 'warming'. -const SERVER_PROBE_TIMEOUT_MS = 2500; +// A host that is not answering must not hold the voice system in 'warming' -- +// but a healthy one has to fit. OpenVox's own /v1/models takes 1.2-2.7s on the +// developer's Mac (measured 29 Sep), so /api/tts/health does too; a 2.5s budget +// missed 5 of 8 calls and sent 1 page load in 5 to system voices with the host +// up. 8s still ends a hung host's probe and leaves room for a tablet's LAN hop. +const SERVER_PROBE_TIMEOUT_MS = 8000; +/* What the two common play() refusals mean, for the learner rather than a + * developer. Anything else is reported by its own name. */ +const UNPLAYABLE_HINTS = { + NotAllowedError: 'the browser blocks sound until you click or tap the page', + NotSupportedError: 'the browser refused the audio', +}; /* The voice used until the learner picks one, and until the host's catalogue is * known. * @@ -300,9 +310,24 @@ class TTSEngine { this._serverAudio = audio; try { await new Promise((resolve) => { + // A refused or undecodable audio used to resolve here in silence. The + // badge stayed green, the host had answered every request, and the + // learner went looking at the speech server -- which was fine: the page's + // own policy was refusing the audio. Say so instead. One notice per + // utterance (the error event and the rejected play() arrive together), + // and none for audio that stop() or a newer utterance cut short. + let reported = false; + const refused = (reason) => { + if (!reported && !this._stopped && gen === this._generation) { + reported = true; + const hint = UNPLAYABLE_HINTS[reason] || reason; + this._notice(`The voice would not play (${hint}), so nothing was spoken.`); + } + resolve(); + }; audio.onended = resolve; - audio.onerror = resolve; - audio.play().catch(() => resolve()); + audio.onerror = () => refused(audio.error ? `media error ${audio.error.code}` : 'media error'); + audio.play().catch((err) => refused((err && err.name) || 'play() refused')); }); } finally { URL.revokeObjectURL(url); @@ -310,6 +335,13 @@ class TTSEngine { } } + /* Tell the learner something went wrong inside the engine. The settings store + turns tts:engine-notice into a toast; a console line alone is read by nobody. */ + _notice(message) { + console.warn('[tts-engine]', message); + window.dispatchEvent(new CustomEvent('tts:engine-notice', { detail: { message } })); + } + _initWebSpeech(reason = 'no-server-speech', detail = '') { if (!('speechSynthesis' in window)) { console.warn('[tts-engine] Web Speech API not available, using silent mode'); diff --git a/packages/studyloop/tests/e2e/test_server_tts.py b/packages/studyloop/tests/e2e/test_server_tts.py index e997f43e..90a3986b 100644 --- a/packages/studyloop/tests/e2e/test_server_tts.py +++ b/packages/studyloop/tests/e2e/test_server_tts.py @@ -17,13 +17,20 @@ from __future__ import annotations +import io import sys +import time +import wave +from contextlib import suppress from pathlib import Path from typing import TYPE_CHECKING import pytest pytest.importorskip("requests") +pytest.importorskip("playwright") + +from playwright.sync_api import Error as PlaywrightError _tests_dir = str(Path(__file__).resolve().parent.parent) if _tests_dir not in sys.path: @@ -32,7 +39,7 @@ from e2e._env import ConsoleWatch, launch_env, shutdown # noqa: E402 if TYPE_CHECKING: - from playwright.sync_api import Browser + from playwright.sync_api import Browser, Page pytestmark = [pytest.mark.e2e] @@ -158,6 +165,189 @@ def test_warm_is_idempotent(env) -> None: assert response.json()["warmed"] is True +# --------------------------------------------------------------------------- +# Playback probes — did the learner HEAR it, not just did the engine ask? +# --------------------------------------------------------------------------- + +#: Installed before any app script: records how every media element's play() +#: ended and every CSP violation, so a test can tell "the engine asked for audio" +#: apart from "the audio played". +_PLAYBACK_PROBE = """ +window.__plays = []; +window.__cspViolations = []; +document.addEventListener('securitypolicyviolation', (e) => { + window.__cspViolations.push(`${e.effectiveDirective} ${e.blockedURI}`); +}); +const __play = HTMLMediaElement.prototype.play; +HTMLMediaElement.prototype.play = function () { + const record = { outcome: 'pending', playing: false }; + window.__plays.push(record); + this.addEventListener('playing', () => { record.playing = true; }); + const promise = __play.call(this); + promise.then( + () => { record.outcome = 'resolved'; }, + (err) => { record.outcome = `${err.name}: ${err.message}`; }, + ); + return promise; +}; +""" + +_PLAYS_SETTLED = ( + "() => window.__plays.length > 0 && window.__plays.every((p) => p.outcome !== 'pending')" +) + + +def _enable_voice(page: Page, base_url: str, *, timeout_ms: int) -> dict: + """Click the header voice toggle, as the learner does, and report how its + "Voice enabled" utterance ended. + + A real click rather than page.evaluate: it is the learner's path, and it + gives the page the user activation a browser requires before audio plays. + """ + page.goto(f"{base_url}/", wait_until="domcontentloaded", timeout=30000) + page.wait_for_function( + "() => !!window.Alpine && !!window.Alpine.store('settings')", timeout=15000 + ) + page.click("button[title^='Toggle voice']") + page.wait_for_function( + "() => !!window.ttsEngine && window.ttsEngine.tier !== null", timeout=timeout_ms + ) + # Only the server tier plays through a media element; on any other tier + # there is nothing to wait for, and the caller's tier assertion says why. + if page.evaluate("() => window.ttsEngine.tier") == "server-openvox": + page.wait_for_function(_PLAYS_SETTLED, timeout=timeout_ms) + return page.evaluate( + "() => ({ plays: window.__plays, csp: window.__cspViolations," + " tier: window.ttsEngine.tier })" + ) + + +#: The host's answers, faked at the network edge so the next two tests need no +#: OpenVox and therefore run in CI. Only /api/tts/* is faked: the page, its real +#: Content-Security-Policy header and the real tts-engine.js come from the server. +_FAKE_HEALTH = { + "available": True, + "model": "kokoro", + "voice_count": 1, + "detail": "", + "voices": [{"id": "bf_emma", "language": "en-gb", "british": True}], +} + + +def _silent_wav(seconds: float = 0.2, rate: int = 24000) -> bytes: + """A real, decodable WAV in the shape OpenVox returns: 24 kHz, 16-bit, mono.""" + buffer = io.BytesIO() + with wave.open(buffer, "wb") as out: + out.setnchannels(1) + out.setsampwidth(2) + out.setframerate(rate) + out.writeframes(b"\x00\x00" * int(seconds * rate)) + return buffer.getvalue() + + +def _fake_host(page: Page, *, audio: bytes, health_delay_s: float = 0.0) -> None: + def health(route) -> None: + if health_delay_s: + time.sleep(health_delay_s) # the page's fetch waits; nothing else runs meanwhile + # The page may have given up and aborted the fetch while we slept. + with suppress(PlaywrightError): + route.fulfill(json=_FAKE_HEALTH) + + page.route("**/api/tts/health", health) + page.route("**/api/tts/warm", lambda route: route.fulfill(json={"warmed": True})) + page.route( + "**/api/tts/speak", + lambda route: route.fulfill(status=200, content_type="audio/wav", body=audio), + ) + + +def test_host_audio_plays_under_the_pages_own_policy(browser: Browser, env) -> None: + """The host's audio must PLAY in the page, not merely be fetched. + + Regression: `default-src 'self'` (2 Sep) with no media-src refused the blob: + URL every server-tier utterance plays through. /api/tts/speak answered 200 + audio/wav, play() rejected, and the learner heard nothing while the badge + said "Kokoro (server)". The real-engine test below caught that only while + OpenVox was running, and so never in CI, which has none. This one fakes the + host's answers instead, so CI runs it. + """ + ctx = browser.new_context(viewport={"width": 1440, "height": 900}) + page = ctx.new_page() + page.add_init_script(_PLAYBACK_PROBE) + _fake_host(page, audio=_silent_wav()) + watch = ConsoleWatch(page) + try: + state = _enable_voice(page, env.base_url, timeout_ms=20000) + + assert state["tier"] == "server-openvox" + assert state["csp"] == [], f"the page's own policy refused: {state['csp']}" + assert [p["outcome"] for p in state["plays"]] == ["resolved"], state["plays"] + assert state["plays"][0]["playing"], "play() resolved but playback never started" + watch.assert_clean("playing the host's audio") + finally: + ctx.close() + + +def test_audio_the_page_cannot_play_is_reported_not_silent(browser: Browser, env) -> None: + """When the host answers but the browser will not play it, the learner is told. + + The engine resolved a refused play() silently, which is why the regression + above looked like a dead speech server: the badge stayed green, OpenVox had + answered every request, and the only trace was one console line. Bytes no + browser can decode stand in for any refusal: a policy, a codec, a corrupt + response. + """ + ctx = browser.new_context(viewport={"width": 1440, "height": 900}) + page = ctx.new_page() + page.add_init_script(_PLAYBACK_PROBE) + _fake_host(page, audio=b"this is not audio " * 64) + try: + state = _enable_voice(page, env.base_url, timeout_ms=20000) + assert state["plays"][0]["outcome"] != "resolved", "undecodable bytes played?" + + toast = _toast_text(page, timeout_ms=5000) + assert "would not play" in toast, f"the refusal was silent (toast: {toast!r})" + # A failure notice has to last long enough to read; the store's 2s + # default is gone before a sentence is finished. + page.wait_for_timeout(3000) + assert page.evaluate("() => window.Alpine.store('toast').visible"), ( + "the failure notice vanished before it could be read" + ) + finally: + ctx.close() + + +def test_a_slow_but_alive_host_still_gets_the_server_tier(browser: Browser, env) -> None: + """A host that answers in a few seconds is present, not absent. + + Measured on the developer's Mac (29 Sep): OpenVox's own /v1/models takes 1.2 + to 2.7s, so /api/tts/health does too, and the page gave it 2.5s. Five of eight + health calls ran over, and one page load in five fell back to system voices + with a healthy host -- a badge that flips between loads for no visible reason. + Four seconds is over the old budget and well inside a sane one. + """ + ctx = browser.new_context(viewport={"width": 1440, "height": 900}) + page = ctx.new_page() + page.add_init_script(_PLAYBACK_PROBE) + _fake_host(page, audio=_silent_wav(), health_delay_s=4.0) + try: + state = _enable_voice(page, env.base_url, timeout_ms=20000) + assert state["tier"] == "server-openvox", ( + f"a host answering in 4s was treated as absent (tier {state['tier']!r})" + ) + finally: + ctx.close() + + +def _toast_text(page: Page, *, timeout_ms: int) -> str: + """The toast's message once one appears, or '' if none does in time.""" + from playwright.sync_api import TimeoutError as PlaywrightTimeoutError + + with suppress(PlaywrightTimeoutError): + page.wait_for_function("() => !!window.Alpine.store('toast').message", timeout=timeout_ms) + return page.evaluate("() => window.Alpine.store('toast').message || ''") + + # --------------------------------------------------------------------------- # Browser leg — the reason this path exists # --------------------------------------------------------------------------- @@ -172,30 +362,28 @@ def test_browser_uses_the_server_tier_without_touching_webgpu(browser: Browser, needs none of them. A tablet on `--lan` is served over plain HTTP, so it is not a secure context and cannot have the first path -- if this assertion ever fails, the tablet has silently lost its voice. + + It asserts that the audio PLAYED. It used to count play() calls, which kept + passing while the page's own policy refused every one (see + test_host_audio_plays_under_the_pages_own_policy below). """ if not _health(env)["available"]: pytest.skip("no OpenVox on this host") ctx = browser.new_context(viewport={"width": 1440, "height": 900}) page = ctx.new_page() + page.add_init_script(_PLAYBACK_PROBE) watch = ConsoleWatch(page) try: - page.goto(f"{env.base_url}/", wait_until="domcontentloaded", timeout=30000) - page.wait_for_timeout(1500) - page.evaluate( - """() => { - window.__played = 0; - const orig = Audio.prototype.play; - Audio.prototype.play = function () { window.__played++; return orig.call(this); }; - }""" - ) - page.evaluate("async () => { await window.ttsEngine.init(); }") + # A cold model costs ~51s on its first utterance, so allow for one. + state = _enable_voice(page, env.base_url, timeout_ms=120000) - assert page.evaluate("() => window.ttsEngine.tier") == "server-openvox" + assert state["tier"] == "server-openvox" assert page.evaluate("() => window.ttsEngine.listVoices().length") > 0 - - page.evaluate("async () => { await window.ttsEngine.speak('Hello from the host.'); }") - assert page.evaluate("() => window.__played") >= 1, "no audio element ever played" + assert state["csp"] == [], f"the page's own policy refused: {state['csp']}" + assert [p["outcome"] for p in state["plays"]] == ["resolved"], ( + f"the host's audio never played: {state['plays']}" + ) assert page.evaluate("() => window.ttsEngine._audioCtx ? 'created' : 'never'") == "never", ( "the server tier should never build an AudioContext" ) diff --git a/packages/studyloop/tests/test_web_app.py b/packages/studyloop/tests/test_web_app.py index 9fc432a9..068b6104 100644 --- a/packages/studyloop/tests/test_web_app.py +++ b/packages/studyloop/tests/test_web_app.py @@ -196,6 +196,54 @@ def test_data_exception_present_when_dev_mode_is_on(self) -> None: assert "data:" in connect_src +class TestMediaSrcAllowsServerSpeech: + """The server voice tier plays through a blob: URL, so media-src must allow + blob: -- in both modes, and in no other directive. + + tts-engine.js ``_speakServer`` plays the host's WAV with + ``new Audio(URL.createObjectURL(blob))``. With no media-src, media falls + back to ``default-src 'self'``, and ``'self'`` does not match a blob: URL. + Reproduced in Chromium against the real host engine on 2026-09-29: + /api/tts/speak answered 200 audio/wav, the browser refused the load + ("Loading media from 'blob:...' violates the following Content Security + Policy directive: "default-src 'self'""), play() rejected with + NotSupportedError, and the learner heard nothing while the badge said + "Kokoro (server)". The same page with the policy bypassed played. + """ + + @staticmethod + def _directives(dev_mode: bool) -> dict[str, list[str]]: + csp = TestClient(create_app(dev_mode=dev_mode)).get("/").headers["content-security-policy"] + directives: dict[str, list[str]] = {} + for part in csp.split(";"): + tokens = part.split() + if tokens: + directives[tokens[0]] = tokens[1:] + return directives + + @pytest.mark.parametrize("dev_mode", [False, True]) + def test_media_src_allows_blob_urls(self, dev_mode: bool) -> None: + directives = self._directives(dev_mode) + assert "media-src" in directives, ( + "no media-src: media falls back to default-src 'self', which refuses " + "the blob: URL every server-tier utterance plays through" + ) + assert "'self'" in directives["media-src"] + assert "blob:" in directives["media-src"] + + @pytest.mark.parametrize("dev_mode", [False, True]) + def test_blob_is_allowed_for_media_only(self, dev_mode: bool) -> None: + """Guard: the relaxation must not spread. blob: in script-src would let + a script run code from a blob it built; media is the one consumer.""" + directives = self._directives(dev_mode) + widened = sorted( + name + for name, sources in directives.items() + if "blob:" in sources and name != "media-src" + ) + assert not widened, f"blob: must stay scoped to media-src; also found in {widened}" + + class TestSecurityHeadersSurviveExceptions: """R-13d: SecurityHeadersMiddleware was a BaseHTTPMiddleware, whose response path is bypassed when a route raises -- the resulting 500