diff --git a/CHANGELOG.md b/CHANGELOG.md index 95fa2e680..534142f5a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,14 @@ metadata and the backend fallback mirror it. ## [Unreleased] +**Highlights** + +- MCP speech tools stay connected through cold starts and slow, progressing renders (#2612) + +### Fixed + +- MCP speech tools wait through model loading and progress-extended CPU renders instead of timing out before the backend (#2609) + ## [0.5.7] — 2026-10-05 **VoiceStudio now runs on PCs without a GPU and recovers instead of giving up.** Voice cloning uses the speech-to-text model you installed from Model Catalogue, GPU-less computers get the small CPU PyTorch build, and slow or busy backends are no longer reported as failed. diff --git a/backend/mcp_server.py b/backend/mcp_server.py index a0ffd113c..b43e36562 100644 --- a/backend/mcp_server.py +++ b/backend/mcp_server.py @@ -282,6 +282,12 @@ async def _write_output(audio_id: str, raw: bytes, format: str = "wav") -> str: # client-side timeout (#2040). _BACKEND_GRACE_S = 30.0 +# Torch-free mirrors of the desktop backstop and model_manager's guard. +# tests/test_generate_abort_budget.py keeps these in sync with their sources. +_GENERATE_SIDECAR_FLOOR_S = 900.0 +_GENERATE_SIDECAR_GRACE_S = 5.0 +_GENERATE_PROGRESS_BUDGETS = 3.0 + def _env_seconds(name: str, default: float) -> float: raw = os.environ.get(name, "").strip() @@ -319,12 +325,32 @@ def _backend_budget_s(kind: str, text: str = "") -> float | None: from core.generate_budget import client_execution_budget_s execution = client_execution_budget_s( - base, len(text or ""), + max(base, _GENERATE_SIDECAR_FLOOR_S), len(text or ""), cpu_auto_possible=not os.environ.get("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "").strip(), + ) + _GENERATE_SIDECAR_GRACE_S + # Classic /generate sends no response until the whole render finishes. + # Cold loading and queueing have separate clocks; fresh chunk-progress + # heartbeats can then extend execution by up to three more budgets. + # Waiting only for queue + execution cuts off healthy CPU renders. + model_load = max(30.0, _env_seconds("OMNIVOICE_MODEL_LOAD_TIMEOUT", 1200.0)) + extension_cap = _env_seconds( + "OMNIVOICE_PROGRESS_EXTENSION_CAP_S", + _env_seconds("OMNIVOICE_MODEL_LOAD_TIMEOUT_S", 1800.0), + ) + extension = max(extension_cap, _GENERATE_PROGRESS_BUDGETS * execution) + queue = _env_seconds("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", 1800.0) + # A clone without a cached reference transcript first runs a separate + # guarded ASR job. That job uses generate_timeout_s("") without an + # engine: no length bonus or sidecar grace, but its own queue and + # progress extension. MCP cannot see whether the profile needs it. + reference = queue + base + max(extension_cap, _GENERATE_PROGRESS_BUDGETS * base) + return ( + model_load + + reference + + queue + + execution + + extension ) - # A generation first waits in the GPU pool's queue, on its own clock - # (model_manager.GPU_QUEUE_TIMEOUT_S), before that budget starts. - return _env_seconds("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", 1800.0) + execution return None diff --git a/docs/performance.md b/docs/performance.md index 1b55f0ed0..5b64950a5 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -139,11 +139,12 @@ editor, profile previews, and streaming). | The host synthesizes on the CPU **and** the text is over 1200 characters | A heads-up that this generation may exceed the time budget | | The host synthesizes on Apple Silicon (MPS) **and** the text is over 1200 characters | The same heads-up — MPS gets the accelerated-host budget (`OMNIVOICE_GENERATE_TIMEOUT_S`), which a long render can still legitimately exceed | -**Why 1200 characters:** it is the same figure the budget itself uses. The first -1200 characters get the flat base budget, and only past that does the budget -start growing (+1 s per 40 characters). Below the threshold you are inside a -budget the backend already considers generous, so ordinary sentences on a CPU -laptop stay quiet. +**Why 1200 characters:** this advisory threshold matches the free allowance in +the legacy accelerated/explicit-budget rule: the first 1200 characters get the +flat base, then the budget grows by 1 s per 40 characters. Default CPU budgeting +uses a separate rule: its 4 s per character exceeds the 600 s floor above 150 +characters, so a 400-character passage receives 1600 s even though no length +warning appears. The warning threshold itself is unchanged. **Which base applies:** @@ -156,16 +157,25 @@ laptop stay quiet. **CPU hosts scale much faster than the +1 s per 40 characters.** A CPU render is often 10-50x slower than on a GPU, so while `OMNIVOICE_CPU_GENERATE_TIMEOUT_S` is left at its default the budget grows at 4 s per input character (a 400-character -passage gets about 27 minutes), up to a 2-hour ceiling that still catches a -genuinely wedged engine. Each streamed chunk is budgeted from its own text, and -loading the model is not part of this clock. Setting the CPU budget explicitly -turns this scaling off and uses your value as the floor (plus the standard -+1 s per 40 characters) — an explicit setting is always authoritative. - -The desktop app and MCP tools never wait less than the backend does: because the -backend budgets the text *after* number normalization (a six-digit number grows -about 11x), a CPU host on the default budget reports its 2-hour ceiling and -clients wait for that rather than guessing from the typed length. +passage gets about 27 minutes), capped at 2 hours of base compute allowance. +Queueing, model loading and the existing progress-extension allowance are +separate. Each streamed chunk is budgeted from its own text; a silent, wedged job +exhausts its compute allowance. Setting the CPU budget explicitly turns this +scaling off and uses your value as the floor (plus the standard +1 s per 40 +characters) — an explicit setting is always authoritative. + +The desktop backstop accounts for the reported automatic CPU ceiling on local +CPU-routed jobs with the default budget. MCP tools conservatively allow that +ceiling whenever the CPU budget is not explicitly set. Both waits also include +model-load, queue, sidecar and progress-extension allowances. The ceiling avoids +guessing the compute budget from typed text that number normalization or +pronunciation rules can expand before synthesis. + +MCP generation also allows a separate reference-transcription job before +synthesis for clone profiles without a cached transcript. That job uses the +generation base budget, its own queue and progress extension; it does not use +the standalone transcription timeout. MCP includes this allowance conservatively +because it cannot inspect the backend's cached reference transcript. Both rows above can be overridden, and the two vars are independent: diff --git a/tests/test_generate_abort_budget.py b/tests/test_generate_abort_budget.py index aca8cc027..ddb74de3b 100644 --- a/tests/test_generate_abort_budget.py +++ b/tests/test_generate_abort_budget.py @@ -30,6 +30,7 @@ def _source_default(path: Path, env: str) -> float: def test_client_budget_mirrors_backend_defaults(monkeypatch): from services import model_manager from worker import deadlines + import mcp_server monkeypatch.delenv("OMNIVOICE_MODEL_LOAD_TIMEOUT", raising=False) manager = ROOT / "backend/services/model_manager.py" @@ -39,6 +40,9 @@ def test_client_budget_mirrors_backend_defaults(monkeypatch): assert client["queueWait"] == _source_default(manager, "OMNIVOICE_GPU_QUEUE_TIMEOUT_S") assert client["progressExtensionCap"] == model_manager.progress_extension_cap_s({}) assert client["progressExtensionBudgets"] == model_manager.PROGRESS_EXTENSION_BUDGETS + assert mcp_server._GENERATE_PROGRESS_BUDGETS == model_manager.PROGRESS_EXTENSION_BUDGETS + assert mcp_server._GENERATE_SIDECAR_FLOOR_S == client["executionBase"] + assert mcp_server._GENERATE_SIDECAR_GRACE_S == client["sidecarGrace"] assert client["freeChars"] == deadlines._FREE_CHARS assert client["charsPerSecond"] == deadlines._CHARS_PER_SECOND from core import generate_budget as gb @@ -69,3 +73,4 @@ def test_client_execution_base_covers_every_default_execution_budget(): ] assert len(bases) > 3, "receive-timeout scan found nothing; fix the regex" assert _client_budget()["executionBase"] >= max(bases) + diff --git a/tests/test_mcp_timeouts_2040.py b/tests/test_mcp_timeouts_2040.py index 668cb1ff3..b4b4c17ca 100644 --- a/tests/test_mcp_timeouts_2040.py +++ b/tests/test_mcp_timeouts_2040.py @@ -9,11 +9,23 @@ "OMNIVOICE_GENERATE_TIMEOUT_S", "OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "OMNIVOICE_GPU_QUEUE_TIMEOUT_S", + "OMNIVOICE_MODEL_LOAD_TIMEOUT", + "OMNIVOICE_PROGRESS_EXTENSION_CAP_S", + "OMNIVOICE_MODEL_LOAD_TIMEOUT_S", ) QUEUE = 1800.0 GRACE = 30.0 +def _generate_wait( + execution, *, queue=QUEUE, model_load=1200.0, extension_cap=1800.0, + reference_base=600.0, +): + execution += 5.0 # sidecar watchdog grace + reference = queue + reference_base + max(extension_cap, 3.0 * reference_base) + return model_load + reference + queue + execution + max(extension_cap, 3.0 * execution) + GRACE + + @pytest.fixture def post_timeout(monkeypatch): for name in _BUDGET_VARS: @@ -39,28 +51,30 @@ def test_generation_covers_the_queue_and_the_length_scaled_budget(post_timeout): # waits for that ceiling whatever the typed length. from core.generate_budget import CPU_AUTO_CAP_S - assert post_timeout("generate", "short") == QUEUE + CPU_AUTO_CAP_S + GRACE - assert post_timeout("generate", "x" * 1600) == QUEUE + CPU_AUTO_CAP_S + GRACE + assert post_timeout("generate", "short") == _generate_wait(CPU_AUTO_CAP_S) + assert post_timeout("generate", "x" * 1600) == _generate_wait(CPU_AUTO_CAP_S) def test_the_larger_cpu_generation_budget_wins(post_timeout, monkeypatch): monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "900") - assert post_timeout("generate", "short") == QUEUE + 900.0 + GRACE + assert post_timeout("generate", "short") == _generate_wait(900.0, reference_base=900.0) # An explicit CPU budget is authoritative and uncapped: no ceiling applies. - assert post_timeout("generate", "x" * 1600) == QUEUE + 900.0 + (1600 * 16 - 1200) / 40.0 + GRACE + assert post_timeout("generate", "x" * 1600) == _generate_wait( + 900.0 + (1600 * 16 - 1200) / 40.0, reference_base=900.0, + ) def test_a_raised_gpu_generation_budget_wins_when_larger(post_timeout, monkeypatch): monkeypatch.setenv("OMNIVOICE_GENERATE_TIMEOUT_S", "1200") monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "600") - assert post_timeout("generate", "short") == QUEUE + 1200.0 + GRACE + assert post_timeout("generate", "short") == _generate_wait(1200.0, reference_base=1200.0) def test_a_shorter_queue_budget_is_followed(post_timeout, monkeypatch): monkeypatch.setenv("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", "60") from core.generate_budget import CPU_AUTO_CAP_S - assert post_timeout("generate", "short") == 60.0 + CPU_AUTO_CAP_S + GRACE + assert post_timeout("generate", "short") == _generate_wait(CPU_AUTO_CAP_S, queue=60.0) def test_an_explicit_mcp_timeout_still_wins(post_timeout, monkeypatch): @@ -94,5 +108,71 @@ def test_transcribe_never_gives_up_before_the_backend(post_timeout): def test_generation_never_gives_up_before_the_backend(post_timeout, device, text): from services import model_manager as mm - backend_worst = mm.GPU_QUEUE_TIMEOUT_S + mm.generate_timeout_s(text, execution_device=device) + from types import SimpleNamespace + + execution = mm.generate_timeout_s( + text, execution_device=device, engine=SimpleNamespace(recv_timeout_s=900.0), + ) + reference = mm.generate_timeout_s("", execution_device=device) + backend_worst = ( + mm._model_load_timeout() + 2 * mm.GPU_QUEUE_TIMEOUT_S + reference + + max(mm.progress_extension_cap_s(), mm.PROGRESS_EXTENSION_BUDGETS * reference) + + execution + + max(mm.progress_extension_cap_s(), mm.PROGRESS_EXTENSION_BUDGETS * execution) + ) assert post_timeout("generate", text) > backend_worst + + +def test_cpu_generation_covers_cold_load_and_progress_extensions(post_timeout): + # Fail-before: queue + base + grace was 9,030 s, but a progressing CPU + # render can legally use 28,800 s of compute alone. Sidecar grace is + # included before sizing its extension, just as in the real guard. + assert post_timeout("generate", "x" * 2000) == 36_050.0 + + +def test_a_raised_model_load_budget_is_followed(post_timeout, monkeypatch): + monkeypatch.setenv("OMNIVOICE_MODEL_LOAD_TIMEOUT", "2000") + assert post_timeout("generate", "short") == 36_850.0 + + +@pytest.mark.parametrize("primary", [None, "", "10000"]) +def test_progress_extension_cap_honors_the_legacy_alias(post_timeout, monkeypatch, primary): + monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "900") + monkeypatch.setenv("OMNIVOICE_MODEL_LOAD_TIMEOUT_S", "20000") + if primary is not None: + monkeypatch.setenv("OMNIVOICE_PROGRESS_EXTENSION_CAP_S", primary) + cap = 10000.0 if primary else 20000.0 + assert post_timeout("generate", "short") == _generate_wait( + 900.0, extension_cap=cap, reference_base=900.0, + ) + + +def test_transcriptless_clone_covers_both_serial_jobs(post_timeout): + # Fail-before: the one-job wait was 31,850 s, but reference ASR and + # synthesis can together consume 36,000 s under their separate guards. + reference = QUEUE + 600.0 + 1800.0 + synthesis = QUEUE + 7200.0 + 21_600.0 + assert post_timeout("generate", "x" * 2000) > 1200.0 + reference + synthesis + + +def test_reference_transcription_uses_generation_not_standalone_asr_budget(post_timeout, monkeypatch): + monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "2000") + monkeypatch.setenv("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", "60") + monkeypatch.setenv("OMNIVOICE_PROGRESS_EXTENSION_CAP_S", "10000") + monkeypatch.setenv("OMNIVOICE_ASR_TRANSCRIBE_TIMEOUT_S", "50000") + assert post_timeout("generate", "short") == _generate_wait( + 2000.0, queue=60.0, extension_cap=10000.0, reference_base=2000.0, + ) + + +def test_model_load_and_progress_settings_do_not_change_transcribe(post_timeout, monkeypatch): + for name in ("OMNIVOICE_MODEL_LOAD_TIMEOUT", "OMNIVOICE_PROGRESS_EXTENSION_CAP_S"): + monkeypatch.setenv(name, "20000") + assert post_timeout("transcribe") == 300.0 + GRACE + + +def test_explicit_mcp_timeout_still_wins_over_every_generate_phase(post_timeout, monkeypatch): + monkeypatch.setenv("OMNIVOICE_MCP_TIMEOUT_S", "45") + for name in ("OMNIVOICE_MODEL_LOAD_TIMEOUT", "OMNIVOICE_PROGRESS_EXTENSION_CAP_S"): + monkeypatch.setenv(name, "20000") + assert post_timeout("generate", "x" * 5000) == 45.0 \ No newline at end of file