From e26f2afbbf6009db715dc4f6ca3840d0de125c37 Mon Sep 17 00:00:00 2001 From: Palash Debnath <4178343+debpalash@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:52:03 +0530 Subject: [PATCH 1/2] fix(mcp): wait through complete generation phase budgets --- CHANGELOG.md | 1 + backend/mcp_server.py | 27 ++++++++++-- docs/performance.md | 35 +++++++++------- tests/test_generate_abort_budget.py | 5 +++ tests/test_mcp_timeouts_2040.py | 65 +++++++++++++++++++++++++---- 5 files changed, 108 insertions(+), 25 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3a4264cf8..dafb7e461 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -103,6 +103,7 @@ metadata and the backend fallback mirror it. - License notice: commercial use is free under the AGPL; the paid licence is for closed-source use, with Pro plans linked (#2578) ### Fixed +- MCP speech tools wait through model loading and progress-extended CPU renders instead of timing out before the backend (#2609) - Fix it with an agent finds Claude Code and other agent CLIs installed in user folders (~/.local/bin, Homebrew, npm global) when launched from Finder or a desktop entry, and rechecks each time the panel opens (#2602) — thanks @flatlinebb on Discord! - YouTube downloads no longer fail with Errno 22 or Broken pipe when the app's stdout is closed; yt-dlp progress and messages now go to the log (#2602) — thanks @marioteka and @shizzy_prod on Discord! - Offline NLLB translation accepts every language in the dubbing pickers (Nepali, Catalan, Latvian, Georgian, Punjabi, Norwegian and about 50 more) for both target and auto-detected source (#2602) — thanks @lamomg on Discord! diff --git a/backend/mcp_server.py b/backend/mcp_server.py index a0ffd113c..0546d31d0 100644 --- a/backend/mcp_server.py +++ b/backend/mcp_server.py @@ -282,6 +282,12 @@ async def _write_output(audio_id: str, raw: bytes, format: str = "wav") -> str: # client-side timeout (#2040). _BACKEND_GRACE_S = 30.0 +# Torch-free mirrors of the desktop backstop and model_manager's guard. +# tests/test_generate_abort_budget.py keeps these in sync with their sources. +_GENERATE_SIDECAR_FLOOR_S = 900.0 +_GENERATE_SIDECAR_GRACE_S = 5.0 +_GENERATE_PROGRESS_BUDGETS = 3.0 + def _env_seconds(name: str, default: float) -> float: raw = os.environ.get(name, "").strip() @@ -311,6 +317,7 @@ def _backend_budget_s(kind: str, text: str = "") -> float | None: base = max( _env_seconds("OMNIVOICE_GENERATE_TIMEOUT_S", 300.0), _env_seconds("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", 600.0), + _GENERATE_SIDECAR_FLOOR_S, ) # Shared with the backend's rule (core.generate_budget): covers the # legacy length bonus AND the automatic CPU ceiling (the backend budgets @@ -321,10 +328,23 @@ def _backend_budget_s(kind: str, text: str = "") -> float | None: execution = client_execution_budget_s( base, len(text or ""), cpu_auto_possible=not os.environ.get("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "").strip(), + ) + _GENERATE_SIDECAR_GRACE_S + # Classic /generate sends no response until the whole render finishes. + # Cold loading and queueing have separate clocks; fresh chunk-progress + # heartbeats can then extend execution by up to three more budgets. + # Waiting only for queue + execution cuts off healthy CPU renders. + model_load = max(30.0, _env_seconds("OMNIVOICE_MODEL_LOAD_TIMEOUT", 1200.0)) + extension_cap = _env_seconds( + "OMNIVOICE_PROGRESS_EXTENSION_CAP_S", + _env_seconds("OMNIVOICE_MODEL_LOAD_TIMEOUT_S", 1800.0), + ) + extension = max(extension_cap, _GENERATE_PROGRESS_BUDGETS * execution) + return ( + model_load + + _env_seconds("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", 1800.0) + + execution + + extension ) - # A generation first waits in the GPU pool's queue, on its own clock - # (model_manager.GPU_QUEUE_TIMEOUT_S), before that budget starts. - return _env_seconds("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", 1800.0) + execution return None @@ -963,3 +983,4 @@ def main(): if __name__ == "__main__": main() + diff --git a/docs/performance.md b/docs/performance.md index 1b55f0ed0..9a1a98b8f 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -139,11 +139,12 @@ editor, profile previews, and streaming). | The host synthesizes on the CPU **and** the text is over 1200 characters | A heads-up that this generation may exceed the time budget | | The host synthesizes on Apple Silicon (MPS) **and** the text is over 1200 characters | The same heads-up — MPS gets the accelerated-host budget (`OMNIVOICE_GENERATE_TIMEOUT_S`), which a long render can still legitimately exceed | -**Why 1200 characters:** it is the same figure the budget itself uses. The first -1200 characters get the flat base budget, and only past that does the budget -start growing (+1 s per 40 characters). Below the threshold you are inside a -budget the backend already considers generous, so ordinary sentences on a CPU -laptop stay quiet. +**Why 1200 characters:** this advisory threshold matches the free allowance in +the legacy accelerated/explicit-budget rule: the first 1200 characters get the +flat base, then the budget grows by 1 s per 40 characters. Default CPU budgeting +uses a separate rule: its 4 s per character exceeds the 600 s floor above 150 +characters, so a 400-character passage receives 1600 s even though no length +warning appears. The warning threshold itself is unchanged. **Which base applies:** @@ -156,16 +157,19 @@ laptop stay quiet. **CPU hosts scale much faster than the +1 s per 40 characters.** A CPU render is often 10-50x slower than on a GPU, so while `OMNIVOICE_CPU_GENERATE_TIMEOUT_S` is left at its default the budget grows at 4 s per input character (a 400-character -passage gets about 27 minutes), up to a 2-hour ceiling that still catches a -genuinely wedged engine. Each streamed chunk is budgeted from its own text, and -loading the model is not part of this clock. Setting the CPU budget explicitly -turns this scaling off and uses your value as the floor (plus the standard -+1 s per 40 characters) — an explicit setting is always authoritative. - -The desktop app and MCP tools never wait less than the backend does: because the -backend budgets the text *after* number normalization (a six-digit number grows -about 11x), a CPU host on the default budget reports its 2-hour ceiling and -clients wait for that rather than guessing from the typed length. +passage gets about 27 minutes), capped at 2 hours of base compute allowance. +Queueing, model loading and the existing progress-extension allowance are +separate. Each streamed chunk is budgeted from its own text; a silent, wedged job +exhausts its compute allowance. Setting the CPU budget explicitly turns this +scaling off and uses your value as the floor (plus the standard +1 s per 40 +characters) — an explicit setting is always authoritative. + +The desktop backstop accounts for the reported automatic CPU ceiling on local +CPU-routed jobs with the default budget. MCP tools conservatively allow that +ceiling whenever the CPU budget is not explicitly set. Both waits also include +model-load, queue, sidecar and progress-extension allowances. The ceiling avoids +guessing the compute budget from typed text that number normalization or +pronunciation rules can expand before synthesis. Both rows above can be overridden, and the two vars are independent: @@ -391,3 +395,4 @@ with hardware-independent budgets: one synthesis per chunk, one assembly, one final Studio effects pass, and linear copied sample volume. No model download or wall-clock speed threshold is involved. These complement the streaming/dub budgets in `tests/test_perf_operation_budgets.py`. + diff --git a/tests/test_generate_abort_budget.py b/tests/test_generate_abort_budget.py index aca8cc027..ddb74de3b 100644 --- a/tests/test_generate_abort_budget.py +++ b/tests/test_generate_abort_budget.py @@ -30,6 +30,7 @@ def _source_default(path: Path, env: str) -> float: def test_client_budget_mirrors_backend_defaults(monkeypatch): from services import model_manager from worker import deadlines + import mcp_server monkeypatch.delenv("OMNIVOICE_MODEL_LOAD_TIMEOUT", raising=False) manager = ROOT / "backend/services/model_manager.py" @@ -39,6 +40,9 @@ def test_client_budget_mirrors_backend_defaults(monkeypatch): assert client["queueWait"] == _source_default(manager, "OMNIVOICE_GPU_QUEUE_TIMEOUT_S") assert client["progressExtensionCap"] == model_manager.progress_extension_cap_s({}) assert client["progressExtensionBudgets"] == model_manager.PROGRESS_EXTENSION_BUDGETS + assert mcp_server._GENERATE_PROGRESS_BUDGETS == model_manager.PROGRESS_EXTENSION_BUDGETS + assert mcp_server._GENERATE_SIDECAR_FLOOR_S == client["executionBase"] + assert mcp_server._GENERATE_SIDECAR_GRACE_S == client["sidecarGrace"] assert client["freeChars"] == deadlines._FREE_CHARS assert client["charsPerSecond"] == deadlines._CHARS_PER_SECOND from core import generate_budget as gb @@ -69,3 +73,4 @@ def test_client_execution_base_covers_every_default_execution_budget(): ] assert len(bases) > 3, "receive-timeout scan found nothing; fix the regex" assert _client_budget()["executionBase"] >= max(bases) + diff --git a/tests/test_mcp_timeouts_2040.py b/tests/test_mcp_timeouts_2040.py index 668cb1ff3..3a20485e5 100644 --- a/tests/test_mcp_timeouts_2040.py +++ b/tests/test_mcp_timeouts_2040.py @@ -9,11 +9,19 @@ "OMNIVOICE_GENERATE_TIMEOUT_S", "OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "OMNIVOICE_GPU_QUEUE_TIMEOUT_S", + "OMNIVOICE_MODEL_LOAD_TIMEOUT", + "OMNIVOICE_PROGRESS_EXTENSION_CAP_S", + "OMNIVOICE_MODEL_LOAD_TIMEOUT_S", ) QUEUE = 1800.0 GRACE = 30.0 +def _generate_wait(execution, *, queue=QUEUE, model_load=1200.0, extension_cap=1800.0): + execution += 5.0 # sidecar watchdog grace + return model_load + queue + execution + max(extension_cap, 3.0 * execution) + GRACE + + @pytest.fixture def post_timeout(monkeypatch): for name in _BUDGET_VARS: @@ -39,28 +47,28 @@ def test_generation_covers_the_queue_and_the_length_scaled_budget(post_timeout): # waits for that ceiling whatever the typed length. from core.generate_budget import CPU_AUTO_CAP_S - assert post_timeout("generate", "short") == QUEUE + CPU_AUTO_CAP_S + GRACE - assert post_timeout("generate", "x" * 1600) == QUEUE + CPU_AUTO_CAP_S + GRACE + assert post_timeout("generate", "short") == _generate_wait(CPU_AUTO_CAP_S) + assert post_timeout("generate", "x" * 1600) == _generate_wait(CPU_AUTO_CAP_S) def test_the_larger_cpu_generation_budget_wins(post_timeout, monkeypatch): monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "900") - assert post_timeout("generate", "short") == QUEUE + 900.0 + GRACE + assert post_timeout("generate", "short") == _generate_wait(900.0) # An explicit CPU budget is authoritative and uncapped: no ceiling applies. - assert post_timeout("generate", "x" * 1600) == QUEUE + 900.0 + (1600 * 16 - 1200) / 40.0 + GRACE + assert post_timeout("generate", "x" * 1600) == _generate_wait(900.0 + (1600 * 16 - 1200) / 40.0) def test_a_raised_gpu_generation_budget_wins_when_larger(post_timeout, monkeypatch): monkeypatch.setenv("OMNIVOICE_GENERATE_TIMEOUT_S", "1200") monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "600") - assert post_timeout("generate", "short") == QUEUE + 1200.0 + GRACE + assert post_timeout("generate", "short") == _generate_wait(1200.0) def test_a_shorter_queue_budget_is_followed(post_timeout, monkeypatch): monkeypatch.setenv("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", "60") from core.generate_budget import CPU_AUTO_CAP_S - assert post_timeout("generate", "short") == 60.0 + CPU_AUTO_CAP_S + GRACE + assert post_timeout("generate", "short") == _generate_wait(CPU_AUTO_CAP_S, queue=60.0) def test_an_explicit_mcp_timeout_still_wins(post_timeout, monkeypatch): @@ -94,5 +102,48 @@ def test_transcribe_never_gives_up_before_the_backend(post_timeout): def test_generation_never_gives_up_before_the_backend(post_timeout, device, text): from services import model_manager as mm - backend_worst = mm.GPU_QUEUE_TIMEOUT_S + mm.generate_timeout_s(text, execution_device=device) + from types import SimpleNamespace + + execution = mm.generate_timeout_s( + text, execution_device=device, engine=SimpleNamespace(recv_timeout_s=900.0), + ) + backend_worst = ( + mm._model_load_timeout() + mm.GPU_QUEUE_TIMEOUT_S + execution + + max(mm.progress_extension_cap_s(), mm.PROGRESS_EXTENSION_BUDGETS * execution) + ) assert post_timeout("generate", text) > backend_worst + + +def test_cpu_generation_covers_cold_load_and_progress_extensions(post_timeout): + # Fail-before: queue + base + grace was 9,030 s, but a progressing CPU + # render can legally use 28,800 s of compute alone. Sidecar grace is + # included before sizing its extension, just as in the real guard. + assert post_timeout("generate", "x" * 2000) == 31_850.0 + + +def test_a_raised_model_load_budget_is_followed(post_timeout, monkeypatch): + monkeypatch.setenv("OMNIVOICE_MODEL_LOAD_TIMEOUT", "2000") + assert post_timeout("generate", "short") == 32_650.0 + + +@pytest.mark.parametrize("primary", [None, "", "10000"]) +def test_progress_extension_cap_honors_the_legacy_alias(post_timeout, monkeypatch, primary): + monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "900") + monkeypatch.setenv("OMNIVOICE_MODEL_LOAD_TIMEOUT_S", "20000") + if primary is not None: + monkeypatch.setenv("OMNIVOICE_PROGRESS_EXTENSION_CAP_S", primary) + cap = 10000.0 if primary else 20000.0 + assert post_timeout("generate", "short") == _generate_wait(900.0, extension_cap=cap) + + +def test_model_load_and_progress_settings_do_not_change_transcribe(post_timeout, monkeypatch): + for name in ("OMNIVOICE_MODEL_LOAD_TIMEOUT", "OMNIVOICE_PROGRESS_EXTENSION_CAP_S"): + monkeypatch.setenv(name, "20000") + assert post_timeout("transcribe") == 300.0 + GRACE + + +def test_explicit_mcp_timeout_still_wins_over_every_generate_phase(post_timeout, monkeypatch): + monkeypatch.setenv("OMNIVOICE_MCP_TIMEOUT_S", "45") + for name in ("OMNIVOICE_MODEL_LOAD_TIMEOUT", "OMNIVOICE_PROGRESS_EXTENSION_CAP_S"): + monkeypatch.setenv(name, "20000") + assert post_timeout("generate", "x" * 5000) == 45.0 From cde96096c53b0131b4daac91b2e97b684244b70e Mon Sep 17 00:00:00 2001 From: Palash Debnath <4178343+debpalash@users.noreply.github.com> Date: Mon, 5 Oct 2026 17:48:59 +0000 Subject: [PATCH 2/2] fix(mcp): include serial reference transcription in generation wait Cover each guarded job's queue and progress-extension allowance, add two-job regressions, and restore Unreleased highlights. --- CHANGELOG.md | 4 +++ backend/mcp_server.py | 13 ++++++--- docs/performance.md | 7 ++++- tests/test_mcp_timeouts_2040.py | 49 ++++++++++++++++++++++++++------- 4 files changed, 58 insertions(+), 15 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 84277ea56..534142f5a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,10 @@ metadata and the backend fallback mirror it. ## [Unreleased] +**Highlights** + +- MCP speech tools stay connected through cold starts and slow, progressing renders (#2612) + ### Fixed - MCP speech tools wait through model loading and progress-extended CPU renders instead of timing out before the backend (#2609) diff --git a/backend/mcp_server.py b/backend/mcp_server.py index 0546d31d0..b43e36562 100644 --- a/backend/mcp_server.py +++ b/backend/mcp_server.py @@ -317,7 +317,6 @@ def _backend_budget_s(kind: str, text: str = "") -> float | None: base = max( _env_seconds("OMNIVOICE_GENERATE_TIMEOUT_S", 300.0), _env_seconds("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", 600.0), - _GENERATE_SIDECAR_FLOOR_S, ) # Shared with the backend's rule (core.generate_budget): covers the # legacy length bonus AND the automatic CPU ceiling (the backend budgets @@ -326,7 +325,7 @@ def _backend_budget_s(kind: str, text: str = "") -> float | None: from core.generate_budget import client_execution_budget_s execution = client_execution_budget_s( - base, len(text or ""), + max(base, _GENERATE_SIDECAR_FLOOR_S), len(text or ""), cpu_auto_possible=not os.environ.get("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "").strip(), ) + _GENERATE_SIDECAR_GRACE_S # Classic /generate sends no response until the whole render finishes. @@ -339,9 +338,16 @@ def _backend_budget_s(kind: str, text: str = "") -> float | None: _env_seconds("OMNIVOICE_MODEL_LOAD_TIMEOUT_S", 1800.0), ) extension = max(extension_cap, _GENERATE_PROGRESS_BUDGETS * execution) + queue = _env_seconds("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", 1800.0) + # A clone without a cached reference transcript first runs a separate + # guarded ASR job. That job uses generate_timeout_s("") without an + # engine: no length bonus or sidecar grace, but its own queue and + # progress extension. MCP cannot see whether the profile needs it. + reference = queue + base + max(extension_cap, _GENERATE_PROGRESS_BUDGETS * base) return ( model_load - + _env_seconds("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", 1800.0) + + reference + + queue + execution + extension ) @@ -983,4 +989,3 @@ def main(): if __name__ == "__main__": main() - diff --git a/docs/performance.md b/docs/performance.md index 9a1a98b8f..5b64950a5 100644 --- a/docs/performance.md +++ b/docs/performance.md @@ -171,6 +171,12 @@ model-load, queue, sidecar and progress-extension allowances. The ceiling avoids guessing the compute budget from typed text that number normalization or pronunciation rules can expand before synthesis. +MCP generation also allows a separate reference-transcription job before +synthesis for clone profiles without a cached transcript. That job uses the +generation base budget, its own queue and progress extension; it does not use +the standalone transcription timeout. MCP includes this allowance conservatively +because it cannot inspect the backend's cached reference transcript. + Both rows above can be overridden, and the two vars are independent: - An explicit `OMNIVOICE_CPU_GENERATE_TIMEOUT_S` always governs CPU-family @@ -395,4 +401,3 @@ with hardware-independent budgets: one synthesis per chunk, one assembly, one final Studio effects pass, and linear copied sample volume. No model download or wall-clock speed threshold is involved. These complement the streaming/dub budgets in `tests/test_perf_operation_budgets.py`. - diff --git a/tests/test_mcp_timeouts_2040.py b/tests/test_mcp_timeouts_2040.py index 3a20485e5..b4b4c17ca 100644 --- a/tests/test_mcp_timeouts_2040.py +++ b/tests/test_mcp_timeouts_2040.py @@ -17,9 +17,13 @@ GRACE = 30.0 -def _generate_wait(execution, *, queue=QUEUE, model_load=1200.0, extension_cap=1800.0): +def _generate_wait( + execution, *, queue=QUEUE, model_load=1200.0, extension_cap=1800.0, + reference_base=600.0, +): execution += 5.0 # sidecar watchdog grace - return model_load + queue + execution + max(extension_cap, 3.0 * execution) + GRACE + reference = queue + reference_base + max(extension_cap, 3.0 * reference_base) + return model_load + reference + queue + execution + max(extension_cap, 3.0 * execution) + GRACE @pytest.fixture @@ -53,15 +57,17 @@ def test_generation_covers_the_queue_and_the_length_scaled_budget(post_timeout): def test_the_larger_cpu_generation_budget_wins(post_timeout, monkeypatch): monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "900") - assert post_timeout("generate", "short") == _generate_wait(900.0) + assert post_timeout("generate", "short") == _generate_wait(900.0, reference_base=900.0) # An explicit CPU budget is authoritative and uncapped: no ceiling applies. - assert post_timeout("generate", "x" * 1600) == _generate_wait(900.0 + (1600 * 16 - 1200) / 40.0) + assert post_timeout("generate", "x" * 1600) == _generate_wait( + 900.0 + (1600 * 16 - 1200) / 40.0, reference_base=900.0, + ) def test_a_raised_gpu_generation_budget_wins_when_larger(post_timeout, monkeypatch): monkeypatch.setenv("OMNIVOICE_GENERATE_TIMEOUT_S", "1200") monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "600") - assert post_timeout("generate", "short") == _generate_wait(1200.0) + assert post_timeout("generate", "short") == _generate_wait(1200.0, reference_base=1200.0) def test_a_shorter_queue_budget_is_followed(post_timeout, monkeypatch): @@ -107,8 +113,11 @@ def test_generation_never_gives_up_before_the_backend(post_timeout, device, text execution = mm.generate_timeout_s( text, execution_device=device, engine=SimpleNamespace(recv_timeout_s=900.0), ) + reference = mm.generate_timeout_s("", execution_device=device) backend_worst = ( - mm._model_load_timeout() + mm.GPU_QUEUE_TIMEOUT_S + execution + mm._model_load_timeout() + 2 * mm.GPU_QUEUE_TIMEOUT_S + reference + + max(mm.progress_extension_cap_s(), mm.PROGRESS_EXTENSION_BUDGETS * reference) + + execution + max(mm.progress_extension_cap_s(), mm.PROGRESS_EXTENSION_BUDGETS * execution) ) assert post_timeout("generate", text) > backend_worst @@ -118,12 +127,12 @@ def test_cpu_generation_covers_cold_load_and_progress_extensions(post_timeout): # Fail-before: queue + base + grace was 9,030 s, but a progressing CPU # render can legally use 28,800 s of compute alone. Sidecar grace is # included before sizing its extension, just as in the real guard. - assert post_timeout("generate", "x" * 2000) == 31_850.0 + assert post_timeout("generate", "x" * 2000) == 36_050.0 def test_a_raised_model_load_budget_is_followed(post_timeout, monkeypatch): monkeypatch.setenv("OMNIVOICE_MODEL_LOAD_TIMEOUT", "2000") - assert post_timeout("generate", "short") == 32_650.0 + assert post_timeout("generate", "short") == 36_850.0 @pytest.mark.parametrize("primary", [None, "", "10000"]) @@ -133,7 +142,27 @@ def test_progress_extension_cap_honors_the_legacy_alias(post_timeout, monkeypatc if primary is not None: monkeypatch.setenv("OMNIVOICE_PROGRESS_EXTENSION_CAP_S", primary) cap = 10000.0 if primary else 20000.0 - assert post_timeout("generate", "short") == _generate_wait(900.0, extension_cap=cap) + assert post_timeout("generate", "short") == _generate_wait( + 900.0, extension_cap=cap, reference_base=900.0, + ) + + +def test_transcriptless_clone_covers_both_serial_jobs(post_timeout): + # Fail-before: the one-job wait was 31,850 s, but reference ASR and + # synthesis can together consume 36,000 s under their separate guards. + reference = QUEUE + 600.0 + 1800.0 + synthesis = QUEUE + 7200.0 + 21_600.0 + assert post_timeout("generate", "x" * 2000) > 1200.0 + reference + synthesis + + +def test_reference_transcription_uses_generation_not_standalone_asr_budget(post_timeout, monkeypatch): + monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "2000") + monkeypatch.setenv("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", "60") + monkeypatch.setenv("OMNIVOICE_PROGRESS_EXTENSION_CAP_S", "10000") + monkeypatch.setenv("OMNIVOICE_ASR_TRANSCRIBE_TIMEOUT_S", "50000") + assert post_timeout("generate", "short") == _generate_wait( + 2000.0, queue=60.0, extension_cap=10000.0, reference_base=2000.0, + ) def test_model_load_and_progress_settings_do_not_change_transcribe(post_timeout, monkeypatch): @@ -146,4 +175,4 @@ def test_explicit_mcp_timeout_still_wins_over_every_generate_phase(post_timeout, monkeypatch.setenv("OMNIVOICE_MCP_TIMEOUT_S", "45") for name in ("OMNIVOICE_MODEL_LOAD_TIMEOUT", "OMNIVOICE_PROGRESS_EXTENSION_CAP_S"): monkeypatch.setenv(name, "20000") - assert post_timeout("generate", "x" * 5000) == 45.0 + assert post_timeout("generate", "x" * 5000) == 45.0 \ No newline at end of file