Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,14 @@ metadata and the backend fallback mirror it.

## [Unreleased]

**Highlights**

- MCP speech tools stay connected through cold starts and slow, progressing renders (#2612)

### Fixed
Comment thread
debpalash marked this conversation as resolved.

- MCP speech tools wait through model loading and progress-extended CPU renders instead of timing out before the backend (#2609)

## [0.5.7] — 2026-10-05

**VoiceStudio now runs on PCs without a GPU and recovers instead of giving up.** Voice cloning uses the speech-to-text model you installed from Model Catalogue, GPU-less computers get the small CPU PyTorch build, and slow or busy backends are no longer reported as failed.
Expand Down
34 changes: 30 additions & 4 deletions backend/mcp_server.py
Original file line number Diff line number Diff line change
Expand Up @@ -282,6 +282,12 @@ async def _write_output(audio_id: str, raw: bytes, format: str = "wav") -> str:
# client-side timeout (#2040).
_BACKEND_GRACE_S = 30.0

# Torch-free mirrors of the desktop backstop and model_manager's guard.
# tests/test_generate_abort_budget.py keeps these in sync with their sources.
_GENERATE_SIDECAR_FLOOR_S = 900.0
_GENERATE_SIDECAR_GRACE_S = 5.0
_GENERATE_PROGRESS_BUDGETS = 3.0


def _env_seconds(name: str, default: float) -> float:
raw = os.environ.get(name, "").strip()
Expand Down Expand Up @@ -319,12 +325,32 @@ def _backend_budget_s(kind: str, text: str = "") -> float | None:
from core.generate_budget import client_execution_budget_s

execution = client_execution_budget_s(
base, len(text or ""),
max(base, _GENERATE_SIDECAR_FLOOR_S), len(text or ""),
cpu_auto_possible=not os.environ.get("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "").strip(),
) + _GENERATE_SIDECAR_GRACE_S
# Classic /generate sends no response until the whole render finishes.
# Cold loading and queueing have separate clocks; fresh chunk-progress
# heartbeats can then extend execution by up to three more budgets.
# Waiting only for queue + execution cuts off healthy CPU renders.
model_load = max(30.0, _env_seconds("OMNIVOICE_MODEL_LOAD_TIMEOUT", 1200.0))
extension_cap = _env_seconds(
"OMNIVOICE_PROGRESS_EXTENSION_CAP_S",
_env_seconds("OMNIVOICE_MODEL_LOAD_TIMEOUT_S", 1800.0),
)
extension = max(extension_cap, _GENERATE_PROGRESS_BUDGETS * execution)
queue = _env_seconds("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", 1800.0)
# A clone without a cached reference transcript first runs a separate
# guarded ASR job. That job uses generate_timeout_s("") without an
# engine: no length bonus or sidecar grace, but its own queue and
# progress extension. MCP cannot see whether the profile needs it.
reference = queue + base + max(extension_cap, _GENERATE_PROGRESS_BUDGETS * base)
return (
model_load
+ reference
+ queue
+ execution
+ extension
)
Comment thread
debpalash marked this conversation as resolved.
# A generation first waits in the GPU pool's queue, on its own clock
# (model_manager.GPU_QUEUE_TIMEOUT_S), before that budget starts.
return _env_seconds("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", 1800.0) + execution
return None


Expand Down
40 changes: 25 additions & 15 deletions docs/performance.md
Original file line number Diff line number Diff line change
Expand Up @@ -139,11 +139,12 @@ editor, profile previews, and streaming).
| The host synthesizes on the CPU **and** the text is over 1200 characters | A heads-up that this generation may exceed the time budget |
| The host synthesizes on Apple Silicon (MPS) **and** the text is over 1200 characters | The same heads-up — MPS gets the accelerated-host budget (`OMNIVOICE_GENERATE_TIMEOUT_S`), which a long render can still legitimately exceed |

**Why 1200 characters:** it is the same figure the budget itself uses. The first
1200 characters get the flat base budget, and only past that does the budget
start growing (+1 s per 40 characters). Below the threshold you are inside a
budget the backend already considers generous, so ordinary sentences on a CPU
laptop stay quiet.
**Why 1200 characters:** this advisory threshold matches the free allowance in
the legacy accelerated/explicit-budget rule: the first 1200 characters get the
flat base, then the budget grows by 1 s per 40 characters. Default CPU budgeting
uses a separate rule: its 4 s per character exceeds the 600 s floor above 150
characters, so a 400-character passage receives 1600 s even though no length
warning appears. The warning threshold itself is unchanged.

**Which base applies:**

Expand All @@ -156,16 +157,25 @@ laptop stay quiet.
**CPU hosts scale much faster than the +1 s per 40 characters.** A CPU render is
often 10-50x slower than on a GPU, so while `OMNIVOICE_CPU_GENERATE_TIMEOUT_S` is
left at its default the budget grows at 4 s per input character (a 400-character
passage gets about 27 minutes), up to a 2-hour ceiling that still catches a
genuinely wedged engine. Each streamed chunk is budgeted from its own text, and
loading the model is not part of this clock. Setting the CPU budget explicitly
turns this scaling off and uses your value as the floor (plus the standard
+1 s per 40 characters) — an explicit setting is always authoritative.

The desktop app and MCP tools never wait less than the backend does: because the
backend budgets the text *after* number normalization (a six-digit number grows
about 11x), a CPU host on the default budget reports its 2-hour ceiling and
clients wait for that rather than guessing from the typed length.
passage gets about 27 minutes), capped at 2 hours of base compute allowance.
Queueing, model loading and the existing progress-extension allowance are
separate. Each streamed chunk is budgeted from its own text; a silent, wedged job
exhausts its compute allowance. Setting the CPU budget explicitly turns this
scaling off and uses your value as the floor (plus the standard +1 s per 40
characters) — an explicit setting is always authoritative.

The desktop backstop accounts for the reported automatic CPU ceiling on local
CPU-routed jobs with the default budget. MCP tools conservatively allow that
ceiling whenever the CPU budget is not explicitly set. Both waits also include
model-load, queue, sidecar and progress-extension allowances. The ceiling avoids
guessing the compute budget from typed text that number normalization or
pronunciation rules can expand before synthesis.

MCP generation also allows a separate reference-transcription job before
synthesis for clone profiles without a cached transcript. That job uses the
generation base budget, its own queue and progress extension; it does not use
the standalone transcription timeout. MCP includes this allowance conservatively
because it cannot inspect the backend's cached reference transcript.

Both rows above can be overridden, and the two vars are independent:

Expand Down
5 changes: 5 additions & 0 deletions tests/test_generate_abort_budget.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ def _source_default(path: Path, env: str) -> float:
def test_client_budget_mirrors_backend_defaults(monkeypatch):
from services import model_manager
from worker import deadlines
import mcp_server

monkeypatch.delenv("OMNIVOICE_MODEL_LOAD_TIMEOUT", raising=False)
manager = ROOT / "backend/services/model_manager.py"
Expand All @@ -39,6 +40,9 @@ def test_client_budget_mirrors_backend_defaults(monkeypatch):
assert client["queueWait"] == _source_default(manager, "OMNIVOICE_GPU_QUEUE_TIMEOUT_S")
assert client["progressExtensionCap"] == model_manager.progress_extension_cap_s({})
assert client["progressExtensionBudgets"] == model_manager.PROGRESS_EXTENSION_BUDGETS
assert mcp_server._GENERATE_PROGRESS_BUDGETS == model_manager.PROGRESS_EXTENSION_BUDGETS
assert mcp_server._GENERATE_SIDECAR_FLOOR_S == client["executionBase"]
assert mcp_server._GENERATE_SIDECAR_GRACE_S == client["sidecarGrace"]
assert client["freeChars"] == deadlines._FREE_CHARS
assert client["charsPerSecond"] == deadlines._CHARS_PER_SECOND
from core import generate_budget as gb
Expand Down Expand Up @@ -69,3 +73,4 @@ def test_client_execution_base_covers_every_default_execution_budget():
]
assert len(bases) > 3, "receive-timeout scan found nothing; fix the regex"
assert _client_budget()["executionBase"] >= max(bases)

94 changes: 87 additions & 7 deletions tests/test_mcp_timeouts_2040.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,11 +9,23 @@
"OMNIVOICE_GENERATE_TIMEOUT_S",
"OMNIVOICE_CPU_GENERATE_TIMEOUT_S",
"OMNIVOICE_GPU_QUEUE_TIMEOUT_S",
"OMNIVOICE_MODEL_LOAD_TIMEOUT",
"OMNIVOICE_PROGRESS_EXTENSION_CAP_S",
"OMNIVOICE_MODEL_LOAD_TIMEOUT_S",
)
QUEUE = 1800.0
GRACE = 30.0


def _generate_wait(
execution, *, queue=QUEUE, model_load=1200.0, extension_cap=1800.0,
reference_base=600.0,
):
execution += 5.0 # sidecar watchdog grace
reference = queue + reference_base + max(extension_cap, 3.0 * reference_base)
return model_load + reference + queue + execution + max(extension_cap, 3.0 * execution) + GRACE


@pytest.fixture
def post_timeout(monkeypatch):
for name in _BUDGET_VARS:
Expand All @@ -39,28 +51,30 @@ def test_generation_covers_the_queue_and_the_length_scaled_budget(post_timeout):
# waits for that ceiling whatever the typed length.
from core.generate_budget import CPU_AUTO_CAP_S

assert post_timeout("generate", "short") == QUEUE + CPU_AUTO_CAP_S + GRACE
assert post_timeout("generate", "x" * 1600) == QUEUE + CPU_AUTO_CAP_S + GRACE
assert post_timeout("generate", "short") == _generate_wait(CPU_AUTO_CAP_S)
assert post_timeout("generate", "x" * 1600) == _generate_wait(CPU_AUTO_CAP_S)


def test_the_larger_cpu_generation_budget_wins(post_timeout, monkeypatch):
monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "900")
assert post_timeout("generate", "short") == QUEUE + 900.0 + GRACE
assert post_timeout("generate", "short") == _generate_wait(900.0, reference_base=900.0)
# An explicit CPU budget is authoritative and uncapped: no ceiling applies.
assert post_timeout("generate", "x" * 1600) == QUEUE + 900.0 + (1600 * 16 - 1200) / 40.0 + GRACE
assert post_timeout("generate", "x" * 1600) == _generate_wait(
900.0 + (1600 * 16 - 1200) / 40.0, reference_base=900.0,
)


def test_a_raised_gpu_generation_budget_wins_when_larger(post_timeout, monkeypatch):
monkeypatch.setenv("OMNIVOICE_GENERATE_TIMEOUT_S", "1200")
monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "600")
assert post_timeout("generate", "short") == QUEUE + 1200.0 + GRACE
assert post_timeout("generate", "short") == _generate_wait(1200.0, reference_base=1200.0)


def test_a_shorter_queue_budget_is_followed(post_timeout, monkeypatch):
monkeypatch.setenv("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", "60")
from core.generate_budget import CPU_AUTO_CAP_S

assert post_timeout("generate", "short") == 60.0 + CPU_AUTO_CAP_S + GRACE
assert post_timeout("generate", "short") == _generate_wait(CPU_AUTO_CAP_S, queue=60.0)


def test_an_explicit_mcp_timeout_still_wins(post_timeout, monkeypatch):
Expand Down Expand Up @@ -94,5 +108,71 @@ def test_transcribe_never_gives_up_before_the_backend(post_timeout):
def test_generation_never_gives_up_before_the_backend(post_timeout, device, text):
from services import model_manager as mm

backend_worst = mm.GPU_QUEUE_TIMEOUT_S + mm.generate_timeout_s(text, execution_device=device)
from types import SimpleNamespace

execution = mm.generate_timeout_s(
text, execution_device=device, engine=SimpleNamespace(recv_timeout_s=900.0),
)
reference = mm.generate_timeout_s("", execution_device=device)
backend_worst = (
mm._model_load_timeout() + 2 * mm.GPU_QUEUE_TIMEOUT_S + reference
+ max(mm.progress_extension_cap_s(), mm.PROGRESS_EXTENSION_BUDGETS * reference)
+ execution
+ max(mm.progress_extension_cap_s(), mm.PROGRESS_EXTENSION_BUDGETS * execution)
)
assert post_timeout("generate", text) > backend_worst


def test_cpu_generation_covers_cold_load_and_progress_extensions(post_timeout):
# Fail-before: queue + base + grace was 9,030 s, but a progressing CPU
# render can legally use 28,800 s of compute alone. Sidecar grace is
# included before sizing its extension, just as in the real guard.
assert post_timeout("generate", "x" * 2000) == 36_050.0


def test_a_raised_model_load_budget_is_followed(post_timeout, monkeypatch):
monkeypatch.setenv("OMNIVOICE_MODEL_LOAD_TIMEOUT", "2000")
assert post_timeout("generate", "short") == 36_850.0


@pytest.mark.parametrize("primary", [None, "", "10000"])
def test_progress_extension_cap_honors_the_legacy_alias(post_timeout, monkeypatch, primary):
monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "900")
monkeypatch.setenv("OMNIVOICE_MODEL_LOAD_TIMEOUT_S", "20000")
if primary is not None:
monkeypatch.setenv("OMNIVOICE_PROGRESS_EXTENSION_CAP_S", primary)
cap = 10000.0 if primary else 20000.0
assert post_timeout("generate", "short") == _generate_wait(
900.0, extension_cap=cap, reference_base=900.0,
)


def test_transcriptless_clone_covers_both_serial_jobs(post_timeout):
# Fail-before: the one-job wait was 31,850 s, but reference ASR and
# synthesis can together consume 36,000 s under their separate guards.
reference = QUEUE + 600.0 + 1800.0
synthesis = QUEUE + 7200.0 + 21_600.0
assert post_timeout("generate", "x" * 2000) > 1200.0 + reference + synthesis


def test_reference_transcription_uses_generation_not_standalone_asr_budget(post_timeout, monkeypatch):
monkeypatch.setenv("OMNIVOICE_CPU_GENERATE_TIMEOUT_S", "2000")
monkeypatch.setenv("OMNIVOICE_GPU_QUEUE_TIMEOUT_S", "60")
monkeypatch.setenv("OMNIVOICE_PROGRESS_EXTENSION_CAP_S", "10000")
monkeypatch.setenv("OMNIVOICE_ASR_TRANSCRIBE_TIMEOUT_S", "50000")
assert post_timeout("generate", "short") == _generate_wait(
2000.0, queue=60.0, extension_cap=10000.0, reference_base=2000.0,
)


def test_model_load_and_progress_settings_do_not_change_transcribe(post_timeout, monkeypatch):
for name in ("OMNIVOICE_MODEL_LOAD_TIMEOUT", "OMNIVOICE_PROGRESS_EXTENSION_CAP_S"):
monkeypatch.setenv(name, "20000")
assert post_timeout("transcribe") == 300.0 + GRACE


def test_explicit_mcp_timeout_still_wins_over_every_generate_phase(post_timeout, monkeypatch):
monkeypatch.setenv("OMNIVOICE_MCP_TIMEOUT_S", "45")
for name in ("OMNIVOICE_MODEL_LOAD_TIMEOUT", "OMNIVOICE_PROGRESS_EXTENSION_CAP_S"):
monkeypatch.setenv(name, "20000")
assert post_timeout("generate", "x" * 5000) == 45.0
Loading