Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 27 additions & 1 deletion s16code/events/engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,32 @@ def _json_object(text: str) -> dict[str, Any]:
return value


def _run_spend_of(result: dict[str, Any]) -> float:
"""What the run cost, read from where the runtime actually reports it.

`Runtime.run` returns spend inside `budget` — `run_budget.snapshot()`, which
carries `total` and `spent`. There is no `spend_usd` key; reading one recorded
every run at $0.00, so `daily_budget` could never bind, a run's ceiling was
never squeezed by what the window had left, and the morning report showed a
night that cost nothing.

`spend_usd` is still honoured, so any caller or fake supplying it keeps working.
"""
direct = result.get("spend_usd")
if direct is not None:
try:
return float(direct)
except (TypeError, ValueError):
return 0.0
budget = result.get("budget")
if isinstance(budget, dict):
try:
return float(budget.get("spent") or 0.0)
except (TypeError, ValueError):
return 0.0
return 0.0


def _spend_of(reply: dict[str, Any]) -> float:
"""What the gate itself cost. Deciding not to act is not free."""
total = 0.0
Expand Down Expand Up @@ -145,7 +171,7 @@ async def decide(subscription: Subscription) -> dict[str, Any]:
initial_evidence={"event": event.model_dump(mode="json")},
)
self.governor.record(subscription.id, kind="run",
usd=float(result.get("spend_usd") or 0.0), now=now)
usd=_run_spend_of(result), now=now)
decision.update(run_id=result["run_id"], run_status=result["status"], acted=True)
self.store.add_decision(event.source, event.id, decision)
return decision
Expand Down
175 changes: 175 additions & 0 deletions tests/test_run_spend_is_recorded.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,175 @@
"""`daily_budget` must be fed what a run actually cost.

`AutonomousEventEngine` records a completed run's spend like this:

self.governor.record(subscription.id, kind="run",
usd=float(result.get("spend_usd") or 0.0), now=now)

`Runtime.run()` never returns a `spend_usd` key. Its result dict is:

{"run_id", "status", "answer", "provider", "model", "graph", "trace",
"events", "principal", "budget", "economics", "allocations"}

The real figure is `result["budget"]["spent"]`. So `.get("spend_usd")` is always
None, every run is recorded at $0.00, and `window_spend(..., kind="run")` stays
zero for the life of the installation.

## Three consequences, one root cause

1. `daily_budget` can never bind. `admit_run` compares the window's run spend
against it, and that spend is always 0. Section 10 makes this the ceiling that
survives "escape the budget by starting more runs".

2. The per-run ceiling is never squeezed by the window. `admit_run` computes
`remaining = daily_budget - spent` and hands the run
`min(per_run, remaining)`; with `spent` pinned at 0, `remaining` is always the
full daily budget. Section 10 asks for "each run's ceiling shrunk by whatever
the window has left". It never shrinks.

3. The morning report's "cost of doing" is always $0.00. Observed live on
2026-08-10 after four real runs against Gemini:

- acted: 4
- cost of watching: $0.00066000
- cost of doing: $0.00000000

An operator reading Section 9's report sees a night that cost nothing.

`max_runs_per_day` still bounds the count, so this is not unbounded spend - but
the money ceiling specifically is inert, and the report understates the bill.

## Why the suite did not catch it

`tests/test_autonomy_governor.py` defines

class _Runtime:
async def run(self, *, prompt, **_):
return {"run_id": ..., "status": "completed", "spend_usd": self.spend}

The stub invents the key production never sends, so
`test_a_daily_budget_caps_the_window_and_shrinks_the_last_run` passes while the
control it covers cannot fire. Section 12: "a green check on a control that never
ran is worse than a red one." This is the second instance of that shape in the
same file - the other was `metered_calls` for triage.

The stub here deliberately mirrors the *real* `Runtime.run` contract instead.
"""

from __future__ import annotations

import json
from datetime import UTC, datetime

import pytest

from s16code.events import (
AutonomousEventEngine,
EventEnvelope,
EventStore,
Subscription,
)


def _event(**changes) -> EventEnvelope:
body = {"id": "e1", "source": "deployments", "type": "deployment.completed",
"subject": "checkout/production", "occurred_at": datetime.now(UTC),
"data": {"previous_error_rate": 0.7, "current_error_rate": 6.4}}
body.update(changes)
return EventEnvelope(**body)


def _subscription(**changes) -> Subscription:
body = {"id": "watch", "instruction": "Assess material error-rate increases.",
"event_types": ["deployment.completed"], "sources": ["deployments"],
"tenant_id": "course", "allowed_side_effects": [], "budget": 0.01}
body.update(changes)
return Subscription(**body)


class _RealShapedRuntime:
"""Returns what `s16code.runtime.Runtime.run` actually returns.

Note what is absent: `spend_usd`. Spend is reported inside `budget`, which is
`run_budget.snapshot()` - `{"total": ..., "spent": ...}`.
"""

def __init__(self, spend: float = 0.004) -> None:
self.runs: list[str] = []
self.spend = spend

async def run(self, *, prompt: str, **_: object) -> dict[str, object]:
self.runs.append(prompt)
return {
"run_id": f"run-{len(self.runs)}",
"status": "completed",
"answer": "assessed",
"provider": "gemini",
"model": "gemini-3.1-flash-lite",
"graph": {"finished": True, "nodes": {}, "edges": []},
"trace": {"planner": {}, "agents": []},
"events": [],
"principal": "course",
"budget": {"total": 0.01, "spent": self.spend},
"economics": None,
"allocations": [],
}


def _relevance_llm(cost: float = 0.00002):
async def llm(prompt: str, system: str): # noqa: ARG001
return {"text": json.dumps({"relevant": True, "reason": "error rate rose",
"goal": "Assess the increase."}),
"metered_calls": [{"cost_usd": cost}]}
return llm


def _day(moment: datetime | None = None) -> str:
return (moment or datetime.now(UTC)).strftime("%Y-%m-%d")


async def test_a_completed_run_records_what_it_spent(tmp_path) -> None:
store = EventStore(tmp_path)
runtime = _RealShapedRuntime(spend=0.004)
engine = AutonomousEventEngine(store, runtime)
store.put_subscription(_subscription())

await engine.process(_event(id="spend-1"), llm=_relevance_llm())

recorded = store.window_spend("watch", _day(), kind="run")
assert recorded == pytest.approx(0.004), (
f"the run spent $0.004 and the ledger recorded ${recorded:.8f}; the engine "
"reads result['spend_usd'], a key Runtime.run does not return"
)


async def test_daily_budget_actually_refuses_once_the_window_is_spent(tmp_path) -> None:
"""The control Section 10 relies on, exercised against the real result shape."""
store = EventStore(tmp_path)
runtime = _RealShapedRuntime(spend=0.004)
engine = AutonomousEventEngine(store, runtime)
store.put_subscription(_subscription(budget=0.004, daily_budget=0.01))

for index in range(4):
await engine.process(_event(id=f"spend-{index}"), llm=_relevance_llm())

assert len(runtime.runs) < 4, (
f"all {len(runtime.runs)} runs were admitted against a $0.01 daily budget "
"at $0.004 each; the window ceiling never fired"
)
assert any(item["control"] == "daily_budget" for item in store.refusals()), (
"no daily_budget refusal was recorded, so nothing tells an operator the "
"window stopped the work"
)


async def test_the_remaining_window_shrinks_a_later_run(tmp_path) -> None:
"""Section 10: a run's ceiling is cut down by whatever the window has left."""
store = EventStore(tmp_path)
runtime = _RealShapedRuntime(spend=0.004)
engine = AutonomousEventEngine(store, runtime)
store.put_subscription(_subscription(budget=0.01, daily_budget=0.01))

await engine.process(_event(id="first"), llm=_relevance_llm())
spent = store.window_spend("watch", _day(), kind="run")
assert spent > 0, "nothing was recorded, so no later run can be squeezed"
assert spent < 0.01