From 50631bb4ef04ec3820fc034b0349e2f09a9945d8 Mon Sep 17 00:00:00 2001 From: Kenny Thomas Date: Fri, 14 Aug 2026 07:57:57 +0100 Subject: [PATCH 1/7] fix(deps): declare tzdata so timezones resolve on Windows `current_datetime` accepts an IANA timezone and hands it to `ZoneInfo`. Windows ships no system tz database, so `zoneinfo` falls back to the `tzdata` package, which is not declared as a dependency. Every zone except UTC therefore raises `ZoneInfoNotFoundError` on a clean Windows install. That is not a corner case for this project: an assistant that cannot resolve "today" in the user's own timezone answers date questions wrongly, and the shipped test for it fails on a fresh checkout. Declared as a marked dependency rather than an unconditional one, since POSIX systems already have the database and do not need the wheel: "tzdata; platform_system == 'Windows'" A maintainer on Linux or macOS cannot reproduce this, which is worth stating explicitly in the pull request. --- pyproject.toml | 4 ++++ uv.lock | 13 ++++++++++++- 2 files changed, 16 insertions(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index d726055..4aae33b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -19,6 +19,10 @@ dependencies = [ "opentelemetry-exporter-otlp>=1.27", "ddgs>=9,<10", "networkx>=3.6,<4", + # Windows ships no system tz database, so zoneinfo falls back to this + # package. Without it current_datetime() raises for every IANA zone except + # UTC, and the agent cannot resolve "today" in the user's own timezone. + "tzdata; platform_system == 'Windows'", ] [dependency-groups] diff --git a/uv.lock b/uv.lock index 653e31e..d0c677f 100644 --- a/uv.lock +++ b/uv.lock @@ -1,5 +1,5 @@ version = 1 -revision = 2 +revision = 3 requires-python = ">=3.11" resolution-markers = [ "python_full_version >= '3.14'", @@ -458,6 +458,7 @@ dependencies = [ { name = "pydantic" }, { name = "python-dotenv" }, { name = "pyyaml" }, + { name = "tzdata", marker = "sys_platform == 'win32'" }, { name = "uvicorn", extra = ["standard"] }, ] @@ -484,6 +485,7 @@ requires-dist = [ { name = "pydantic", specifier = ">=2.6" }, { name = "python-dotenv", specifier = ">=1.0" }, { name = "pyyaml", specifier = ">=6.0" }, + { name = "tzdata", marker = "sys_platform == 'win32'" }, { name = "uvicorn", extras = ["standard"], specifier = ">=0.27" }, ] @@ -1643,6 +1645,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/dc/9b/47798a6c91d8bdb567fe2698fe81e0c6b7cb7ef4d13da4114b41d239f65d/typing_inspection-0.4.2-py3-none-any.whl", hash = "sha256:4ed1cacbdc298c220f1bd249ed5287caa16f34d44ef4e9c3d0cbad5b521545e7", size = 14611, upload-time = "2025-10-01T02:14:40.154Z" }, ] +[[package]] +name = "tzdata" +version = "2026.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/92/ff/5a28bdfd8c3ebec42564ac7d0e54ca3db65044a9314a97f9564fa7a1e926/tzdata-2026.3.tar.gz", hash = "sha256:4a1518b8993086a7982523e071643f3c0e5f213e75b21318e78bcabfff9d1415", size = 198674, upload-time = "2026-07-10T08:50:37.887Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e5/6d/b53b99a9f2766d095985947a5782f1702cabb129a34f7a802d7197af832f/tzdata-2026.3-py2.py3-none-any.whl", hash = "sha256:dc096730c87af6cab1b171c9d532be840741ff5d459015e7f6947bd7d7e54931", size = 348168, upload-time = "2026-07-10T08:50:36.46Z" }, +] + [[package]] name = "urllib3" version = "2.7.0" From 38a575d31814933b4b0af7b8c7a74bc842380590 Mon Sep 17 00:00:00 2001 From: Kenny Thomas Date: Fri, 14 Aug 2026 07:59:39 +0100 Subject: [PATCH 2/7] fix(runtime): parse file:// URIs correctly on Windows `verify_artifact` resolved a file URI with: parsed = httpx.URL(task.input["uri"]) candidate = Path(str(parsed.path)).resolve() `Path.as_uri()` puts a slash before the drive letter, so a Windows artifact URI reads `file:///C:/Users/...` and its path component is `/C:/Users/...`. Handing that to `Path()` produces `\C:\Users\...`, which is not a valid path, so every `verify_artifact` call on Windows fails on an artifact the run had just written itself. `url2pathname` does the platform-correct conversion. It also percent-decodes, which is why the value must not be unquoted first: doing both turns a `%20` in a filename into a space and then decodes the space again. The same mistake is in the shipped test, which used `uri.removeprefix("file://")` and therefore only ever passed on POSIX. Both sites are corrected, so the test now exercises the fix rather than agreeing with the bug. Invisible to a maintainer on Linux or macOS, where the path component happens to be a valid path already. Worth saying so in the pull request. Tests: tests/test_runtime_regressions.py, existing test corrected; 6 pass. --- s16code/runtime.py | 11 +++++++++-- tests/test_runtime_regressions.py | 7 ++++++- 2 files changed, 15 insertions(+), 3 deletions(-) diff --git a/s16code/runtime.py b/s16code/runtime.py index b8fcef7..aa49159 100644 --- a/s16code/runtime.py +++ b/s16code/runtime.py @@ -15,6 +15,8 @@ from datetime import date from pathlib import Path from typing import Any +from urllib.parse import urlparse +from urllib.request import url2pathname import httpx @@ -347,10 +349,15 @@ async def run_copy_file(task: TaskSpec) -> dict[str, Any]: overwrite=bool(task.input.get("overwrite", False))) async def run_verify_artifact(task: TaskSpec) -> dict[str, Any]: - parsed = httpx.URL(task.input["uri"]) + parsed = urlparse(task.input["uri"]) if parsed.scheme != "file": raise ValueError("verify_artifact requires a file:// URI") - candidate = Path(str(parsed.path)).resolve() + # Path.as_uri() puts a slash before the drive letter, so a file URI + # reads file:///C:/... and its path component is /C:/... . Handing + # that straight to Path() on Windows yields \C:\... which is not a + # valid path. url2pathname does the platform-correct conversion and + # percent-decodes, so it must not be pre-unquoted. + candidate = Path(url2pathname(parsed.path)).resolve() owned = (runtime.root / "artifacts" / run_id).resolve() if candidate == owned or owned not in candidate.parents or not candidate.is_file(): raise PermissionError("artifact is not a file owned by this run") diff --git a/tests/test_runtime_regressions.py b/tests/test_runtime_regressions.py index baee8d4..6cf0d7e 100644 --- a/tests/test_runtime_regressions.py +++ b/tests/test_runtime_regressions.py @@ -2,6 +2,8 @@ import json from pathlib import Path +from urllib.parse import urlparse +from urllib.request import url2pathname import s16code.routes as agent_route import s16code.runtime as runtime_module @@ -195,7 +197,10 @@ def decide(context): "allowed_side_effects": ["remember_explicit_fact", "create_calendar_events"]}).json() artifacts = body["graph"]["nodes"]["calendar"]["result"]["artifacts"] assert len(artifacts) == 2 - assert all(Path(uri.removeprefix("file://")).read_text().startswith("BEGIN:VCALENDAR") for uri in artifacts) + # Not removeprefix("file://"): that leaves /C:/... on Windows, which Path + # turns into the invalid \C:\... . url2pathname converts per platform. + assert all(Path(url2pathname(urlparse(uri).path)).read_text().startswith("BEGIN:VCALENDAR") + for uri in artifacts) def test_failed_file_read_is_visible_to_the_final_answer(app_client, monkeypatch, tmp_path): From b2934d717db5bd0eab76d4a3d398812f782a90e0 Mon Sep 17 00:00:00 2001 From: Kenny Thomas Date: Fri, 14 Aug 2026 08:02:07 +0100 Subject: [PATCH 3/7] fix(events): record spend so daily_budget and daily_triage_budget can enforce `daily_budget` and `daily_triage_budget` are enforced by comparing recorded spend against the ceiling. Spend was never recorded as anything but 0.0, so neither comparison could ever be true and neither ceiling could refuse anything. The morning report also showed $0.00 regardless of what was spent. `max_runs_per_day` is unaffected, because it counts rather than sums money. The agent was therefore bounded by one of its three controls rather than three. Four independent causes, all of which had to be fixed: 1. `_spend_of` looked for `cost_usd` or `usd`. The economics controller writes its per-call cost under `cost` (economics.budget.Charge), so every real metered call summed to zero. 2. `GatewayClient.complete()` discarded the gateway's `cost` object and returned only text, provider, model and token counts. The gateway prices every call and returns `cost.total_usd`; the price was known at the boundary and thrown away one layer later. This is the triage path, which is exactly what `daily_triage_budget` bounds. 3. `spend_usd` was read but never written. The engine records run spend from `result.get("spend_usd")`; grepping the package finds four references, all reads. `RunBudget.spent` had always accumulated the real figure and was simply never surfaced. 4. `chat()` dropped `cost` before `complete()` could read it. This one only became visible after 2 was fixed and live triage still recorded zero: `complete()` does not call the gateway, it calls `chat()`, which rebuilds the response as a field whitelist that never named `cost`. Measured with the fix for 2 already in place: gateway POST /v1/chat -> cost.total_usd = 1.5e-06 GatewayClient.complete -> cost_usd = 0.0, metered_calls = [] Cause 4 is worth calling out in review: the first round of tests for this fix were all green while the bug was still live, because each started from a reply dict that already carried a price. They pinned the arithmetic and never asserted that a price arrives at all. Tests: tests/test_autonomy_spend_accounting.py, 9 tests. Two stub the HTTP layer and assert the price survives `chat()` and reaches `_spend_of`, which is the seam that was actually broken. The last drives the governor directly and asserts a spent-out `daily_triage_budget` returns a refusal with that control name, which is the behaviour the ceiling exists for. --- s16code/events/engine.py | 29 +++-- s16code/gateway.py | 19 ++++ s16code/runtime.py | 4 + tests/test_autonomy_spend_accounting.py | 136 ++++++++++++++++++++++++ 4 files changed, 182 insertions(+), 6 deletions(-) create mode 100644 tests/test_autonomy_spend_accounting.py diff --git a/s16code/events/engine.py b/s16code/events/engine.py index ab32753..357129a 100644 --- a/s16code/events/engine.py +++ b/s16code/events/engine.py @@ -26,16 +26,33 @@ def _json_object(text: str) -> dict[str, Any]: def _spend_of(reply: dict[str, Any]) -> float: - """What the gate itself cost. Deciding not to act is not free.""" + """What the gate itself cost. Deciding not to act is not free. + + A metered call record is produced by the economics controller and its cost + field is named ``cost`` (see economics.budget.Charge). Reading only + ``cost_usd``/``usd`` silently returned zero for every real call, which made + ``daily_triage_budget`` unenforceable. All three spellings are accepted so + the helper works against the controller's records and the gateway client's. + """ total = 0.0 for call in reply.get("metered_calls", []) or []: if not isinstance(call, dict): continue - try: - total += float(call.get("cost_usd") or call.get("usd") or 0.0) - except (TypeError, ValueError): - continue - return total + for key in ("cost", "cost_usd", "usd"): + value = call.get(key) + if value is not None: + try: + total += float(value) + except (TypeError, ValueError): + pass + break + if total: + return total + # Fall back to a single-call reply that carries its price at the top level. + try: + return float(reply.get("cost_usd") or (reply.get("cost") or {}).get("total_usd") or 0.0) + except (TypeError, ValueError, AttributeError): + return 0.0 class AutonomousEventEngine: diff --git a/s16code/gateway.py b/s16code/gateway.py index 6cedb1d..40903db 100644 --- a/s16code/gateway.py +++ b/s16code/gateway.py @@ -94,6 +94,13 @@ async def chat( "cache_creation_input_tokens": body.get("cache_creation_input_tokens") or 0, "latency_ms": body.get("latency_ms"), "stop_reason": body.get("stop_reason"), + # The gateway is the only component that knows what a call cost: it + # owns the provider keys, the model routing and the price table. A + # caller cannot recompute this from tokens without duplicating that + # table and drifting from it. Carrying it through is what lets the + # autonomy governor enforce a budget in the currency the budget is + # written in. + "cost": body.get("cost") or {}, } async def complete( @@ -107,9 +114,21 @@ async def complete( """The unbudgeted path, for a run created without a ceiling.""" body: dict[str, Any] = {"session": session, **(request or {})} result = await self.chat(prompt=prompt, system=system, request=body) + # The gateway prices every call and returns cost.total_usd. Dropping it + # here is what made the relevance gate look free: the autonomy governor + # sums this to enforce daily_triage_budget, so with no cost reaching it + # the ceiling could never be reached and "cost of watching" was always + # zero in the morning report. + cost = result.get("cost") or {} + spend = float(cost.get("total_usd") or 0.0) return { "text": result["text"], "provider": result["provider"], "model": result["model"], "input_tokens": result["input_tokens"], "output_tokens": result["output_tokens"], + "cost": cost, "cost_usd": spend, + # Same shape a metered node result carries, so one summing helper + # works for both the triage path and the run path. + "metered_calls": [{"cost": spend, "provider": result.get("provider"), + "model": result.get("model")}] if spend else [], } async def health(self) -> dict[str, Any]: diff --git a/s16code/runtime.py b/s16code/runtime.py index aa49159..a75b231 100644 --- a/s16code/runtime.py +++ b/s16code/runtime.py @@ -970,6 +970,10 @@ async def execute_once(task: TaskSpec) -> dict[str, Any] | Deferred: "edges": snapshot.edges}, "trace": {"planner": getattr(planner, "last_selection", {"mode": "deterministic"}), "agents": trace}, "events": [event.__dict__ for event in self.graph.events(run_id)], "principal": who, + # The autonomy governor reads spend_usd to enforce daily_budget. + # It was only ever read, never written, so the ceiling summed + # zero no matter what a run actually cost. + "spend_usd": float(run_budget.spent) if run_budget is not None else 0.0, "budget": run_budget.snapshot() if run_budget is not None else None, "economics": economics_config.describe() if economics_config is not None else None, "allocations": list(getattr(planner, "allocations", []))} diff --git a/tests/test_autonomy_spend_accounting.py b/tests/test_autonomy_spend_accounting.py new file mode 100644 index 0000000..da905f3 --- /dev/null +++ b/tests/test_autonomy_spend_accounting.py @@ -0,0 +1,136 @@ +"""Spend must actually be recorded, or the window ceilings cannot enforce. + +`daily_budget` and `daily_triage_budget` are enforced by comparing recorded +spend against the ceiling. If spend is always recorded as zero the comparison +is never true, the ceilings never refuse anything, and the morning report shows +$0.00 no matter how much was spent. These tests pin the recording, not the +prose. +""" + +from __future__ import annotations + +from s16code.events.engine import _spend_of + + +class TestSpendExtraction: + """The economics controller names its cost field `cost` (economics.budget. + Charge). Reading only `cost_usd`/`usd` silently returned zero for every + real call. + """ + + def test_reads_the_controller_field_name(self) -> None: + reply = {"metered_calls": [{"cost": 0.004, "provider": "gemini"}]} + assert _spend_of(reply) == 0.004 + + def test_sums_every_call_in_a_node(self) -> None: + reply = {"metered_calls": [{"cost": 0.001}, {"cost": 0.002}, {"cost": 0.003}]} + assert abs(_spend_of(reply) - 0.006) < 1e-9 + + def test_still_reads_the_alternative_spellings(self) -> None: + assert _spend_of({"metered_calls": [{"cost_usd": 0.5}]}) == 0.5 + assert _spend_of({"metered_calls": [{"usd": 0.25}]}) == 0.25 + + def test_falls_back_to_a_top_level_price(self) -> None: + # The gateway client returns one priced call rather than a list. + assert _spend_of({"cost_usd": 0.007}) == 0.007 + assert _spend_of({"cost": {"total_usd": 0.008}}) == 0.008 + + def test_absent_or_malformed_prices_are_zero_not_an_error(self) -> None: + assert _spend_of({}) == 0.0 + assert _spend_of({"metered_calls": []}) == 0.0 + assert _spend_of({"metered_calls": [{"cost": None}, "not-a-dict"]}) == 0.0 + assert _spend_of({"metered_calls": [{"cost": "free"}]}) == 0.0 + + + +class TestTheGatewayPriceSurvivesTheClient: + """`_spend_of` can only read a price that reached it. + + The gateway prices every call and returns `cost.total_usd`; it is the only + component that can, since it owns the routing and the price table. But + `chat()` rebuilds the response field by field, and a field it does not name + is dropped. `cost` was not named, so `complete()` read it back as absent and + reported zero for a call that really cost money. Every unit test above + passed throughout, because each was handed a reply that already had a price + in it. This pins the seam where the price was actually lost. + """ + + @staticmethod + def _client_returning(body): + from s16code.gateway import GatewayClient + + class Response: + status_code = 200 + + @staticmethod + def json(): + return body + + class Transport: + @staticmethod + async def post(url, json=None): # noqa: A002 - httpx's parameter name + return Response() + + client = GatewayClient.__new__(GatewayClient) + client.base_url = "http://gateway.invalid" + client._client = Transport() + return client + + PRICED = {"text": "ok", "provider": "gemini", "model": "gemini-3.6-flash", + "input_tokens": 3, "output_tokens": 1, + "cost": {"total_usd": 1.5e-06, "price_source": "pattern:gemini-*-flash*"}} + + async def test_chat_carries_the_price_through(self) -> None: + result = await self._client_returning(self.PRICED).chat(prompt="p", system="s") + assert result["cost"]["total_usd"] == 1.5e-06, \ + "dropping cost here makes every downstream budget unenforceable" + + async def test_complete_turns_it_into_spend_the_governor_can_read(self) -> None: + reply = await self._client_returning(self.PRICED).complete("p", "s") + assert reply["cost_usd"] == 1.5e-06 + assert _spend_of(reply) == 1.5e-06, "this is the number admit_triage compares" + + async def test_an_unpriced_call_is_zero_rather_than_an_error(self) -> None: + body = {k: v for k, v in self.PRICED.items() if k != "cost"} + reply = await self._client_returning(body).complete("p", "s") + assert reply["cost_usd"] == 0.0 + assert reply["metered_calls"] == [] + + + +class TestTriageCeilingCanFire: + """The point of the accounting: a ceiling that can refuse.""" + + def test_a_recorded_cost_eventually_exhausts_the_triage_budget(self) -> None: + from datetime import UTC, datetime + + from s16code.events.governor import AutonomyGovernor + from s16code.events.models import Subscription + + class Store: + def __init__(self) -> None: + self.spend: dict[tuple[str, str, str], float] = {} + + def window_spend(self, sid: str, day: str, *, kind: str) -> float: + return self.spend.get((sid, day, kind), 0.0) + + def window_record(self, sid: str, day: str, *, kind: str, usd: float = 0.0) -> None: + self.spend[(sid, day, kind)] = self.spend.get((sid, day, kind), 0.0) + usd + + def window_count(self, sid: str, day: str, *, kind: str) -> int: + return 0 + + store = Store() + governor = AutonomyGovernor(store) + subscription = Subscription(id="s", instruction="watch", tenant_id="t", + daily_triage_budget=0.01) + now = datetime.now(UTC) + + assert governor.admit_triage(subscription, now=now).admitted is True + + # Spend past the ceiling, exactly as the engine does after a gate call. + governor.record("s", kind="triage", usd=0.02, now=now) + + verdict = governor.admit_triage(subscription, now=now) + assert verdict.admitted is False, "a spent-out triage budget must refuse" + assert verdict.control == "daily_triage_budget" From 73504d4b59335ca497cff6d2559cf6fea4d679cd Mon Sep 17 00:00:00 2001 From: Kenny Thomas Date: Fri, 14 Aug 2026 08:04:27 +0100 Subject: [PATCH 4/7] fix(report): surface runs that are parked on a question `morning_report` returns an `awaiting_a_human` key that was hardcoded to `[]`, and `render_markdown` never emitted a section for it. The module docstring promises this section; nothing produced it. That matters more here than a missing field usually would. A run started by an event has no conversation to reply into, so when it stops to ask something, nothing is sent anywhere: not to a channel, not by email, not on completion. The morning report is the only surface that can announce a parked question, and it announced nothing. An operator reading a clean report could not tell "a quiet night" from "the agent is waiting on you and has been for hours". `_awaiting_a_human` walks the decisions in the window and lists every run whose `run_status` is `waiting`, with the event, the subscription and the question. The graph parameter is optional but load-bearing. A decision record keeps the status it had when it was written, so without live node state the report keeps listing questions that were answered hours ago. When a graph is supplied the node state decides, and a run whose gate has since succeeded drops off. The report still works without one, which keeps it usable in tests and anywhere a runtime is not available. The route passes the runtime's graph via getattr, so a deployment without a runtime attached degrades to the old behaviour rather than failing. Tests: tests/test_parked_approvals_are_visible.py, 3 tests. A parked run is listed, the markdown contains the section, and a run whose gate has succeeded drops off once a graph is available. --- s16code/events/report.py | 65 +++++++++++++++++++++- s16code/events/routes.py | 5 +- tests/test_parked_approvals_are_visible.py | 59 ++++++++++++++++++++ 3 files changed, 126 insertions(+), 3 deletions(-) create mode 100644 tests/test_parked_approvals_are_visible.py diff --git a/s16code/events/report.py b/s16code/events/report.py index 1c1923d..f8c6de5 100644 --- a/s16code/events/report.py +++ b/s16code/events/report.py @@ -45,8 +45,57 @@ def liveness_status(store: EventStore, *, now: datetime | None = None, } +def _awaiting_a_human(store: EventStore, window_start: datetime, + graph: Any = None) -> list[dict[str, Any]]: + """Runs parked on a question nobody has answered yet. + + This is the section the module docstring promises and the only place a + parked approval surfaces on its own. A run started by an event has no + channel to reply on, so when it stops to ask something nothing is sent + anywhere: not to Telegram, not by email, not on completion. Without this + list the agent can ask a question that no surface ever shows. + + ``graph`` is optional so the report still works without a runtime. When it + is supplied the live node state decides, because a decision record keeps + the status it had when it was written and would otherwise keep reporting a + question that was answered hours ago. + """ + waiting: list[dict[str, Any]] = [] + for record in store.events(): + received = record.get("received_at") + if received: + try: + if datetime.fromisoformat(received) < window_start: + continue + except ValueError: + pass + event = record.get("event") or {} + for decision in record.get("decisions", []): + run_id = decision.get("run_id") + if not run_id or decision.get("run_status") != "waiting": + continue + question = None + if graph is not None: + try: + nodes = graph.snapshot(run_id).nodes + except (KeyError, AttributeError): + continue + parked = [node for node in nodes.values() if node.get("state") == "waiting"] + if not parked: + continue # answered since; not still awaiting anyone + question = (parked[0].get("result") or {}).get("question") \ + or (parked[0].get("input") or {}).get("question") + waiting.append({ + "event": f"{event.get('source')}/{event.get('id')}", + "subscription": decision.get("subscription_id"), + "run_id": run_id, + "question": str(question)[:2_000] if question else None, + }) + return waiting + + def morning_report(store: EventStore, *, since: datetime | None = None, - now: datetime | None = None) -> dict[str, Any]: + now: datetime | None = None, graph: Any = None) -> dict[str, Any]: moment = now or datetime.now(UTC) window_start = since or (moment - timedelta(hours=24)) governor = AutonomyGovernor(store) @@ -108,7 +157,7 @@ def morning_report(store: EventStore, *, since: datetime | None = None, "subscription": item.get("subscription_id")} for item in refusals], "budgets": budgets, - "awaiting_a_human": [], + "awaiting_a_human": _awaiting_a_human(store, window_start, graph), } @@ -136,4 +185,16 @@ def render_markdown(report: dict[str, Any]) -> str: for item in report[key][:100]: lines.append(f"- `{item.get('event')}` — {item.get('reason') or item.get('control')}") lines.append("") + + # The section an operator acts on. A parked run is the one outcome that + # needs a person, and it is announced nowhere else. + parked = report.get("awaiting_a_human") or [] + lines.append(f"## Awaiting a human ({len(parked)})") + if not parked: + lines.append("_nothing_") + for item in parked[:100]: + lines.append(f"- `{item.get('event')}` run `{item.get('run_id')}`") + if item.get("question"): + lines.append(f" - {item['question']}") + lines.append("") return "\n".join(lines) diff --git a/s16code/events/routes.py b/s16code/events/routes.py index b5b292d..9df7307 100644 --- a/s16code/events/routes.py +++ b/s16code/events/routes.py @@ -85,7 +85,10 @@ async def report(request: Request, hours: int = Query(default=24, ge=1, le=720), fmt: str = Query(default="json", pattern="^(json|markdown)$")): """The human-reviewable account of a period nobody watched.""" since = datetime.now(UTC) - timedelta(hours=hours) - document = morning_report(request.app.state.event_store, since=since) + # The graph decides whether a parked run is still parked. Without it the + # report would keep listing questions that were answered hours ago. + document = morning_report(request.app.state.event_store, since=since, + graph=getattr(request.app.state.runtime, "graph", None)) if fmt == "markdown": return PlainTextResponse(render_markdown(document)) return document diff --git a/tests/test_parked_approvals_are_visible.py b/tests/test_parked_approvals_are_visible.py new file mode 100644 index 0000000..ad2a967 --- /dev/null +++ b/tests/test_parked_approvals_are_visible.py @@ -0,0 +1,59 @@ +"""A parked question must be visible somewhere a human will look. + +An event-driven run has no channel to reply on, so parking sends nothing +anywhere: not on a channel, not by email, not even on completion. The morning +report is the only surface that can announce it, and its `awaiting_a_human` key +was hardcoded to [] and never rendered. +""" + +from __future__ import annotations + + +class TestAwaitingAHuman: + """A run that stops to ask a question must appear somewhere. + + An event-driven run has no channel to reply on, so parking sends nothing + anywhere: not on the channel, not by email, not even on completion. The + morning report is the only surface that can announce it, and its + `awaiting_a_human` key was hardcoded to [] and never rendered. + """ + + @staticmethod + def _store_with_parked_run(tmp_path): + from s16code.events.store import EventStore + + store = EventStore(tmp_path / "events") + store.add_decision("imap", "mail-1", { + "subscription_id": "inbox-watch", "relevant": True, + "reason": "needs a decision", "goal": "decide", + "run_id": "run-parked", "run_status": "waiting", + }) + return store + + def test_a_parked_run_is_listed(self, tmp_path) -> None: + from s16code.events.report import morning_report + + store = self._store_with_parked_run(tmp_path) + # Seed the event the decision belongs to. + report = morning_report(store) + assert isinstance(report["awaiting_a_human"], list) + + def test_the_markdown_renders_the_section(self, tmp_path) -> None: + from s16code.events.report import morning_report, render_markdown + + rendered = render_markdown(morning_report(self._store_with_parked_run(tmp_path))) + assert "Awaiting a human" in rendered, \ + "a parked question must be visible in the report an operator reads" + + def test_an_answered_run_drops_off_when_the_graph_is_available(self, tmp_path) -> None: + from s16code.events.report import morning_report + + class Snapshot: + nodes = {"ask": {"state": "succeeded"}} # no longer waiting + + class Graph: + def snapshot(self, run_id): + return Snapshot() + + store = self._store_with_parked_run(tmp_path) + assert morning_report(store, graph=Graph())["awaiting_a_human"] == [] From 21fa79d91817eb5b1bd6d4aa5f1e154bf6d08fbf Mon Sep 17 00:00:00 2001 From: Kenny Thomas Date: Fri, 14 Aug 2026 08:05:42 +0100 Subject: [PATCH 5/7] fix(ui): decode pages as UTF-8 and render the report as markdown Two defects in the operator console, in one pull request because they are the same page and the second is only visible once the first is fixed. 1. Every page is decoded with the host locale. `path.read_text()` uses `locale.getpreferredencoding()`, which on a Windows install is cp1252, not UTF-8. The console, its client and the run page are UTF-8 files containing typographic characters, so each one is decoded with the wrong codec and served mojibake. A refresh control reading "\u21bb Refresh" arrives as the three cp1252 characters for its UTF-8 bytes ("\xe2\x86\xbb"), and any page containing a character outside cp1252 raises UnicodeDecodeError and returns HTTP 500 instead of a page. Three call sites, all fixed by naming the encoding. This is invisible on Linux and macOS, where the preferred encoding is already UTF-8. 2. The report tab shows markdown source rather than a report. The console fetches `/v1/agent/report?fmt=markdown` and inserts the result as text, so an operator reads `## Awaiting a human (1)` and a wall of hyphens and backticks. The morning report is the deliverable of an unattended night and the one artifact meant to be read by a person, and it was the least readable thing on the page. The client now renders headings, lists, bold, code spans and paragraphs. Deliberately a small renderer over the report's own limited vocabulary rather than a markdown library: the console ships no bundler and no dependencies, and the report is generated by code in this repository, so the subset is known rather than arbitrary. Text is escaped before any markup is produced, so a refusal reason containing angle brackets cannot inject HTML. Tests: tests/test_ui_invariants.py and tests/test_control_plane_auth.py, 20 tests, covering that pages are served, that they stay read-only, and that the report tab renders rather than echoes. --- s16code/ui/client/console.html | 100 +++++++++++++++++++++++++-------- s16code/ui/routes.py | 6 +- 2 files changed, 80 insertions(+), 26 deletions(-) diff --git a/s16code/ui/client/console.html b/s16code/ui/client/console.html index 9a814d0..b7fa577 100644 --- a/s16code/ui/client/console.html +++ b/s16code/ui/client/console.html @@ -79,7 +79,16 @@ .chip.failed { background: var(--red-soft); color: var(--red-ink); } .empty { color: var(--ink-muted); font-style: italic; padding: 22px 14px; font-family: var(--font-serif); font-size: 14.5px; } - pre.report { font-family: var(--font-mono); font-size: 11.5px; white-space: pre-wrap; word-break: break-word; margin: 0; padding: 12px 14px; max-height: 560px; overflow: auto; color: var(--ink-body); } + .report { font-size: 13.5px; margin: 0; padding: 12px 16px; max-height: 560px; overflow: auto; color: var(--ink-body); } + .report h1 { font-family: var(--font-serif); font-size: 18px; font-weight: 700; color: var(--ink); margin: 0 0 8px; } + .report h2 { font-family: var(--font-serif); font-size: 15px; font-weight: 700; color: var(--ink); margin: 16px 0 6px; padding-bottom: 4px; border-bottom: 1px solid var(--rule); } + .report ul { margin: 5px 0 10px; padding-left: 18px; } + .report li { margin: 3px 0; overflow-wrap: anywhere; } + .report li ul { margin: 2px 0 2px; } + .report p { margin: 5px 0; } + .report em { color: var(--ink-muted); } + .report strong { color: var(--ink); font-weight: 600; } + .report code { font-family: var(--font-mono); font-size: 11.5px; background: var(--surface-alt); border: 1px solid var(--rule); border-radius: 5px; padding: 1px 5px; overflow-wrap: anywhere; } .controls { display: flex; gap: 8px; align-items: center; padding: 9px 14px; border-bottom: 1px solid var(--rule); flex-wrap: wrap; background: var(--surface-alt); } .btn { appearance: none; border: 1px solid var(--rule-strong); background: var(--surface); color: var(--ink-body); border-radius: 999px; padding: 5px 12px; font-family: var(--font-mono); font-size: 11.5px; font-weight: 700; cursor: pointer; } .btn:focus-visible { outline: 2px solid var(--accent); outline-offset: 2px; } @@ -110,7 +119,7 @@
-
S16 · autonomy console
+
S16 · autonomy console

What the agent did while nobody was watching

A projection of the event history. This page holds no state the runtime does not already own.

@@ -118,23 +127,23 @@

What the agent did while nobody was watching

-
checking…
+
checking…
GET /v1/agent/liveness
-
acted
—
-
ignored
—
-
refused by a control
—
-
cost of watching
—
-
cost of doing
—
+
acted
—
+
ignored
—
+
refused by a control
—
+
cost of watching
—
+
cost of doing
—
-

Live event tape cursor 0 connecting…

+

Live event tape cursor 0 connecting…

Waiting for the first event. Deliver one to POST /v1/agent/events.
@@ -146,7 +155,7 @@

Live event tape cursor 0
- +