diff --git a/.gitignore b/.gitignore
index e665851..c7389db 100644
--- a/.gitignore
+++ b/.gitignore
@@ -75,6 +75,11 @@ yarn-error.log*
.idea/
*.swp
+# --- Agent tooling scratch ---
+# Per-session metrics written by claude-flow hooks. Machine-local, no bearing
+# on the build; six of these were tracked by accident before this rule.
+.claude-flow/
+
# --- Vercel ---
# Added by `vercel link`, which also appends a bare `.env*`. That pattern is
# kept narrower here: it sits after the `!.env.example` negation above and would
diff --git a/README.md b/README.md
index a95846e..b19d58d 100644
--- a/README.md
+++ b/README.md
@@ -24,7 +24,11 @@ facts memorized into model weights.
- **Evaluated by an independent judge.** A held-out set of 26 domain
scenarios (`petri/seeds/humanitarian_test_scenarios.json`) is graded by a
model from a different family than the one being tested, against explicit
- expected facts — not a self-graded, keyword-ratio heuristic. See
+ expected facts — not a self-graded, keyword-ratio heuristic. Current
+ published baseline: **1 of 26 scenarios passing (4%)**, published as-is —
+ the suite is a regression instrument, not a trophy (see
+ [`docs/STRATEGY.md`](docs/STRATEGY.md#what-the-evals-are-for) and the full
+ reports in [`evals/reports/`](evals/reports/)). See
[`evals/README.md`](evals/README.md).

@@ -33,13 +37,13 @@ facts memorized into model weights.
```mermaid
flowchart LR
- UI["Next.js UI\n(chat, playbooks, guides)"] --> API["/api/chat\nAI SDK v7"]
- API --> Safety["Safety layer\nPII interception"]
- Safety --> LLM["Local Ollama\nqwen2.5:14b\n(swappable via env)"]
- LLM --> T1["search_standards\nSupabase pgvector\nhybrid search"]
- LLM --> T2["crisis_updates\nIFRC GO / ReliefWeb"]
- LLM --> T3["humanitarian_data\nHDX HAPI"]
- LLM --> I18n["i18n: EN / FR / AR / ES\n(RTL for Arabic)"]
+ UI["Next.js UI (chat, playbooks, guides) i18n: EN / FR / AR / ES"] --> API["/api/chat AI SDK v7"]
+ API --> Safety["Safety layer PII interception"]
+ Safety --> LLM["Local Ollama qwen2.5:14b (swappable via env)"]
+ LLM --> T1["search_standards Supabase pgvector hybrid search"]
+ LLM --> T2["crisis_updates IFRC GO / ReliefWeb"]
+ LLM --> T3["humanitarian_data HDX HAPI"]
+ LLM --> T4["hazards_context USGS / GDACS World Bank / OCHA HPC"]
```
- **UI**: Next.js app (`app/`) — chat, six role playbooks, three guides, and
@@ -65,6 +69,11 @@ flowchart LR
- `humanitarian_data` — structured country indicators from **HDX HAPI**
(population, food security, funding, humanitarian needs). No key
required.
+ - `hazards_context` — live hazard and context signals aggregated from four
+ keyless sources: **USGS** (earthquakes), **GDACS** (multi-hazard alerts),
+ **World Bank** (country indicators) and **OCHA HPC** (response plans and
+ funding). Each source degrades its own section rather than failing the
+ whole answer.
- **i18n**: UI chrome in English, French, Arabic, Spanish, with RTL layout
for Arabic. The model answers in whatever language the user writes in,
independent of the UI locale.
diff --git a/app/.env.example b/app/.env.example
index d3174ad..b9d0bef 100644
--- a/app/.env.example
+++ b/app/.env.example
@@ -26,6 +26,27 @@ LLM_BASE_URL=http://localhost:11434/v1
LLM_MODEL=qwen2.5:14b
LLM_API_KEY=ollama
+# Optional. The model /api/deliverables uses; defaults to LLM_MODEL.
+#
+# Worth setting on a hosted deployment, because Groq's token buckets are per
+# model — both the per-minute one and the per-day one. Verified against the
+# deployed key: qwen/qwen3.8-27b and openai/gpt-oss-120b each reported their own
+# independent 8,000 tokens/minute and decremented separately.
+#
+# A situation brief costs roughly 25,000 tokens against a 200,000/day ceiling,
+# so sharing one model with chat means a busy chat afternoon quietly consumes
+# the ability to produce a document. Pointing deliverables at a second model
+# gives the two features independent daily budgets on the same key.
+#
+# LLM_DELIVERABLES_MODEL=openai/gpt-oss-120b # hosted demo's value
+LLM_DELIVERABLES_MODEL=
+
+# Optional. Only if deliverables should use a different provider entirely
+# rather than a second model on the same one. Each falls back to its LLM_*
+# equivalent.
+LLM_DELIVERABLES_BASE_URL=
+LLM_DELIVERABLES_API_KEY=
+
# IMPORTANT for local Ollama: the 4096-token default context is too small for
# HAI, and overflowing it silently drops the system prompt — after which the
# model loses its grounding and language rules mid-answer. Either build the
@@ -82,8 +103,30 @@ HDX_APP_IDENTIFIER=
# RATE_LIMIT_RPM paces one client: a per-IP sliding window, default 20/min. It
# lives in process memory, so on Vercel each serverless instance keeps its own
# copy and it resets on every deploy — a politeness control, not a spend control.
+#
+# It governs /api/chat only. /api/deliverables keeps its own, much tighter
+# counter (3 runs per 10 minutes, not configurable), because one run is about
+# nineteen model calls rather than one.
RATE_LIMIT_RPM=
+# LLM_TOKENS_PER_MINUTE no longer needs setting on an ordinary hosted
+# deployment, and its meaning has narrowed.
+#
+# /api/deliverables now reads x-ratelimit-* off every response (see
+# src/lib/llm/rate-limit.ts) and paces against the endpoint's own numbers rather
+# than a configured guess, so Groq's free tier and a paid tier are both paced
+# correctly with no configuration. Pacing switches itself on when an endpoint
+# reports a ceiling and stays off when none is reported — a local Ollama is
+# never paced, and no URL check decides that.
+#
+# What this variable is for now is the endpoint that enforces a ceiling but does
+# not report it in headers: set it to that ceiling in tokens per minute. A
+# reported limit always wins over it. Leave it unset otherwise.
+#
+# Only the deliverables engine reads it; chat is one call at a time and never
+# approaches the limit.
+LLM_TOKENS_PER_MINUTE=
+
# MAX_DAILY_REQUESTS is the spend control: one shared counter in Postgres,
# default 500/day across the whole deployment, enforced atomically so concurrent
# instances cannot overshoot it. Requires the daily_request_cap migration.
diff --git a/app/README.md b/app/README.md
index 81064f8..c3e7e98 100644
--- a/app/README.md
+++ b/app/README.md
@@ -9,8 +9,8 @@ This file covers just this package.
| Path | What's in it |
|---|---|
-| `src/app/` | Routes — chat (`/`), playbooks, guides, `/api/chat` |
-| `src/lib/tools/` | The three tools the model calls: `search-standards.ts`, `crisis-updates.ts` (IFRC GO / ReliefWeb), `humanitarian-data.ts` (HDX HAPI) |
+| `src/app/` | Routes — chat (`/`), playbooks, guides, `/about`, `/deliverables`, `/api/chat`, `/api/deliverables` |
+| `src/lib/tools/` | The four tools the model calls: `search-standards.ts`, `crisis-updates.ts` (IFRC GO / ReliefWeb), `humanitarian-data.ts` (HDX HAPI), `hazards-context.ts` (USGS / GDACS / World Bank / OCHA HPC) |
| `src/lib/retrieval/` | `search.ts` — embeds the query via Ollama and calls the `search_standards_hybrid` Supabase RPC |
| `src/lib/safety/` | `pii.ts` (deterministic regex/heuristic screening), `llm-screen.ts` (optional second-pass), `intercept.ts` (turns findings into the banner copy) |
| `src/lib/prompts/` | `system.ts` (base system prompt), `coach.ts` (coach-mode addition) |
diff --git a/app/package.json b/app/package.json
index 839e497..1fa5354 100644
--- a/app/package.json
+++ b/app/package.json
@@ -1,7 +1,14 @@
{
- "name": "app",
+ "name": "hai-app",
"version": "0.1.0",
"private": true,
+ "description": "The HAI Next.js application: chat, role playbooks, guides and the deliverables engine. Built from the repository root — see the root package.json for why.",
+ "license": "MIT",
+ "repository": {
+ "type": "git",
+ "url": "https://github.com/samfrons/HAI.git",
+ "directory": "app"
+ },
"scripts": {
"dev": "next dev",
"build": "next build",
@@ -15,6 +22,7 @@
"@ai-sdk/react": "^4.0.82",
"@supabase/supabase-js": "^2.58.0",
"ai": "^7.0.79",
+ "fast-xml-parser": "^5.11.1",
"gray-matter": "^4.0.3",
"next": "16.3.2",
"react": "19.2.8",
diff --git a/app/pnpm-lock.yaml b/app/pnpm-lock.yaml
index 47d835c..770f208 100644
--- a/app/pnpm-lock.yaml
+++ b/app/pnpm-lock.yaml
@@ -20,6 +20,9 @@ importers:
ai:
specifier: ^7.0.79
version: 7.0.79(zod@4.4.3)
+ fast-xml-parser:
+ specifier: ^5.11.1
+ version: 5.11.1
gray-matter:
specifier: ^4.0.3
version: 4.0.3
@@ -476,6 +479,9 @@ packages:
cpu: [x64]
os: [win32]
+ '@nodable/entities@3.0.0':
+ resolution: {integrity: sha512-8L9xFeTYKhm49xfIypoe2W5wV1m/3Z58kT+7kR9A8OyFxcPduI4VmxaUMQyKYrRjUoLLSXv6EKKID5Tvj9cUVw==}
+
'@nodelib/fs.scandir@2.1.5':
resolution: {integrity: sha512-vq24Bq3ym5HEQm2NKCr3yXDwjc7vTsEThRDnkp2DK9p1uqLR+DHurm/NOTo0KG7HYHU7eppKZj3MyqYuMBf62g==}
engines: {node: '>= 8'}
@@ -998,6 +1004,9 @@ packages:
resolution: {integrity: sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg==}
engines: {node: '>=8'}
+ anynum@1.0.1:
+ resolution: {integrity: sha512-N6//FLET/tXYNM/F6ABca1oH6fWB+KlTt909Le28WMDBk8oaT4vY17DCrwg2MvmuqUKt3Ni4N5dGJ/EoBgcO6A==}
+
argparse@1.0.10:
resolution: {integrity: sha512-o5Roy6tNG4SL/FOkCAN6RzjiakZS25RLYFrcMttJqbdd8BWrnA+fGz57iN5Pb06pvBGvl5gQ0B48dJlslXvoTg==}
@@ -1448,6 +1457,13 @@ packages:
fast-levenshtein@2.0.6:
resolution: {integrity: sha512-DCXu6Ifhqcks7TZKY3Hxp3y6qphY5SJZmrWMDrKcERSOXWQdMhU9Ig/PYrzyw/ul9jOIyh0N4M0tbC5hodg8dw==}
+ fast-xml-builder@1.3.1:
+ resolution: {integrity: sha512-pIM/1n3ntFXKYrUZwW7QCK0gAW7XY+wzj1YMIV3tLDvPj/V+zTGJK5e3/4WJfwj0qWw2ElNXiTixda/R+3YSug==}
+
+ fast-xml-parser@5.11.1:
+ resolution: {integrity: sha512-TBw6K/fxoQGGjCmZDw9w/ZwP3uDcnTM4YH/g+PFRWr8sbe5idXtxNN6vITh4+1ruCZaho6uBFurElsA7F0zzgw==}
+ hasBin: true
+
fastq@1.20.1:
resolution: {integrity: sha512-GGToxJ/w1x32s/D2EKND7kTil4n8OVk/9mycTc4VDza13lOvpUZTGX3mFSCtV9ksdGBVzvsyAVLM6mHFThxXxw==}
@@ -1736,6 +1752,9 @@ packages:
resolution: {integrity: sha512-p3EcsicXjit7SaskXHs1hA91QxgTw46Fv6EFKKGS5DRFLD8yKnohjF3hxoju94b/OcMZoQukzpPpBE9uLVKzgQ==}
engines: {node: '>= 0.4'}
+ is-unsafe@2.0.2:
+ resolution: {integrity: sha512-HgbIHPBH0KHHCcjLfGsCvhtPTVxjaAZlXjwdz7/GQC40SjSe4sfQsar8J5VFo8JOSbarkpV0OLG95bbaNd9aAQ==}
+
is-weakmap@2.0.2:
resolution: {integrity: sha512-K5pXYOm9wqY1RgjpL3YTkF39tni1XajUIkawTLUo9EZEVUFga5gSQJF8nNS7ZwJQ02y+1YCNYcMh+HIf1ZqE+w==}
engines: {node: '>= 0.4'}
@@ -2243,6 +2262,10 @@ packages:
resolution: {integrity: sha512-ak9Qy5Q7jYb2Wwcey5Fpvg2KoAc/ZIhLSLOSBmRmygPsGwkVVt0fZa0qrtMz+m6tJTAHfZQ8FnmB4MG4LWy7/w==}
engines: {node: '>=8'}
+ path-expression-matcher@1.6.2:
+ resolution: {integrity: sha512-enSlaiat05iasnzmgNxRj8reFdj3puY2QpNgP1aPIaVfT6nn9ICuPoFlKHk8EN22HcwewshO+mN2DGbkCEOtqQ==}
+ engines: {node: '>=14.0.0'}
+
path-key@3.1.1:
resolution: {integrity: sha512-ojmeN0qd+y0jszEtoY48r0Peq5dwMEkIlCOu6Q5f41lfkswXuKtYrhgoTpLnyIcHm24Uhqx+5Tqm2InSwLhE6Q==}
engines: {node: '>=8'}
@@ -2496,6 +2519,9 @@ packages:
resolution: {integrity: sha512-6fPc+R4ihwqP6N/aIv2f1gMH8lOVtWQHoqC4yK6oSDVVocumAsfCqjkXnqiYMhmMwS/mEHLp7Vehlt3ql6lEig==}
engines: {node: '>=8'}
+ strnum@2.4.2:
+ resolution: {integrity: sha512-rDG3Ah4TV0k1hWvLSzkZtMmLN9+eS+h3knq4MP6A42Y3Yh5qGNnOUs1jJkoSr8FG5dsL28c7KgkIBzSEykqtuw==}
+
style-to-js@1.1.21:
resolution: {integrity: sha512-RjQetxJrrUJLQPHbLku6U/ocGtzyjbJMP9lCNK7Ag0CNh690nSH8woqWH9u16nMjYBAok+i7JO1NP2pOy8IsPQ==}
@@ -2774,6 +2800,10 @@ packages:
resolution: {integrity: sha512-BN22B5eaMMI9UMtjrGd5g5eCYPpCPDUy0FJXbYsaT5zYxjFOckS53SQDE3pWkVoWpHXVb3BrYcEN4Twa55B5cA==}
engines: {node: '>=0.10.0'}
+ xml-naming@0.3.0:
+ resolution: {integrity: sha512-ghig2TBE/H11aOVgmahA3MhimvkBr6JIYknH/Dhdk10nXwdbIqBJsbfMxpvFPG8bAw77gN29aQWvKpmVoPlvPQ==}
+ engines: {node: '>=16.0.0'}
+
yallist@3.1.1:
resolution: {integrity: sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g==}
@@ -3197,6 +3227,8 @@ snapshots:
'@next/swc-win32-x64-msvc@16.3.2':
optional: true
+ '@nodable/entities@3.0.0': {}
+
'@nodelib/fs.scandir@2.1.5':
dependencies:
'@nodelib/fs.stat': 2.0.5
@@ -3653,6 +3685,8 @@ snapshots:
dependencies:
color-convert: 2.0.1
+ anynum@1.0.1: {}
+
argparse@1.0.10:
dependencies:
sprintf-js: 1.0.3
@@ -4260,6 +4294,20 @@ snapshots:
fast-levenshtein@2.0.6: {}
+ fast-xml-builder@1.3.1:
+ dependencies:
+ path-expression-matcher: 1.6.2
+ xml-naming: 0.3.0
+
+ fast-xml-parser@5.11.1:
+ dependencies:
+ '@nodable/entities': 3.0.0
+ fast-xml-builder: 1.3.1
+ is-unsafe: 2.0.2
+ path-expression-matcher: 1.6.2
+ strnum: 2.4.2
+ xml-naming: 0.3.0
+
fastq@1.20.1:
dependencies:
reusify: 1.1.0
@@ -4566,6 +4614,8 @@ snapshots:
dependencies:
which-typed-array: 1.1.22
+ is-unsafe@2.0.2: {}
+
is-weakmap@2.0.2: {}
is-weakref@1.1.1:
@@ -5252,6 +5302,8 @@ snapshots:
path-exists@4.0.0: {}
+ path-expression-matcher@1.6.2: {}
+
path-key@3.1.1: {}
path-parse@1.0.7: {}
@@ -5616,6 +5668,10 @@ snapshots:
strip-json-comments@3.1.1: {}
+ strnum@2.4.2:
+ dependencies:
+ anynum: 1.0.1
+
style-to-js@1.1.21:
dependencies:
style-to-object: 1.0.14
@@ -5917,6 +5973,8 @@ snapshots:
word-wrap@1.2.5: {}
+ xml-naming@0.3.0: {}
+
yallist@3.1.1: {}
yocto-queue@0.1.0: {}
diff --git a/app/src/app/api/chat/route.ts b/app/src/app/api/chat/route.ts
index 2865d73..441fabb 100644
--- a/app/src/app/api/chat/route.ts
+++ b/app/src/app/api/chat/route.ts
@@ -4,12 +4,15 @@ import {
createUIMessageStreamResponse,
stepCountIs,
streamText,
+ type InferUIMessageChunk,
type InferUITools,
type UIMessage,
+ type UIMessageStreamWriter,
} from 'ai';
import { claimDailyRequest, dailyCapMessage } from '@/lib/limits/daily-cap';
-import { getChatModel, getProviderOptions } from '@/lib/llm/provider';
+import { getChatModel, getModelBudget, getProviderOptions } from '@/lib/llm/provider';
+import { humaniseUpstreamError } from '@/lib/llm/rate-limit';
import { COACH_SYSTEM_PROMPT } from '@/lib/prompts/coach';
import { SYSTEM_PROMPT } from '@/lib/prompts/system';
import { warmEmbeddingsEndpoint } from '@/lib/retrieval/embeddings';
@@ -43,6 +46,21 @@ export const maxDuration = 60;
/** Custom data parts HAI streams alongside text. */
export type HaiDataParts = {
'safety-notice': SafetyNoticeData;
+ /**
+ * The turn is waiting behind the endpoint's own token budget rather than
+ * behind a slow model. Streamed transiently — it is a fact about right now,
+ * not part of the answer, and it must not be replayed in the history of a
+ * conversation whose answer arrived fine thirty seconds later.
+ */
+ queued: {
+ retryAfterMs: number;
+ /**
+ * Which ceiling. A per-minute queue clears itself while the reader waits; a
+ * per-day one does not, and a countdown implying otherwise would be a lie
+ * with a number attached.
+ */
+ scope: 'tokens-per-minute' | 'tokens-per-day';
+ };
};
/** Shared with the client so message parts are typed against the real tools. */
@@ -107,6 +125,63 @@ function isRateLimited(key: string): boolean {
return false;
}
+/* ------------------------------------------------------------------ *
+ * Dead air, explained
+ * ------------------------------------------------------------------ */
+
+/** How often the budget is re-read while the first chunk is outstanding. */
+const QUEUE_POLL_MS = 1_000;
+/**
+ * How long that polling runs. Bounded rather than open-ended so this can never
+ * hold the response open on its own: the watch stops well inside `maxDuration`,
+ * and the answer streams on regardless of whether anything was ever announced.
+ */
+const QUEUE_WATCH_MS = 30_000;
+
+/**
+ * Say when a turn is queued behind the free tier rather than merely slow.
+ *
+ * The provider's fetch already folds every response — including a 429 and its
+ * `retry-after` — into the model's `TokenBudget` (see `lib/llm/rate-limit.ts`),
+ * so the fact exists server-side the moment the endpoint refuses. What it did
+ * not have was a way out to the browser: the SDK retries a 429 internally and
+ * silently, and from the client that is indistinguishable from a model taking
+ * twenty seconds to think. This writes one transient data part when the budget
+ * says the wait is a queue, and nothing at all otherwise — a local endpoint
+ * never reports a limit, so it never becomes measured, so this never fires.
+ */
+async function watchTokenQueue(
+ writer: UIMessageStreamWriter,
+ firstChunk: Promise,
+): Promise {
+ const budget = getModelBudget();
+ const deadline = Date.now() + QUEUE_WATCH_MS;
+
+ while (Date.now() < deadline) {
+ const arrived = await Promise.race([
+ firstChunk.then(() => true),
+ new Promise((resolve) => setTimeout(() => resolve(false), QUEUE_POLL_MS)),
+ ]);
+ if (arrived) return;
+
+ // `waitFor(0)` is exactly the unexpired part of a refusal's `retry-after`:
+ // spending nothing needs no refill, so anything left is the block itself.
+ const blockedMs = budget.waitFor(0);
+ const daily = budget.snapshot().dailyExhausted;
+ if (daily || blockedMs > 0) {
+ writer.write({
+ type: 'data-queued',
+ data: {
+ retryAfterMs: daily?.untilMs ?? blockedMs,
+ scope: daily ? 'tokens-per-day' : 'tokens-per-minute',
+ },
+ transient: true,
+ });
+ return;
+ }
+ }
+}
+
/* ------------------------------------------------------------------ *
* Data-responsibility screening
* ------------------------------------------------------------------ */
@@ -261,8 +336,55 @@ export async function POST(request: Request) {
stopWhen: stepCountIs(4),
});
- return result.toUIMessageStreamResponse({
- onError: (error) =>
- error instanceof Error ? error.message : 'The assistant hit an unexpected error.',
+ /*
+ * Upstream messages are written for whoever pays the endpoint's bill, not for
+ * whoever is asking about water quantity in Sudan. Left alone, a rate-limited
+ * turn put this in front of the user verbatim, billing link included:
+ *
+ * Rate limit reached for model `qwen/qwen3.8-27b` in organization
+ * `org_01kk…` service tier `on_demand` on tokens per day (TPD): Limit
+ * 200000, Used 199889 … Need more tokens? Upgrade to Dev Tier today at
+ * https://console.groq.com/settings/billing
+ *
+ * The engine already sanitised this on the deliverables side; chat did not,
+ * so both now go through the same function.
+ */
+ const onError = (error: unknown) =>
+ error instanceof Error
+ ? humaniseUpstreamError(error.message)
+ : 'The assistant hit an unexpected error.';
+
+ const stream = createUIMessageStream({
+ onError,
+ execute: async ({ writer }) => {
+ // Resolved by the first chunk that says the model actually answered.
+ // `start` and `start-step` are emitted by the SDK before the request has
+ // been answered at all, so they are precisely the two that prove nothing.
+ let responded: () => void = () => {};
+ const firstChunk = new Promise((resolve) => {
+ responded = resolve;
+ });
+
+ writer.merge(
+ result.toUIMessageStream({ onError }).pipeThrough(
+ new TransformStream, InferUIMessageChunk>(
+ {
+ transform(chunk, controller) {
+ if (chunk.type !== 'start' && chunk.type !== 'start-step') responded();
+ controller.enqueue(chunk);
+ },
+ // A turn that ends without ever producing one — an error, an
+ // abort — must still release the watch rather than hold the
+ // response open for its full window.
+ flush: () => responded(),
+ },
+ ),
+ ),
+ );
+
+ await watchTokenQueue(writer, firstChunk);
+ },
});
+
+ return createUIMessageStreamResponse({ stream });
}
diff --git a/app/src/app/api/deliverables/route.ts b/app/src/app/api/deliverables/route.ts
new file mode 100644
index 0000000..8874d63
--- /dev/null
+++ b/app/src/app/api/deliverables/route.ts
@@ -0,0 +1,211 @@
+import { createUIMessageStream, createUIMessageStreamResponse, type UIMessage } from 'ai';
+
+import { runWorkflow } from '@/lib/agent/engine';
+import type { TraceEvent } from '@/lib/agent/types';
+import { WORKFLOWS, isWorkflowId } from '@/lib/agent/workflows';
+import { claimDailyRequest, dailyCapMessage } from '@/lib/limits/daily-cap';
+import { clientKey, createBurstLimiter } from '@/lib/limits/burst';
+import { buildInterceptionMessage } from '@/lib/safety/intercept';
+import { llmScreen } from '@/lib/safety/llm-screen';
+import { screenForPii } from '@/lib/safety/pii';
+import { warmEmbeddingsEndpoint } from '@/lib/retrieval/embeddings';
+import { warmSupabaseConnection } from '@/lib/retrieval/search';
+
+export const dynamic = 'force-dynamic';
+
+/*
+ * The wall clock a complete brief needs, with margin — and the binding
+ * constraint on this route rather than a comfortable one, which is the opposite
+ * of the chat route's situation and worth being explicit about.
+ *
+ * A situation brief is nineteen model calls. Against hosted inference those are
+ * fast; what takes the time is the pacing. The free tier meters 8,000 tokens a
+ * minute and a full run spends roughly 25,000, so `lib/agent/pacer.ts`
+ * deliberately waits out the difference. A measured end-to-end Sudan brief —
+ * six sections, every one populated, no request refused — took 142 seconds, of
+ * which about 105 were spent idle inside those waits on purpose.
+ *
+ * So this is 300, Vercel's ceiling with Fluid compute (the default for new
+ * projects, Hobby included), and the measured run leaves roughly a two-times
+ * margin under it. That margin is real but it is not unlimited: the token
+ * bucket is shared with anything else on the same key, so a run competing with
+ * chat traffic paces harder and takes longer.
+ *
+ * Two things follow, neither hidden from the user. The stream is written
+ * incrementally, so a run cut short still leaves the reader the sections that
+ * finished rather than nothing, and the client says the document is partial.
+ * And the engine stops itself before the platform can — see `deadline` below —
+ * so the last thing a truncated run does is explain itself rather than having
+ * its connection dropped mid-sentence.
+ *
+ * Local `next dev` and `next start` ignore this value entirely.
+ */
+export const maxDuration = 300;
+
+/**
+ * Margin left for the run to end tidily inside the function's lifetime.
+ *
+ * A serverless function that hits its ceiling is killed, and a killed stream is
+ * the one failure the reader cannot interpret: the document simply stops, with
+ * no caveat, no timestamp, and no indication whether the missing sections were
+ * empty or never attempted. Ending the run ourselves a few seconds early costs
+ * one section and buys an honest ending.
+ */
+const DEADLINE_MARGIN_MS = 15_000;
+
+// cost: $0.00 per run with the default local Ollama endpoint. A hosted
+// LLM_BASE_URL may bill per token, and one run is ~16 calls, not one.
+
+/**
+ * One run's worth of trace, streamed as data parts.
+ *
+ * Every event goes over the wire as `data-trace`. The client keeps the whole
+ * sequence and derives both views from it — the assembling document on the left
+ * and the trace panel on the right — so there is exactly one source of truth for
+ * what happened, and the document can never show a section the trace does not
+ * account for.
+ */
+export type DeliverableDataParts = {
+ trace: TraceEvent;
+};
+
+export type DeliverableUIMessage = UIMessage;
+
+/*
+ * Three runs per ten minutes per IP.
+ *
+ * Much tighter than chat's twenty a minute, because the unit is not comparable:
+ * a run is roughly sixteen model calls and several live API round trips, and it
+ * occupies a serverless function for minutes. Someone legitimately using this
+ * generates a brief, reads it, and edits it; nobody needs a fourth inside ten
+ * minutes, and a script that does is the case this exists for.
+ */
+const runLimiter = createBurstLimiter(3, 10 * 60_000);
+
+/** Bounds what can be spent on resolving a subject line. */
+const MAX_SUBJECT_CHARS = 120;
+
+export async function POST(request: Request) {
+ // Same rationale as the chat route: the first tool call is still a model round
+ // trip away, and warming these now means retrieval is hot when a gather step
+ // reaches it rather than paying a cold start inside the run.
+ warmEmbeddingsEndpoint();
+ warmSupabaseConnection();
+
+ if (runLimiter.isLimited(clientKey(request))) {
+ return Response.json(
+ {
+ error:
+ 'Rate limit exceeded — 3 deliverables per 10 minutes. Each run makes many model calls; wait a few minutes and try again.',
+ },
+ { status: 429, headers: { 'Retry-After': '600' } },
+ );
+ }
+
+ /*
+ * Claimed once, like a chat message, even though a run costs far more than
+ * one. Weighting it properly would mean either changing the shared counter's
+ * contract or making sixteen sequential round trips before any work starts,
+ * and a partial claim that then fails would burn budget for nothing. The
+ * burst limiter above is what bounds the real cost here; this keeps the
+ * hosted deployment's single day-counter honest about traffic.
+ */
+ const daily = await claimDailyRequest();
+ if (!daily.allowed) {
+ return Response.json({ error: dailyCapMessage(daily) }, { status: 429 });
+ }
+
+ let workflowId: string;
+ let subject: string;
+ try {
+ ({ workflowId, subject } = (await request.json()) as {
+ workflowId: string;
+ subject: string;
+ });
+ } catch {
+ return Response.json({ error: 'Malformed request body.' }, { status: 400 });
+ }
+
+ if (typeof workflowId !== 'string' || !isWorkflowId(workflowId)) {
+ return Response.json({ error: 'Unknown deliverable template.' }, { status: 400 });
+ }
+ if (typeof subject !== 'string' || !subject.trim()) {
+ return Response.json(
+ { error: 'Name the country or topic this deliverable is about.' },
+ { status: 400 },
+ );
+ }
+ if (subject.length > MAX_SUBJECT_CHARS) {
+ return Response.json(
+ { error: `Keep the subject under ${MAX_SUBJECT_CHARS} characters.` },
+ { status: 400 },
+ );
+ }
+
+ /*
+ * Screened before anything else touches it — no retrieval, no model call, no
+ * logging. The subject line is short, but it is a free-text field on a
+ * humanitarian tool, which makes it exactly the place someone pastes a
+ * beneficiary name or a phone number while meaning to name a caseload.
+ *
+ * The refusal streams as a normal run that produces no document, rather than
+ * as a 4xx. A red error banner frames a correct data-responsibility decision
+ * as a broken app, which is the reading most likely to send someone to a tool
+ * with no screening at all — the same reasoning as the chat route's.
+ */
+ const deterministic = screenForPii(subject);
+ if (deterministic.flagged) {
+ return refusalStream(buildInterceptionMessage(deterministic.findings));
+ }
+ const semantic = await llmScreen(subject);
+ if (semantic) {
+ return refusalStream(buildInterceptionMessage([semantic]));
+ }
+
+ const workflow = WORKFLOWS[workflowId];
+
+ const stream = createUIMessageStream({
+ execute: async ({ writer }) => {
+ writer.write({ type: 'start' });
+ writer.write({ type: 'start-step' });
+
+ for await (const event of runWorkflow({
+ workflow,
+ subject,
+ signal: request.signal,
+ deadline: Date.now() + maxDuration * 1_000 - DEADLINE_MARGIN_MS,
+ })) {
+ writer.write({ type: 'data-trace', data: event });
+ }
+
+ writer.write({ type: 'finish-step' });
+ writer.write({ type: 'finish' });
+ },
+ onError: (error) =>
+ error instanceof Error ? error.message : 'The run hit an unexpected error.',
+ });
+
+ return createUIMessageStreamResponse({ stream });
+}
+
+/**
+ * A refused subject line, delivered as a completed run with one error event.
+ * The client already knows how to render `workflow-error`, so this needs no
+ * special case on the other side.
+ */
+function refusalStream(message: string): Response {
+ const stream = createUIMessageStream({
+ execute: ({ writer }) => {
+ writer.write({ type: 'start' });
+ writer.write({ type: 'start-step' });
+ writer.write({
+ type: 'data-trace',
+ data: { type: 'workflow-error', at: Date.now(), message },
+ });
+ writer.write({ type: 'finish-step' });
+ writer.write({ type: 'finish' });
+ },
+ });
+
+ return createUIMessageStreamResponse({ stream });
+}
diff --git a/app/src/app/deliverables/layout.tsx b/app/src/app/deliverables/layout.tsx
new file mode 100644
index 0000000..6080036
--- /dev/null
+++ b/app/src/app/deliverables/layout.tsx
@@ -0,0 +1,15 @@
+import { Footer } from '@/components/footer';
+import { SiteHeader } from '@/components/site-header';
+
+export default function DeliverablesLayout({ children }: LayoutProps<'/deliverables'>) {
+ return (
+
+
+ {/* Wider than the playbooks and guides pages: this one carries a document
+ and its trace side by side, and squeezing the trace into a narrow rail
+ would make it the footnote it is specifically not meant to be. */}
+ {children}
+
+
+ );
+}
diff --git a/app/src/app/deliverables/page.tsx b/app/src/app/deliverables/page.tsx
new file mode 100644
index 0000000..b37a180
--- /dev/null
+++ b/app/src/app/deliverables/page.tsx
@@ -0,0 +1,13 @@
+import type { Metadata } from 'next';
+
+import { Deliverables } from '@/components/deliverables';
+
+export const metadata: Metadata = {
+ title: 'Deliverables — HAI',
+ description:
+ 'Generate a country situation brief or a donor report section from live humanitarian data and the standards corpus, with every source consulted and every claim checked shown alongside.',
+};
+
+export default function DeliverablesPage() {
+ return ;
+}
diff --git a/app/src/app/globals.css b/app/src/app/globals.css
index a257c04..e2f6a56 100644
--- a/app/src/app/globals.css
+++ b/app/src/app/globals.css
@@ -103,6 +103,20 @@ h1, h2, h3, .hai-heading {
text-transform: uppercase;
}
+/* Arabic is a cursive script: its letters join, so `letter-spacing` inserts
+ its gap *inside* words rather than tracking them out, and the eyebrow reads
+ as broken instead of wide-set. That matters most in the header nav, which is
+ set entirely in eyebrows. Blink quietly suppresses the spacing across
+ cursive joins, so this is invisible in Chrome — WebKit and Gecko do apply
+ it, which is where the disjointed rendering and the extra label width (and
+ with it the pressure on a 390px RTL header) actually show up. Uppercasing is
+ a no-op for Arabic; it is cleared alongside so the rule reads as one
+ statement: no Latin small-caps treatment under `lang="ar"`. */
+:lang(ar) .hai-eyebrow {
+ letter-spacing: normal;
+ text-transform: none;
+}
+
/* Tabular data voice — citations' source · section · page, tool detail
strings, counts. Set in the mono face so numbers align and read as data. */
.hai-data {
diff --git a/app/src/components/chat.tsx b/app/src/components/chat.tsx
index a6ffbd5..672d6f4 100644
--- a/app/src/components/chat.tsx
+++ b/app/src/components/chat.tsx
@@ -17,6 +17,7 @@ import { Markdown } from './markdown';
import { NavLinks } from './nav-links';
import { PendingStatus } from './pending-status';
import { SafetyNotice } from './safety-notice';
+import { ShowWorking } from './show-working';
import { SourcePanel } from './source-panel';
import { collectRetrievalNotice, collectSources, type Source } from './sources';
import { ToolActivity } from './tool-activity';
@@ -24,6 +25,18 @@ import { ToolActivity } from './tool-activity';
/** Seconds must pass before the elapsed counter appears — see PendingStatus. */
const ELAPSED_THRESHOLD_S = 3;
+/**
+ * Seconds of silence after which the pending line stops claiming a phase.
+ *
+ * "Contacting model…" is a fair description of the first few seconds and an
+ * increasingly poor one after eight, when the honest thing left to say is that
+ * the turn is alive and nothing has arrived yet. Set against the free tier's
+ * behaviour rather than picked round: a queued turn on Groq's free budget waits
+ * tens of seconds, and the eight-second mark is where a normal answer has
+ * already started and a queued one plainly has not.
+ */
+const DEAD_AIR_THRESHOLD_S = 8;
+
type MessagePart = HaiUIMessage['parts'][number];
function isToolPart(part: MessagePart): boolean {
@@ -70,8 +83,17 @@ export function Chat({ hosted = false }: { hosted?: boolean }) {
const [coachMode, setCoachMode] = useState(false);
const scrollAnchor = useRef(null);
+ // Set by the route's transient `queued` part: the endpoint has refused on its
+ // token budget and this turn is waiting behind that refusal. Transient means
+ // it never lands in `message.parts`, so it is held here for the life of the
+ // turn and cleared by the next send.
+ const [queued, setQueued] = useState<'tokens-per-minute' | 'tokens-per-day' | null>(null);
+
const { messages, sendMessage, status, stop, error } = useChat({
transport: new DefaultChatTransport({ api: '/api/chat' }),
+ onData: (part) => {
+ if (part.type === 'data-queued') setQueued(part.data.scope);
+ },
});
const busy = status === 'submitted' || status === 'streaming';
@@ -100,6 +122,21 @@ export function Chat({ hosted = false }: { hosted?: boolean }) {
const phase = pendingPhase(status, lastMessage);
const elapsedLabel = busy && elapsedSeconds !== null ? t.pending.elapsed(elapsedSeconds) : null;
+ // What the pending line says, in order of how much it knows. A queue is a
+ // specific fact and outranks everything; past eight seconds of silence the
+ // phase words have outlived their accuracy; otherwise the phase is right.
+ const pendingText = !phase
+ ? null
+ : queued === 'tokens-per-day'
+ ? // Not a queue. Nothing is going to arrive by waiting, and a line saying
+ // "queued" would have the reader wait for it anyway.
+ t.deliverables.budgetDailyExhausted
+ : queued
+ ? t.pending.queued
+ : elapsedSeconds !== null && elapsedSeconds >= DEAD_AIR_THRESHOLD_S
+ ? t.pending.stillWorking
+ : t.pending[phase];
+
useEffect(() => {
scrollAnchor.current?.scrollIntoView({ behavior: 'smooth', block: 'end' });
}, [messages, status]);
@@ -115,6 +152,7 @@ export function Chat({ hosted = false }: { hosted?: boolean }) {
const trimmed = text.trim();
if (!trimmed) return;
setInput('');
+ setQueued(null);
void sendMessage({ text: trimmed }, { body: { mode: coachMode ? 'coach' : 'default' } });
},
[sendMessage, coachMode],
@@ -169,6 +207,7 @@ export function Chat({ hosted = false }: { hosted?: boolean }) {
onSelectSource={setActiveSource}
isLast={isLast}
pendingPhase={isLast ? phase : null}
+ pendingText={isLast ? pendingText : null}
elapsedLabel={elapsedLabel}
/>
);
@@ -176,8 +215,8 @@ export function Chat({ hosted = false }: { hosted?: boolean }) {
{/* The turn hasn't produced an assistant message yet at all — status
'submitted', or 'start' has fired but no parts exist yet. Once the
assistant message appears, MessageBlock above takes over. */}
- {phase && lastMessage?.role !== 'assistant' ? (
-
+ {pendingText && lastMessage?.role !== 'assistant' ? (
+
) : null}
)}
@@ -211,6 +250,7 @@ function MessageBlock({
onSelectSource,
isLast,
pendingPhase,
+ pendingText,
elapsedLabel,
}: {
message: HaiUIMessage;
@@ -218,10 +258,10 @@ function MessageBlock({
isLast: boolean;
/** Meaningful only when `isLast` — see `pendingPhase()` in the parent. */
pendingPhase: 'contacting' | 'writing' | null;
+ /** The line to show for that phase, already resolved by the parent. */
+ pendingText: string | null;
elapsedLabel: string | null;
}) {
- const { t } = useLocale();
-
if (message.role === 'user') {
const text = message.parts
.filter((part) => part.type === 'text')
@@ -266,11 +306,15 @@ function MessageBlock({
return null;
})}
- {pendingPhase === 'writing' ? (
-
+ {pendingPhase === 'writing' && pendingText ? (
+
) : null}
+
+ {/* Only once the turn has settled: a disclosure that appears mid-stream
+ invites a click onto a list that is still growing underneath it. */}
+ {isLast && pendingPhase !== null ? null : }
);
}
diff --git a/app/src/components/deliverables.tsx b/app/src/components/deliverables.tsx
new file mode 100644
index 0000000..99069bd
--- /dev/null
+++ b/app/src/components/deliverables.tsx
@@ -0,0 +1,394 @@
+'use client';
+
+/**
+ * The deliverables surface: pick a template, name a subject, watch a document
+ * assemble beside an account of how it was assembled.
+ *
+ * The two-column layout is the argument the whole feature makes. A generated
+ * humanitarian brief with no visible working is an artefact nobody should
+ * forward — the figures look identical whether they were retrieved or invented.
+ * Putting the trace next to the prose, at the same weight, means the reader
+ * never has to go looking for the provenance; it is already on screen, ticking,
+ * while the section it belongs to is being written.
+ */
+
+import { useChat } from '@ai-sdk/react';
+import { DefaultChatTransport } from 'ai';
+import { useCallback, useEffect, useMemo, useRef, useState } from 'react';
+
+import type { DeliverableUIMessage } from '@/app/api/deliverables/route';
+import { assembleDocument, documentFilename, foldRun } from '@/lib/agent/render';
+import type { TraceEvent } from '@/lib/agent/types';
+import { WORKFLOWS, WORKFLOW_IDS, type WorkflowId } from '@/lib/agent/workflows';
+import { useLocale } from '@/lib/i18n/context';
+import { IconCheck, IconCopy, IconDeliverable, IconDownload, IconWarning } from './icons';
+import { Markdown } from './markdown';
+import { PendingStatus } from './pending-status';
+import { TracePanel } from './trace-panel';
+
+/** Seconds before the elapsed counter appears — matches the chat's threshold. */
+const ELAPSED_THRESHOLD_S = 3;
+
+/** Trace events out of the run's single assistant message. */
+function traceEvents(message: DeliverableUIMessage | undefined): TraceEvent[] {
+ if (!message) return [];
+ return message.parts
+ .filter((part): part is { type: 'data-trace'; data: TraceEvent } => part.type === 'data-trace')
+ .map((part) => part.data);
+}
+
+/* ------------------------------------------------------------------ *
+ * The page
+ * ------------------------------------------------------------------ */
+
+export function Deliverables() {
+ const { t } = useLocale();
+ const d = t.deliverables;
+
+ const [templateId, setTemplateId] = useState('situation-brief');
+ const [subject, setSubject] = useState('');
+ const [started, setStarted] = useState(false);
+
+ const { messages, sendMessage, status, stop, error, setMessages } =
+ useChat({
+ transport: new DefaultChatTransport({
+ api: '/api/deliverables',
+ // The route takes a template and a subject, not a message history: a
+ // run is not a conversation, and shipping the whole `messages` array
+ // would send the server a shape it has no use for.
+ prepareSendMessagesRequest: ({ messages: sent, body }) => ({
+ body: {
+ workflowId: body?.workflowId,
+ subject: sent
+ .at(-1)
+ ?.parts.filter((part) => part.type === 'text')
+ .map((part) => part.text)
+ .join(' ')
+ .trim(),
+ },
+ }),
+ }),
+ });
+
+ const busy = status === 'submitted' || status === 'streaming';
+ const events = useMemo(
+ () => traceEvents(messages.find((message) => message.role === 'assistant')),
+ [messages],
+ );
+
+ const template = WORKFLOWS[templateId];
+ const run = useMemo(() => foldRun(events, template.title), [events, template.title]);
+
+ // One continuous count for the whole run, as in chat: a wait that moves
+ // through six sections is still one wait to the person watching it.
+ const [elapsedSeconds, setElapsedSeconds] = useState(null);
+ useEffect(() => {
+ if (!busy) return;
+ const startedAt = Date.now();
+ const tick = () => {
+ const seconds = Math.floor((Date.now() - startedAt) / 1000);
+ setElapsedSeconds(seconds >= ELAPSED_THRESHOLD_S ? seconds : null);
+ };
+ tick();
+ const id = setInterval(tick, 1000);
+ return () => clearInterval(id);
+ }, [busy]);
+
+ const elapsedLabel = busy && elapsedSeconds !== null ? t.pending.elapsed(elapsedSeconds) : null;
+
+ const start = useCallback(() => {
+ const trimmed = subject.trim();
+ if (!trimmed || busy) return;
+ setStarted(true);
+ void sendMessage({ text: trimmed }, { body: { workflowId: templateId } });
+ }, [subject, busy, sendMessage, templateId]);
+
+ const reset = useCallback(() => {
+ stop();
+ setMessages([]);
+ setStarted(false);
+ }, [stop, setMessages]);
+
+ if (!started) {
+ return (
+
+ );
+ }
+
+ const markdown = assembleDocument(run.title, run.sections);
+ const hasContent = run.sections.some((section) => section.markdown.trim());
+ // Stopped, errored, or the stream ended without the run saying it finished —
+ // the serverless timeout looks exactly like the last of those from here.
+ const partial = !busy && !run.finished && hasContent;
+
+ return (
+
+
+
+
{run.title}
+
{d.templates[templateId].name}
+
+
+ {busy ? (
+
+ ) : (
+
+ )}
+
+
+
+
+ {error ? (
+
+ {error.message || t.errorFallback}
+
+ ) : null}
+
+ {/* Document left, working right — on a narrow screen the trace drops
+ below the document rather than beside it, since the document is what
+ someone on a phone came for. */}
+
+ );
+}
+
+/* ------------------------------------------------------------------ *
+ * Export
+ * ------------------------------------------------------------------ */
+
+const buttonClass =
+ 'hai-eyebrow border border-border-strong px-2.5 py-1 text-muted transition-colors hover:text-foreground focus:outline-none focus-visible:ring-2 focus-visible:ring-accent disabled:cursor-not-allowed disabled:opacity-40';
+
+/**
+ * Copy and download, both entirely client-side.
+ *
+ * The markdown is already in the browser — it was assembled from the trace
+ * there — so a round trip to ask the server for it again would be a second code
+ * path producing a second version of the same document. A blob URL keeps the
+ * export exactly equal to what is on screen.
+ */
+function ExportButtons({
+ markdown,
+ title,
+ disabled,
+}: {
+ markdown: string;
+ title: string;
+ disabled: boolean;
+}) {
+ const { t } = useLocale();
+ const d = t.deliverables;
+ const [copied, setCopied] = useState(false);
+ const copiedTimer = useRef | undefined>(undefined);
+
+ useEffect(() => () => clearTimeout(copiedTimer.current), []);
+
+ const copy = useCallback(async () => {
+ try {
+ await navigator.clipboard.writeText(markdown);
+ setCopied(true);
+ clearTimeout(copiedTimer.current);
+ copiedTimer.current = setTimeout(() => setCopied(false), 2000);
+ } catch {
+ // Clipboard access is refused in some browsers and every insecure
+ // context. Download still works, so there is no need to alarm anyone.
+ }
+ }, [markdown]);
+
+ const download = useCallback(() => {
+ const blob = new Blob([markdown], { type: 'text/markdown;charset=utf-8' });
+ const url = URL.createObjectURL(blob);
+ const anchor = document.createElement('a');
+ anchor.href = url;
+ anchor.download = documentFilename(title);
+ anchor.click();
+ URL.revokeObjectURL(url);
+ }, [markdown, title]);
+
+ return (
+ <>
+
+
+ >
+ );
+}
+
+function Banner({ tone, children }: { tone: 'notice' | 'accent'; children: React.ReactNode }) {
+ return (
+
+
+ {children}
+
+ );
+}
diff --git a/app/src/components/icons.tsx b/app/src/components/icons.tsx
index 4048905..ff345b4 100644
--- a/app/src/components/icons.tsx
+++ b/app/src/components/icons.tsx
@@ -193,6 +193,25 @@ export function IconDocument(props: IconProps) {
);
}
+/**
+ * Live hazard alerts — concentric waves off an epicentre.
+ *
+ * Deliberately not the warning triangle: `IconWarning` means "something went
+ * wrong here" throughout the app, and a hazard feed returning three flood
+ * alerts is the tool working, not failing.
+ */
+export function IconHazard(props: IconProps) {
+ return (
+
+ );
+}
+
/** Advisory / caution. */
export function IconWarning(props: IconProps) {
return (
@@ -233,9 +252,83 @@ export const PLAYBOOK_ICONS: Record = {
'field-logistics': IconLogisticsRole,
};
-/** Chat tool name → the icon shown while it runs and once it resolves. */
-export const TOOL_ICONS: Record<'search_standards' | 'crisis_updates' | 'humanitarian_data', IconComponent> = {
+/**
+ * Chat tool name → the icon shown while it runs and once it resolves.
+ *
+ * Keyed by the registry names in `lib/tools/index.ts`. A tool registered
+ * without an entry here still renders — `toolIcon` below falls back — but it
+ * renders as a generic live-data glyph, so add it when you add the tool.
+ */
+export const TOOL_ICONS: Record<
+ 'search_standards' | 'crisis_updates' | 'humanitarian_data' | 'hazards_context',
+ IconComponent
+> = {
search_standards: IconSearch,
crisis_updates: IconLiveData,
humanitarian_data: IconLiveData,
+ hazards_context: IconHazard,
};
+
+/** A verified claim — the check the self-check step draws. */
+export function IconCheck(props: IconProps) {
+ return (
+
+ );
+}
+
+/** A claim the evidence did not support. Flown, not ticked. */
+export function IconFlag(props: IconProps) {
+ return (
+
+ );
+}
+
+/** Copy to clipboard — two sheets, offset. */
+export function IconCopy(props: IconProps) {
+ return (
+
+ );
+}
+
+/** Download — out of the document, onto the disk. */
+export function IconDownload(props: IconProps) {
+ return (
+
+ );
+}
+
+/** Deliverables — a document assembled from separate blocks. */
+export function IconDeliverable(props: IconProps) {
+ return (
+
+ );
+}
+
+/**
+ * The icon for a tool by name, for surfaces that render whatever the registry
+ * happens to hold rather than a fixed set — the trace panel, which must not
+ * break when a tool is added to `lib/tools/index.ts` before anyone gets round
+ * to drawing a glyph for it.
+ */
+export function toolIcon(name: string): IconComponent {
+ return (
+ (TOOL_ICONS as Record)[name] ??
+ (name.includes('standard') ? IconSearch : IconLiveData)
+ );
+}
diff --git a/app/src/components/nav-links.tsx b/app/src/components/nav-links.tsx
index 347e639..cd62cfe 100644
--- a/app/src/components/nav-links.tsx
+++ b/app/src/components/nav-links.tsx
@@ -16,6 +16,7 @@ export function NavLinks() {
const links = [
{ href: '/', label: t.nav.chat },
+ { href: '/deliverables', label: t.nav.deliverables },
{ href: '/playbooks', label: t.nav.playbooks },
{ href: '/guides', label: t.nav.guides },
{ href: '/about', label: t.nav.about },
diff --git a/app/src/components/show-working.tsx b/app/src/components/show-working.tsx
new file mode 100644
index 0000000..6e9b6c8
--- /dev/null
+++ b/app/src/components/show-working.tsx
@@ -0,0 +1,53 @@
+'use client';
+
+/**
+ * The per-message "show working" disclosure in chat.
+ *
+ * Collapsed by default, and that default is the whole design decision. In the
+ * deliverables view the trace is a column of its own because the reader is
+ * producing a document they will forward, and provenance is the point. In chat
+ * they are mid-conversation and the answer is the point — so the working is one
+ * click away rather than in the way, and the row of tool activity above it
+ * already tells them grounding happened at all.
+ *
+ * The rows themselves are `TraceList`, exactly as rendered on the deliverables
+ * page, so what "consulted this source" looks like cannot drift between the two
+ * surfaces.
+ */
+
+import { useMemo, useState } from 'react';
+
+import { traceFromMessage } from '@/lib/agent/chat-trace';
+import { useLocale } from '@/lib/i18n/context';
+import { TraceList } from './trace-panel';
+
+export function ShowWorking({
+ message,
+}: {
+ message: { parts: ReadonlyArray<{ type: string } & Record> };
+}) {
+ const { t } = useLocale();
+ const [open, setOpen] = useState(false);
+
+ const events = useMemo(() => traceFromMessage(message), [message]);
+ if (events.length === 0) return null;
+
+ return (
+
+
+
+ {open ? (
+
+
+
+ ) : null}
+
+ );
+}
diff --git a/app/src/components/tool-activity.tsx b/app/src/components/tool-activity.tsx
index d6bac4c..58486c8 100644
--- a/app/src/components/tool-activity.tsx
+++ b/app/src/components/tool-activity.tsx
@@ -6,7 +6,19 @@ import { IconWarning, TOOL_ICONS } from './icons';
type Part = HaiUIMessage['parts'][number];
-const KNOWN_TOOLS = ['search_standards', 'crisis_updates', 'humanitarian_data'] as const;
+/**
+ * Tools this row knows how to describe, keyed by their registry names in
+ * `lib/tools/index.ts`. A tool missing from here renders nothing at all — the
+ * user watches a silent gap while it runs — so adding a tool to the registry
+ * means adding it here, to `TOOL_ICONS`, and to `toolActivity` in the
+ * dictionary.
+ */
+const KNOWN_TOOLS = [
+ 'search_standards',
+ 'crisis_updates',
+ 'humanitarian_data',
+ 'hazards_context',
+] as const;
type KnownTool = (typeof KNOWN_TOOLS)[number];
function toolName(partType: string): KnownTool | undefined {
@@ -22,6 +34,21 @@ function detail(part: Part): string | undefined {
const country = typeof input.country === 'string' ? input.country : undefined;
return country ? `${country} — ${input.query}` : input.query;
}
+ // hazards_context, checked before the bare `country_iso3` branch below
+ // because it carries that field too — matching on it first would describe a
+ // country hazard sweep as "SDN" and drop the part that matters.
+ //
+ // Which feeds were asked for is worth showing here in a way it is not for the
+ // other tools: this is the one call that fans out across four upstream
+ // sources, so "GDACS had nothing" and "we never asked GDACS" look identical
+ // without it.
+ if (input.scope === 'global' || input.scope === 'country') {
+ const sources = Array.isArray(input.sources)
+ ? input.sources.filter((entry): entry is string => typeof entry === 'string')
+ : [];
+ const scope = input.scope === 'global' ? 'global' : String(input.country_iso3 ?? 'country');
+ return sources.length > 0 ? `${scope} — ${sources.join(', ')}` : scope;
+ }
if (typeof input.country_iso3 === 'string') {
const dataset =
typeof input.dataset === 'string' ? input.dataset.replace(/_/g, ' ') : '';
diff --git a/app/src/components/trace-panel.tsx b/app/src/components/trace-panel.tsx
new file mode 100644
index 0000000..31ee1c3
--- /dev/null
+++ b/app/src/components/trace-panel.tsx
@@ -0,0 +1,386 @@
+'use client';
+
+/**
+ * The trace panel: what the machine did, in the order it did it.
+ *
+ * This is not a debug view that happened to get shipped. For a humanitarian
+ * deliverable the working *is* part of the product — an analyst who cannot see
+ * which source a figure came from, or that the funding section ran while the
+ * funding API was down, cannot responsibly forward the document. So the panel
+ * is written for that reader rather than for an engineer: tool calls read as
+ * sources consulted, verification reads as claims checked, and a failure reads
+ * as a named source that did not answer.
+ *
+ * Two exported surfaces. `TracePanel` is the full right-hand column of the
+ * deliverables run view. `TraceList` is the same event rendering without the
+ * plan checklist or the frame, which the chat's per-message "show working"
+ * disclosure reuses so the two views can never drift apart.
+ */
+
+import { useEffect, useMemo, useState } from 'react';
+
+import { useLocale } from '@/lib/i18n/context';
+import type { PlannedSection, TraceEvent, Verdict } from '@/lib/agent/types';
+import { IconCheck, IconFlag, IconMark, IconWarning, toolIcon } from './icons';
+
+/* ------------------------------------------------------------------ *
+ * Plan checklist
+ * ------------------------------------------------------------------ */
+
+type SectionStatus = 'pending' | 'active' | 'done' | 'flagged';
+
+/**
+ * Where each planned section has got to, derived from the events rather than
+ * tracked separately — there is one source of truth for a run, and it is the
+ * event stream.
+ */
+export function sectionStatuses(events: TraceEvent[]): Map {
+ const statuses = new Map();
+
+ for (const event of events) {
+ if (event.type === 'plan-created') {
+ for (const section of event.sections) statuses.set(section.id, 'pending');
+ } else if (event.type === 'step-started' && event.sectionId) {
+ if (statuses.get(event.sectionId) !== 'done') statuses.set(event.sectionId, 'active');
+ } else if (event.type === 'draft-section') {
+ statuses.set(event.sectionId, 'done');
+ } else if (event.type === 'section-verified') {
+ statuses.set(event.sectionId, event.flagged > 0 ? 'flagged' : 'done');
+ }
+ }
+
+ return statuses;
+}
+
+function PlanChecklist({
+ sections,
+ statuses,
+}: {
+ sections: PlannedSection[];
+ statuses: Map;
+}) {
+ const { t } = useLocale();
+
+ return (
+
+ {sections.map((section, index) => {
+ const status = statuses.get(section.id) ?? 'pending';
+ return (
+
+ );
+ })}
+
+ );
+}
+
+/* ------------------------------------------------------------------ *
+ * Event rows
+ * ------------------------------------------------------------------ */
+
+const VERDICT_TONE: Record = {
+ supported: 'text-subtle',
+ unsupported: 'text-accent',
+ unverifiable: 'text-accent',
+};
+
+/**
+ * One event as a line the reader can act on.
+ *
+ * Returning `null` is normal and intended: `draft-delta` fires hundreds of
+ * times a run and belongs in the document, not the log, and `step-finished`
+ * only says anything when the step degraded. A panel that rendered every event
+ * would be unreadable at exactly the moment someone needs to read it.
+ */
+function TraceRow({ event }: { event: TraceEvent }) {
+ const { t } = useLocale();
+ const d = t.deliverables;
+
+ switch (event.type) {
+ case 'step-started':
+ return (
+
+ {d.steps[event.kind]}
+ {event.label}
+
+ );
+
+ case 'tool-called':
+ return (
+
+ {event.tool}
+ {event.args ? {event.args} : null}
+
+ );
+
+ case 'tool-result':
+ return (
+
+ ) : (
+
+ )
+ }
+ >
+ {event.summary}
+
+ );
+
+ case 'check-run':
+ return (
+
+ ) : (
+
+ )
+ }
+ >
+
+ {d.verdicts[event.verdict]}
+
+ {truncate(event.claim, 110)}
+ {event.source ? (
+ · {event.source}
+ ) : null}
+
+ );
+
+ case 'source-error':
+ return (
+ }>
+ {event.source}
+ {truncate(event.message, 140)}
+
+ );
+
+ case 'step-finished':
+ if (event.ok || !event.note) return null;
+ return (
+ }>
+ {truncate(event.note, 140)}
+
+ );
+
+ case 'budget-wait':
+ // The per-minute wait is rendered live by `TraceList` from the tail of
+ // the stream, not here: once it is over there is nothing left to say,
+ // and a row frozen at "resumes in 44s" would be a lie about the past.
+ // The per-day one is not a wait at all — it is where the run ended —
+ // so it stays in the log where the reader can still see it.
+ if (event.scope !== 'tokens-per-day') return null;
+ return (
+ }>
+ {d.budgetDailyExhausted}
+
+ );
+
+ case 'workflow-error':
+ return (
+ }>
+ {event.message}
+
+ );
+
+ case 'workflow-done':
+ return (
+
+
+ {event.flagged > 0 ? d.doneWithFlags(event.flagged) : d.done}
+
+
+ );
+
+ default:
+ return null;
+ }
+}
+
+/**
+ * Resolved outside the component body on purpose: `toolIcon` looks a name up in
+ * the registry at call time, and binding its result to a capitalised local
+ * inside a render reads to React's lint rules as defining a component per
+ * render. The glyph still varies by tool, which is the point — the trace shows
+ * whatever the registry holds, including tools added after this was written.
+ */
+function renderToolIcon(name: string) {
+ const Icon = toolIcon(name);
+ return ;
+}
+
+function Row({
+ icon,
+ tone,
+ children,
+}: {
+ icon?: React.ReactNode;
+ tone?: 'heading';
+ children: React.ReactNode;
+}) {
+ return (
+
+ {icon}
+ {children}
+
+ );
+}
+
+/* ------------------------------------------------------------------ *
+ * The pacing row
+ * ------------------------------------------------------------------ */
+
+type BudgetWait = Extract;
+
+/**
+ * The per-minute wait the run is inside *right now*, or null.
+ *
+ * "Right now" is exactly the tail of the stream: the engine emits nothing
+ * between a `budget-wait` and the `budget-resumed` that ends it, so a wait that
+ * is still the last event is a wait still being served, and any event after it
+ * — the resume, the step that followed, the workflow error after a daily wall —
+ * has already cleared it.
+ */
+function activeBudgetWait(events: TraceEvent[]): BudgetWait | null {
+ const last = events[events.length - 1];
+ if (!last || last.type !== 'budget-wait') return null;
+ return last.scope === 'tokens-per-minute' ? last : null;
+}
+
+/**
+ * A wait, named and counted down.
+ *
+ * The count is a one-second interval rather than a re-render of the whole
+ * panel: only this row's number changes, and the trace above it must not
+ * reflow once a second while somebody is reading it.
+ *
+ * The deadline is anchored to when this row mounted plus `waitMs`, not to
+ * `event.at + waitMs`, because `at` is the *server's* clock and this subtraction
+ * would otherwise be as wrong as the difference between the two machines — a
+ * browser thirty seconds behind would show a countdown that never ends. The
+ * event arrives over an open stream within milliseconds of being emitted, so
+ * mount time is the better anchor by a wide margin.
+ *
+ * Deliberately still: a muted, unpulsed mark. The pulse in this design means
+ * "work is happening here", and no work is happening here.
+ */
+function BudgetWaitRow({ event }: { event: BudgetWait }) {
+ const { t } = useLocale();
+ const [remaining, setRemaining] = useState(() => Math.ceil(event.waitMs / 1000));
+
+ useEffect(() => {
+ const deadline = Date.now() + event.waitMs;
+ const tick = () => setRemaining(Math.max(0, Math.ceil((deadline - Date.now()) / 1000)));
+ tick();
+ const id = setInterval(tick, 1000);
+ return () => clearInterval(id);
+ }, [event.waitMs]);
+
+ return (
+ }>
+ {t.deliverables.budgetWaiting}
+ {/* Hidden from assistive tech: the sentence beside it is the message,
+ and a number that changes every second inside the panel's polite
+ live region would talk over everything else the run has to say. */}
+
+ {' · '}
+ {t.deliverables.budgetResumesIn(remaining)}
+
+
+ );
+}
+
+function truncate(text: string, max: number): string {
+ const flat = text.replace(/\s+/g, ' ').trim();
+ return flat.length <= max ? flat : `${flat.slice(0, max).trimEnd()}…`;
+}
+
+/* ------------------------------------------------------------------ *
+ * Exported surfaces
+ * ------------------------------------------------------------------ */
+
+/** The event log alone — reused by the chat's "show working" disclosure. */
+export function TraceList({ events }: { events: TraceEvent[] }) {
+ const { t } = useLocale();
+ const wait = activeBudgetWait(events);
+
+ if (events.length === 0) {
+ return
{t.deliverables.traceEmpty}
;
+ }
+
+ return (
+
+ {events.map((event, index) => (
+
+ ))}
+ {/* Keyed by the wait it belongs to, so a second wait gets a second row
+ with a freshly anchored countdown rather than inheriting the first's. */}
+ {wait ? : null}
+
+ );
+}
+
+/** The full right-hand column: the plan, then the log. */
+export function TracePanel({ events, busy }: { events: TraceEvent[]; busy: boolean }) {
+ const { t } = useLocale();
+ const d = t.deliverables;
+
+ const plan = useMemo(
+ () => events.find((event) => event.type === 'plan-created'),
+ [events],
+ );
+ const statuses = useMemo(() => sectionStatuses(events), [events]);
+
+ return (
+
+ );
+}
diff --git a/app/src/lib/agent/__testing__/mock-model.ts b/app/src/lib/agent/__testing__/mock-model.ts
new file mode 100644
index 0000000..b52f739
--- /dev/null
+++ b/app/src/lib/agent/__testing__/mock-model.ts
@@ -0,0 +1,130 @@
+/**
+ * A scripted language model for engine tests.
+ *
+ * The engine's job is to sequence model calls and turn what comes back into a
+ * document plus a trace. Testing that against a real endpoint would test the
+ * endpoint; testing it against a model that replies by inspecting the prompt it
+ * was given tests the engine — including the paths that only occur when the
+ * model misbehaves, which are the ones that matter here and which a real model
+ * cannot be asked to produce on demand.
+ *
+ * Not a `.test.ts` file, so vitest does not collect it as a suite.
+ */
+
+import type { LanguageModel } from 'ai';
+import { MockLanguageModelV4, convertArrayToReadableStream } from 'ai/test';
+
+const USAGE = {
+ inputTokens: { total: 100, noCache: 100, cacheRead: 0, cacheWrite: 0 },
+ outputTokens: { total: 50, text: 50, reasoning: 0 },
+};
+
+/**
+ * The provider spec takes a finish reason as `{ unified, raw }`, not a bare
+ * string, and the SDK reads `.unified` to decide whether client-side tools may
+ * run. A mock that returns the string silently produces a step with a tool call
+ * in it and no tool result — the call streams, `execute` is never invoked, and
+ * the loop stops after one step. Worth stating plainly: it fails as "the engine
+ * did not gather anything" rather than as a type error.
+ */
+function finishReason(toolCalls: boolean) {
+ return toolCalls
+ ? { unified: 'tool-calls' as const, raw: 'tool_calls' }
+ : { unified: 'stop' as const, raw: 'stop' };
+}
+
+type Part =
+ | { kind: 'text'; text: string }
+ | { kind: 'tool'; toolName: string; input: Record };
+
+export function text(value: string): Part {
+ return { kind: 'text', text: value };
+}
+
+export function toolCall(toolName: string, input: Record = {}): Part {
+ return { kind: 'tool', toolName, input };
+}
+
+/** Everything the model was sent, flattened, so a script can branch on it. */
+export function promptText(options: { prompt: unknown }): string {
+ const messages = options.prompt as Array<{ role: string; content: unknown }>;
+ const chunks: string[] = [];
+ for (const message of messages ?? []) {
+ if (typeof message.content === 'string') {
+ chunks.push(message.content);
+ continue;
+ }
+ for (const part of (message.content as Array>) ?? []) {
+ if (part && part.type === 'text' && typeof part.text === 'string') chunks.push(part.text);
+ }
+ }
+ return chunks.join('\n');
+}
+
+/**
+ * Build a model whose reply is chosen by `script` from the prompt it receives.
+ * Returning an empty array is a model that says nothing, which is itself a case
+ * the engine has to handle.
+ */
+export function scriptedModel(script: (prompt: string, callIndex: number) => Part[]): LanguageModel {
+ let callIndex = 0;
+
+ return new MockLanguageModelV4({
+ doStream: async (options) => {
+ const parts = script(promptText(options), callIndex++);
+ const chunks: unknown[] = [{ type: 'stream-start', warnings: [] }];
+ let toolCalls = 0;
+
+ for (const [index, part] of parts.entries()) {
+ if (part.kind === 'text') {
+ const id = `t${index}`;
+ chunks.push({ type: 'text-start', id });
+ // Split so the engine's delta handling is exercised rather than
+ // receiving each section as a single block.
+ for (const word of part.text.split(/(?<= )/)) {
+ chunks.push({ type: 'text-delta', id, delta: word });
+ }
+ chunks.push({ type: 'text-end', id });
+ } else {
+ toolCalls += 1;
+ chunks.push({
+ type: 'tool-call',
+ toolCallId: `call-${callIndex}-${index}`,
+ toolName: part.toolName,
+ input: JSON.stringify(part.input),
+ });
+ }
+ }
+
+ chunks.push({
+ type: 'finish',
+ finishReason: finishReason(toolCalls > 0),
+ usage: USAGE,
+ });
+
+ // Cast at the boundary: the provider spec's stream-part union is not
+ // exported from a direct dependency, and spelling it out here would pin
+ // the test helper to a package the app does not otherwise import.
+ return { stream: convertArrayToReadableStream(chunks) } as never;
+ },
+
+ doGenerate: async (options) => {
+ const parts = script(promptText(options), callIndex++);
+ return {
+ content: parts.map((part) =>
+ part.kind === 'text'
+ ? { type: 'text' as const, text: part.text }
+ : {
+ type: 'tool-call' as const,
+ toolCallId: `gen-${callIndex}`,
+ toolName: part.toolName,
+ input: JSON.stringify(part.input),
+ },
+ ),
+ finishReason: finishReason(parts.some((part) => part.kind === 'tool')),
+ usage: USAGE,
+ warnings: [],
+ } as never;
+ },
+ }) as unknown as LanguageModel;
+}
diff --git a/app/src/lib/agent/chat-trace.test.ts b/app/src/lib/agent/chat-trace.test.ts
new file mode 100644
index 0000000..e9bea2c
--- /dev/null
+++ b/app/src/lib/agent/chat-trace.test.ts
@@ -0,0 +1,132 @@
+import { describe, expect, it } from 'vitest';
+
+import { traceFromMessage } from './chat-trace';
+
+describe('traceFromMessage', () => {
+ it('reports a finished search as a source consulted', () => {
+ const events = traceFromMessage({
+ parts: [
+ { type: 'text', text: 'The minimum is 15 litres.' },
+ {
+ type: 'tool-search_standards',
+ toolCallId: 'c1',
+ state: 'output-available',
+ input: { query: 'water per person per day', source: 'all' },
+ output: { chunks: [{ source: 'sphere' }, { source: 'sphere' }] },
+ },
+ ],
+ });
+
+ expect(events).toEqual([
+ {
+ type: 'tool-called',
+ at: 0,
+ stepId: 'chat',
+ callId: 'c1',
+ tool: 'search_standards',
+ args: 'query=water per person per day source=all',
+ },
+ {
+ type: 'tool-result',
+ at: 0,
+ stepId: 'chat',
+ callId: 'c1',
+ tool: 'search_standards',
+ ok: true,
+ summary: '2 chunks',
+ },
+ ]);
+ });
+
+ it('reports a dataset with no coverage as a call that did not answer', () => {
+ const events = traceFromMessage({
+ parts: [
+ {
+ type: 'tool-humanitarian_data',
+ toolCallId: 'c2',
+ state: 'output-available',
+ input: { country_iso3: 'SDN', dataset: 'funding' },
+ output: { available: false, detail: 'HAPI holds no funding data for SDN.' },
+ },
+ ],
+ });
+
+ expect(events[1]).toMatchObject({
+ ok: false,
+ summary: 'HAPI holds no funding data for SDN.',
+ });
+ });
+
+ it('reports an empty retrieval as a call that did not answer', () => {
+ const events = traceFromMessage({
+ parts: [
+ {
+ type: 'tool-search_standards',
+ toolCallId: 'c3',
+ state: 'output-available',
+ input: { query: 'blockchain' },
+ output: { chunks: [], notice: 'No passages matched this query.' },
+ },
+ ],
+ });
+
+ expect(events[1]).toMatchObject({ ok: false, summary: 'No passages matched this query.' });
+ });
+
+ it('carries a tool failure through rather than reporting a success', () => {
+ const events = traceFromMessage({
+ parts: [
+ {
+ type: 'tool-crisis_updates',
+ toolCallId: 'c4',
+ state: 'output-error',
+ input: { query: 'displacement' },
+ errorText: 'feed unreachable',
+ },
+ ],
+ });
+
+ expect(events[1]).toMatchObject({ ok: false, summary: 'feed unreachable' });
+ });
+
+ it('leaves a call still in flight without a result row', () => {
+ const events = traceFromMessage({
+ parts: [
+ {
+ type: 'tool-search_standards',
+ toolCallId: 'c5',
+ state: 'input-available',
+ input: { query: 'shelter' },
+ },
+ ],
+ });
+
+ expect(events).toHaveLength(1);
+ expect(events[0].type).toBe('tool-called');
+ });
+
+ it('produces nothing for a turn that called no tool', () => {
+ expect(traceFromMessage({ parts: [{ type: 'text', text: 'Hello.' }] })).toEqual([]);
+ });
+
+ /*
+ * Load-bearing, and the reason this file exists rather than the chat route
+ * emitting events: chat answers are not verified. A verdict here would read as
+ * the same guarantee the deliverables page makes, and it is not one.
+ */
+ it('never claims a claim was checked', () => {
+ const events = traceFromMessage({
+ parts: [
+ {
+ type: 'tool-search_standards',
+ toolCallId: 'c6',
+ state: 'output-available',
+ input: { query: 'water' },
+ output: { chunks: [{ source: 'sphere' }] },
+ },
+ ],
+ });
+
+ expect(events.some((event) => event.type === 'check-run')).toBe(false);
+ });
+});
diff --git a/app/src/lib/agent/chat-trace.ts b/app/src/lib/agent/chat-trace.ts
new file mode 100644
index 0000000..76b88c5
--- /dev/null
+++ b/app/src/lib/agent/chat-trace.ts
@@ -0,0 +1,121 @@
+/**
+ * The chat's answers, expressed in the same trace vocabulary as a deliverable.
+ *
+ * # Why this reads the message rather than the server emitting it
+ *
+ * The deliverables route emits `TraceEvent`s because the engine knows things the
+ * client cannot reconstruct — which evidence a section was given, what the
+ * verifier decided. A chat turn has no such hidden state. Everything it did is
+ * already in the message the client is holding: which tools were called, with
+ * what arguments, and what came back. Emitting a parallel event stream from the
+ * chat route would add a second source of truth for facts the first one already
+ * carries, and put a write on a hot path shared with other work in flight.
+ *
+ * So the chat trace is derived here, from the message parts, and rendered by the
+ * same `TraceList` the deliverables panel uses. If the two ever disagree it will
+ * be because the message and the panel disagree, which is visible.
+ *
+ * # What it deliberately does not claim
+ *
+ * No `check-run` events. Chat answers are not verified — the self-check is a
+ * workflow step, and the token budget for a conversational turn does not stretch
+ * to one. Showing a verdict here would be the single most misleading thing this
+ * file could do: a reader who has seen claims checked on the deliverables page
+ * would reasonably read the same panel in chat as the same guarantee. The chat
+ * trace says which sources were consulted, and nothing more.
+ */
+
+import { summariseArgs } from './evidence';
+import type { TraceEvent } from './types';
+
+/** The shape this needs from a UI message, kept structural so `HaiUIMessage` stays in the route. */
+interface MessageLike {
+ parts: ReadonlyArray<{ type: string } & Record>;
+}
+
+const STEP_ID = 'chat';
+
+/**
+ * A tool result summarised for the trace, without the harvesting the engine
+ * does. Chat tool output is rendered properly by the citations panel; this only
+ * needs to say whether the call worked and roughly what came back.
+ */
+function summarise(output: unknown): { ok: boolean; summary: string } {
+ if (output === null || output === undefined) return { ok: false, summary: 'no result' };
+
+ if (typeof output === 'object') {
+ const record = output as Record;
+
+ if (typeof record.error === 'string' && record.error) {
+ return { ok: false, summary: record.error };
+ }
+ if (record.available === false) {
+ const detail = typeof record.detail === 'string' ? record.detail : undefined;
+ const reason = typeof record.reason === 'string' ? record.reason : undefined;
+ return { ok: false, summary: detail ?? reason ?? 'no data available' };
+ }
+ if (typeof record.notice === 'string' && record.notice) {
+ return { ok: false, summary: record.notice };
+ }
+
+ // The payload is whichever array came back — `chunks`, `updates`, `figures`.
+ for (const [key, value] of Object.entries(record)) {
+ if (key !== 'errors' && Array.isArray(value) && value.length > 0) {
+ return {
+ ok: true,
+ summary: `${value.length} ${key.replace(/([a-z])([A-Z])/g, '$1 $2').toLowerCase()}`,
+ };
+ }
+ }
+ return { ok: true, summary: 'result returned' };
+ }
+
+ return { ok: true, summary: String(output).slice(0, 120) };
+}
+
+/**
+ * Trace events for one assistant message, in message order.
+ *
+ * `at` is 0 throughout: a finished chat message carries no timestamps, and
+ * inventing them from render time would put a plausible-looking wrong number in
+ * front of someone who has been told this panel is the honest account.
+ */
+export function traceFromMessage(message: MessageLike): TraceEvent[] {
+ const events: TraceEvent[] = [];
+
+ for (const [index, part] of message.parts.entries()) {
+ if (!part.type.startsWith('tool-')) continue;
+
+ const tool = part.type.slice('tool-'.length);
+ const callId = typeof part.toolCallId === 'string' ? part.toolCallId : `${tool}-${index}`;
+ const state = typeof part.state === 'string' ? part.state : undefined;
+
+ events.push({
+ type: 'tool-called',
+ at: 0,
+ stepId: STEP_ID,
+ callId,
+ tool,
+ args: summariseArgs(part.input),
+ });
+
+ if (state === 'output-error') {
+ events.push({
+ type: 'tool-result',
+ at: 0,
+ stepId: STEP_ID,
+ callId,
+ tool,
+ ok: false,
+ summary: typeof part.errorText === 'string' ? part.errorText : 'the tool failed',
+ });
+ } else if (state === 'output-available') {
+ const { ok, summary } = summarise(part.output);
+ events.push({ type: 'tool-result', at: 0, stepId: STEP_ID, callId, tool, ok, summary });
+ }
+ // A call still in flight gets no result row — `ToolActivity` above the
+ // disclosure is already showing it running.
+ }
+
+ return events;
+}
diff --git a/app/src/lib/agent/engine.test.ts b/app/src/lib/agent/engine.test.ts
new file mode 100644
index 0000000..6102f4c
--- /dev/null
+++ b/app/src/lib/agent/engine.test.ts
@@ -0,0 +1,385 @@
+import { tool, type ToolSet } from 'ai';
+import { describe, expect, it } from 'vitest';
+import { z } from 'zod';
+
+import { scriptedModel, text, toolCall } from './__testing__/mock-model';
+import { runWorkflow, selectTools } from './engine';
+import { TokenPacer } from './pacer';
+import { TokenBudget } from '@/lib/llm/rate-limit';
+import { UNVERIFIED_MARK } from './render';
+import type { TraceEvent, WorkflowDefinition } from './types';
+
+const pacer = () => new TokenPacer(new TokenBudget());
+
+/** A two-section brief plus the synthesised caveats block — a full run, fast. */
+const workflow: WorkflowDefinition = {
+ id: 'test-brief',
+ title: 'Test brief',
+ subjectKind: 'country',
+ description: 'test',
+ sections: [
+ {
+ id: 'needs',
+ heading: 'Needs',
+ tools: ['humanitarian_data'],
+ gatherBrief: 'Get the caseload.',
+ brief: 'Report the caseload.',
+ },
+ {
+ id: 'sources',
+ heading: 'Sources and caveats',
+ tools: [],
+ gatherBrief: '',
+ brief: '',
+ synthesised: 'sources-and-caveats',
+ skipVerification: true,
+ },
+ ],
+};
+
+const tools: ToolSet = {
+ humanitarian_data: tool({
+ description: 'Country humanitarian figures.',
+ inputSchema: z.object({ country_iso3: z.string(), dataset: z.string() }),
+ execute: async () => ({
+ figures: [
+ {
+ source: 'HDX HAPI',
+ indicator: 'people_in_need',
+ value: 24800000,
+ reference_period: '2026-01 to 2026-12',
+ },
+ ],
+ }),
+ }),
+};
+
+/**
+ * The script a well-behaved model would follow for the workflow above: search
+ * once, then stop searching. Fresh per test, because the gather step is a loop
+ * and "have I already called the tool" is part of what is being scripted.
+ */
+function goodRun(): (prompt: string) => ReturnType[] {
+ let gathered = false;
+ return (prompt) => {
+ if (prompt.includes('ISO 3166-1 alpha-3 code')) return [text('Sudan|SDN')];
+ if (prompt.includes('Retrieve the evidence')) {
+ if (gathered) return [text('')];
+ gathered = true;
+ return [toolCall('humanitarian_data', { country_iso3: 'SDN', dataset: 'humanitarian_needs' })];
+ }
+ if (prompt.includes('CLAIMS:')) return [text('1|supported|HDX HAPI · 2026-01 to 2026-12')];
+ return [
+ text('An estimated 24,800,000 people are in need, reference period 2026-01 to 2026-12 [e1].'),
+ ];
+ };
+}
+
+async function collect(generator: AsyncGenerator): Promise {
+ const events: TraceEvent[] = [];
+ for await (const event of generator) events.push(event);
+ return events;
+}
+
+function ofType(
+ events: TraceEvent[],
+ type: T,
+): Extract[] {
+ return events.filter((event) => event.type === type) as Extract[];
+}
+
+describe('runWorkflow', () => {
+ it('publishes the section checklist before any of it exists', async () => {
+ const events = await collect(
+ runWorkflow(
+ { workflow, subject: 'sudan' },
+ { model: scriptedModel(goodRun()), tools, pacer: pacer() },
+ ),
+ );
+
+ const plan = ofType(events, 'plan-created')[0];
+ expect(plan).toBeDefined();
+ expect(plan.subject).toBe('Sudan');
+ expect(plan.iso3).toBe('SDN');
+ expect(plan.sections.map((section) => section.id)).toEqual(['needs', 'sources']);
+
+ // The plan must arrive before the first section's work, or the trace panel
+ // has nothing to tick through.
+ expect(events.indexOf(plan)).toBeLessThan(
+ events.findIndex((event) => event.type === 'tool-called'),
+ );
+ });
+
+ it('reports each tool call and what it returned', async () => {
+ const events = await collect(
+ runWorkflow(
+ { workflow, subject: 'Sudan' },
+ { model: scriptedModel(goodRun()), tools, pacer: pacer() },
+ ),
+ );
+
+ const called = ofType(events, 'tool-called')[0];
+ expect(called.tool).toBe('humanitarian_data');
+ expect(called.args).toContain('country_iso3=SDN');
+
+ const result = ofType(events, 'tool-result')[0];
+ expect(result.ok).toBe(true);
+ expect(result.summary).toContain('HDX HAPI');
+ expect(result.callId).toBe(called.callId);
+ });
+
+ it('streams the section as it is written, then publishes it whole', async () => {
+ const events = await collect(
+ runWorkflow(
+ { workflow, subject: 'Sudan' },
+ { model: scriptedModel(goodRun()), tools, pacer: pacer() },
+ ),
+ );
+
+ const deltas = ofType(events, 'draft-delta').filter((event) => event.sectionId === 'needs');
+ expect(deltas.length).toBeGreaterThan(1);
+
+ const section = ofType(events, 'draft-section').find((event) => event.sectionId === 'needs');
+ expect(section?.markdown).toContain('24,800,000');
+ // The citation id has become a label the reader can check.
+ expect(section?.markdown).toContain('(HDX HAPI · 2026-01 to 2026-12)');
+ expect(section?.markdown).not.toContain('[e1]');
+ });
+
+ it('checks each claim and reports the verdict', async () => {
+ const events = await collect(
+ runWorkflow(
+ { workflow, subject: 'Sudan' },
+ { model: scriptedModel(goodRun()), tools, pacer: pacer() },
+ ),
+ );
+
+ const checks = ofType(events, 'check-run');
+ expect(checks).toHaveLength(1);
+ expect(checks[0].verdict).toBe('supported');
+ expect(ofType(events, 'section-verified')[0].flagged).toBe(0);
+ expect(ofType(events, 'workflow-done')[0].flagged).toBe(0);
+ });
+
+ /*
+ * The whole point of the feature, exercised end to end: the model writes a
+ * figure that the tool never returned. It must survive to the reader marked,
+ * be counted in the run total, and be named in the caveats section.
+ */
+ it('flags a drafted figure that the evidence does not contain', async () => {
+ let gathered = false;
+ const model = scriptedModel((prompt) => {
+ if (prompt.includes('ISO 3166-1 alpha-3 code')) return [text('Sudan|SDN')];
+ if (prompt.includes('Retrieve the evidence')) {
+ if (gathered) return [text('')];
+ gathered = true;
+ return [toolCall('humanitarian_data', { country_iso3: 'SDN', dataset: 'humanitarian_needs' })];
+ }
+ if (prompt.includes('CLAIMS:')) return [text('1|unsupported|-')];
+ return [text('An estimated 30,100,000 people are in need across the country.')];
+ });
+
+ const events = await collect(
+ runWorkflow({ workflow, subject: 'Sudan' }, { model, tools, pacer: pacer() }),
+ );
+
+ expect(ofType(events, 'check-run')[0].verdict).toBe('unsupported');
+
+ const verified = ofType(events, 'section-verified')[0];
+ expect(verified.flagged).toBe(1);
+ expect(verified.markdown).toContain(UNVERIFIED_MARK);
+ expect(ofType(events, 'workflow-done')[0].flagged).toBe(1);
+
+ const caveats = ofType(events, 'draft-section').find((event) => event.sectionId === 'sources');
+ expect(caveats?.markdown).toContain('1 claim');
+ });
+
+ /*
+ * The `situation.py` contract: a source that fails degrades its section and
+ * lands in the caveats. It must not end the run, and it must not vanish.
+ */
+ it('degrades the section and names the source when a tool fails', async () => {
+ const failing: ToolSet = {
+ humanitarian_data: tool({
+ description: 'Country humanitarian figures.',
+ inputSchema: z.object({ country_iso3: z.string(), dataset: z.string() }),
+ execute: async () => ({
+ available: false,
+ reason: 'no_coverage',
+ detail: 'HAPI holds no needs data for SDN.',
+ }),
+ }),
+ };
+
+ let gathered = false;
+ const model = scriptedModel((prompt) => {
+ if (prompt.includes('ISO 3166-1 alpha-3 code')) return [text('Sudan|SDN')];
+ if (prompt.includes('Retrieve the evidence')) {
+ if (gathered) return [text('')];
+ gathered = true;
+ return [toolCall('humanitarian_data', { country_iso3: 'SDN', dataset: 'humanitarian_needs' })];
+ }
+ if (prompt.includes('CLAIMS:')) return [text('1|unverifiable|-')];
+ return [text('No caseload figures were available for this country from HDX HAPI.')];
+ });
+
+ const events = await collect(
+ runWorkflow({ workflow, subject: 'Sudan' }, { model, tools: failing, pacer: pacer() }),
+ );
+
+ const sourceErrors = ofType(events, 'source-error');
+ expect(sourceErrors.map((event) => event.message)).toContain(
+ 'HAPI holds no needs data for SDN.',
+ );
+
+ // The run still finishes and still produces a document.
+ expect(ofType(events, 'workflow-done')).toHaveLength(1);
+ const caveats = ofType(events, 'draft-section').find((event) => event.sectionId === 'sources');
+ expect(caveats?.markdown).toContain('HAPI holds no needs data for SDN.');
+ expect(caveats?.markdown).toContain('Source issues (degraded sections)');
+ });
+
+ /*
+ * The grounding failure the whole system prompt exists to prevent, at the
+ * step level: a model that answers a gather step from memory rather than
+ * calling anything. The engine must not accept it.
+ */
+ it('forces a tool call when the gather step tried to answer from memory', async () => {
+ const prompts: string[] = [];
+ const model = scriptedModel((prompt) => {
+ prompts.push(prompt);
+ if (prompt.includes('ISO 3166-1 alpha-3 code')) return [text('Sudan|SDN')];
+ if (prompt.includes('Retrieve the evidence')) {
+ // First attempt: prose instead of a tool call.
+ const attempts = prompts.filter((entry) => entry.includes('Retrieve the evidence')).length;
+ if (attempts === 1) return [text('I already know Sudan has 24.8 million people in need.')];
+ return attempts === 2
+ ? [toolCall('humanitarian_data', { country_iso3: 'SDN', dataset: 'humanitarian_needs' })]
+ : [text('')];
+ }
+ if (prompt.includes('CLAIMS:')) return [text('1|supported|HDX HAPI · people_in_need')];
+ return [text('An estimated 24,800,000 people are in need [e1].')];
+ });
+
+ const events = await collect(
+ runWorkflow({ workflow, subject: 'Sudan' }, { model, tools, pacer: pacer() }),
+ );
+
+ expect(ofType(events, 'tool-called')).toHaveLength(1);
+ expect(ofType(events, 'tool-result')[0].ok).toBe(true);
+ });
+
+ it('assembles the caveats section from the run rather than from the model', async () => {
+ const events = await collect(
+ runWorkflow(
+ { workflow, subject: 'Sudan' },
+ { model: scriptedModel(goodRun()), tools, pacer: pacer() },
+ ),
+ );
+
+ const caveats = ofType(events, 'draft-section').find((event) => event.sectionId === 'sources');
+ expect(caveats?.markdown).toContain('Sources consulted');
+ expect(caveats?.markdown).toContain('HDX HAPI — 1 record');
+ expect(caveats?.markdown).toContain('has not been reviewed by anyone');
+ expect(caveats?.markdown).toMatch(/Generated \d{4}-\d{2}-\d{2}T/);
+ });
+
+ it('ends with an error event rather than a broken stream', async () => {
+ const model = scriptedModel(() => {
+ throw new Error('endpoint down');
+ });
+
+ const events = await collect(
+ runWorkflow(
+ { workflow: { ...workflow, subjectKind: 'topic' }, subject: 'Sudan' },
+ { model, tools, pacer: pacer() },
+ ),
+ );
+
+ // The gather failure is contained; the run still reaches a terminal event.
+ expect(events.at(-1)?.type).toBe('workflow-done');
+ expect(ofType(events, 'source-error').length).toBeGreaterThan(0);
+ });
+});
+
+describe('degraded runs', () => {
+ /*
+ * Both from the same live Sudan brief, and both worse than the failure they
+ * followed. The funding section's draft call hit Groq's per-minute ceiling;
+ * the engine then verified the resulting error message as if it were prose,
+ * so the document carried "This section could not be drafted: Failed after 3
+ * attempts… **[unverified]**" — and Groq's own text, ending "Upgrade to Dev
+ * Tier today at https://console.groq.com/settings/billing", was rendered
+ * verbatim into a humanitarian brief's caveats.
+ */
+ const rateLimit = new Error(
+ 'Failed after 3 attempts. Last error: AI_APICallError: Rate limit reached for model `qwen/qwen3.8-27b`, used 8000, limit 8000. Please try again in 3m48.96s. Need more tokens? Upgrade to Dev Tier today at https://console.groq.com/settings/billing',
+ );
+
+ function runWithFailingDraft() {
+ let gathered = false;
+ return scriptedModel((prompt) => {
+ if (prompt.includes('ISO 3166-1 alpha-3 code')) return [text('Sudan|SDN')];
+ if (prompt.includes('Retrieve the evidence')) {
+ if (gathered) return [text('')];
+ gathered = true;
+ return [toolCall('humanitarian_data', { country_iso3: 'SDN', dataset: 'humanitarian_needs' })];
+ }
+ throw rateLimit;
+ });
+ }
+
+ it('does not verify a section that failed to draft', async () => {
+ const events = await collect(
+ runWorkflow(
+ { workflow, subject: 'Sudan' },
+ { model: runWithFailingDraft(), tools, pacer: pacer() },
+ ),
+ );
+
+ expect(ofType(events, 'check-run')).toHaveLength(0);
+ const section = ofType(events, 'draft-section').find((event) => event.sectionId === 'needs');
+ expect(section?.markdown).not.toContain(UNVERIFIED_MARK);
+ });
+
+ it('never puts an upstream billing URL in the document', async () => {
+ const events = await collect(
+ runWorkflow(
+ { workflow, subject: 'Sudan' },
+ { model: runWithFailingDraft(), tools, pacer: pacer() },
+ ),
+ );
+
+ const caveats = ofType(events, 'draft-section').find((event) => event.sectionId === 'sources');
+ expect(caveats?.markdown).not.toContain('console.groq.com');
+ expect(caveats?.markdown).not.toContain('Upgrade to Dev Tier');
+ expect(caveats?.markdown).toContain('token budget was exhausted');
+ // The operative fact survives: when it is worth retrying. Rendered in
+ // seconds rather than echoing the endpoint's own "3m48.96s", because the
+ // rest of that sentence is the part being thrown away.
+ expect(caveats?.markdown).toContain('229s');
+ // And it does not claim to know *which* ceiling. This fixture's message
+ // names neither TPM nor TPD, and the free tier has both — the old wording
+ // asserted "per-minute" for a refusal that never said so, which is exactly
+ // the guess that hid a per-day exhaustion in production for twelve minutes.
+ expect(caveats?.markdown).not.toContain('per-minute');
+ });
+});
+
+describe('selectTools', () => {
+ /*
+ * The registry is edited by other work in parallel with this engine. A
+ * template naming a tool that does not exist yet must degrade to the tools
+ * that do, not throw — and a tool added to the registry must become available
+ * to every template that names it with no change here.
+ */
+ it('skips a tool that is not registered rather than failing the section', () => {
+ const selected = selectTools(tools, ['humanitarian_data', 'not_yet_built']);
+ expect(selected.names).toEqual(['humanitarian_data']);
+ expect(Object.keys(selected.set)).toEqual(['humanitarian_data']);
+ });
+
+ it('returns nothing when a template names only unknown tools', () => {
+ expect(selectTools(tools, ['nope']).names).toEqual([]);
+ });
+});
diff --git a/app/src/lib/agent/engine.ts b/app/src/lib/agent/engine.ts
new file mode 100644
index 0000000..17f44f5
--- /dev/null
+++ b/app/src/lib/agent/engine.ts
@@ -0,0 +1,959 @@
+/**
+ * The workflow engine: a declared sequence of bounded model calls that produces
+ * a deliverable and a full account of how it was produced.
+ *
+ * # Why this is not an agent loop
+ *
+ * The obvious way to build "write me a situation brief" is one long
+ * tool-calling turn with a big prompt and a high step cap. HAI cannot use that,
+ * for three reasons that all point the same way.
+ *
+ * The first is the token budget. The deployed configuration talks to Groq's
+ * free tier at 8,000 tokens per minute, and a single accumulating conversation
+ * carries every tool result it has ever seen into every subsequent request. Six
+ * sections of gathered evidence in one context window is a request that the
+ * endpoint refuses outright — the same failure documented at length in
+ * `lib/llm/provider.ts`, arriving faster. Sections that gather and draft
+ * independently keep each request small: evidence for section four is never in
+ * the request that writes section five.
+ *
+ * The second is honesty. A loop that chooses its own next move can be logged
+ * but not planned, so the reader watching it has no idea what is left. A
+ * declared section list can be shown as a checklist before any of it exists,
+ * which is what makes the trace panel a progress indicator rather than a
+ * scroll of noise.
+ *
+ * The third is verification, and it is the one that matters. Claims can only be
+ * checked against the evidence that was actually retrieved for them. In one
+ * long turn every claim is nominally checkable against everything, which in
+ * practice means a figure gathered for the funding section can "support" a
+ * sentence in the needs section. Per-section evidence makes the check strict:
+ * this section's claims, against this section's sources, or it gets flagged.
+ *
+ * # What a run emits
+ *
+ * A `TraceEvent` stream, consumed by the route and forwarded to the client as
+ * data parts. The generator never throws for anything a source did: a tool that
+ * fails degrades its section and lands in the caveats. It throws only if the
+ * model endpoint itself is unusable, and even then the last event out is a
+ * `workflow-error` rather than a broken stream.
+ */
+
+import {
+ generateText,
+ stepCountIs,
+ streamText,
+ type LanguageModel,
+ type ToolSet,
+} from 'ai';
+
+import { getDeliverablesBudget, getDeliverablesModel, getProviderOptions } from '@/lib/llm/provider';
+import { humaniseUpstreamError } from '@/lib/llm/rate-limit';
+import { haiTools } from '@/lib/tools';
+
+import {
+ formatEvidence,
+ harvest,
+ summariseArgs,
+ summariseResult,
+ type EvidenceItem,
+ type SourceError,
+} from './evidence';
+import {
+ TokenPacer,
+ estimateToolLoopTokens,
+ estimateTokens,
+ isDailyLimit,
+ sleep,
+ withRateLimitRetry,
+} from './pacer';
+import { renderSection, renderSourcesAndCaveats } from './render';
+import type { SectionSpec, TraceEvent, WorkflowDefinition } from './types';
+import { VERIFY_EVIDENCE_CHARS, verifySection } from './verify';
+
+/* ------------------------------------------------------------------ *
+ * Step-scoped prompts
+ * ------------------------------------------------------------------ */
+
+/**
+ * The policy every step carries, and all it carries.
+ *
+ * Note what is absent: this is not `SYSTEM_PROMPT`. The chat system prompt is
+ * some 900 words of corpus description, principles, style, and refusal policy,
+ * and it earns that length in a conversation where the model must decide for
+ * itself what kind of question it has been asked. A workflow step has already
+ * been told what to do by its own brief. Re-sending the full policy on all
+ * eighteen calls of a situation brief would spend the better part of a minute's
+ * token budget restating things the step cannot act on — so each step gets the
+ * two rules it can violate (invent evidence, leak personal data) and its own
+ * task, and nothing else.
+ */
+const STEP_POLICY = `You are a component of HAI, a humanitarian operations assistant. You are executing one step of a document-generation workflow, not holding a conversation.
+
+Absolute rules:
+- Use ONLY the evidence supplied to you. Never add a figure, date, standard, section number, or organisation name that is not in it. If the evidence does not cover something the section needs, write one sentence saying so and move on. A named gap is correct output; a plausible invented figure is the worst possible output.
+- Never write identifiable personal data about affected people — no names, no case or registration numbers, no household-level detail, no precise individual locations. Report in aggregate only.
+- Do not narrate yourself. No "let me", no "I will now", no apologies, no meta-commentary about the workflow.`;
+
+/**
+ * Max tool-calling steps allowed inside one gather.
+ *
+ * Two, lowered from three after the first live Sudan brief, and the reason is
+ * the token budget rather than the search quality. A gather's cost is dominated
+ * by results accumulating across its own steps (see `estimateToolLoopTokens`),
+ * so a third step costs roughly what the first two did together — and at three
+ * steps one section's gather reserves over 80% of a minute's ceiling, leaving
+ * the draft and verify calls that follow it to wait out most of the next
+ * minute.
+ *
+ * What that third step actually bought on the live run was mostly re-queries:
+ * `crisis_updates` asked about Sudan three times with different words and got
+ * the same empty answer. The useful fan-out — population, needs, and food
+ * security together — happened inside a single step, because the model issues
+ * several calls at once when the provider allows it. Two steps keep that and
+ * drop the retry.
+ */
+const GATHER_STEP_CAP = 2;
+/** Guard on section length — a brief is scanned under time pressure. */
+const DRAFT_MAX_WORDS = 200;
+
+/**
+ * Explicit output ceilings on every call, and why omitting them was a bug
+ * rather than a default.
+ *
+ * The endpoint admits a request against `prompt + reserved completion`, not
+ * against what the completion turns out to be — its refusals say so in as many
+ * words: "Limit 8000, Requested 16395". When `maxOutputTokens` is left unset the
+ * reservation becomes the model's own maximum, which for the deployed models is
+ * 16,384 or 65,536 tokens: two to eight times the entire per-minute ceiling. A
+ * gather step with a 2,000-token prompt was therefore being refused for
+ * requesting 18,000, at a moment when the endpoint's own headers reported
+ * nearly 7,000 tokens free — a 429 that no amount of pacing could have avoided,
+ * because the number being rejected was never the number we were pacing.
+ *
+ * So every call states its ceiling. Sizing them, however, is not simply a
+ * matter of how much prose the step should produce, and getting that wrong
+ * broke a run in a much quieter way than a 429 does.
+ *
+ * A reasoning model spends output tokens thinking before it writes anything,
+ * and those tokens come out of this same ceiling: a draft call measured against
+ * `openai/gpt-oss-120b` used 133 reasoning tokens before its first visible
+ * character. Cap the output at what the prose needs and a long prompt produces
+ * a completion that is entirely reasoning, truncated at the limit, with no text
+ * at all — and the step reports success, because nothing failed. On the first
+ * run with caps set to 600 that happened to four of five sections, and the only
+ * one that survived was the one with the shortest prompt.
+ *
+ * So these are sized as reasoning headroom plus the visible answer, and stay far
+ * enough under the per-minute ceiling that prompt plus reservation still fits
+ * inside it with room for the accumulating tool results.
+ */
+const GATHER_MAX_OUTPUT = 1_200;
+const DRAFT_MAX_OUTPUT = 1_600;
+/**
+ * The plan step returns one `Name|ISO3` line — but it is a model call like any
+ * other, and at 24 tokens a reasoning model spent every one of them thinking
+ * and returned nothing. The subject then failed to resolve to an ISO3 code and
+ * every country-scoped tool in the run lost its most useful argument.
+ */
+const RESOLVE_MAX_OUTPUT = 400;
+
+/**
+ * Raised when the endpoint's per-day ceiling is shut.
+ *
+ * Distinct from any other failure because the only useful response is to stop:
+ * the sections already written stand, the reader is told plainly that the day's
+ * free budget is spent rather than watching a counter, and nothing else is
+ * attempted. Caught by `runWorkflow`'s outer handler like anything else, so the
+ * run still ends on a `workflow-error` rather than a broken stream.
+ */
+class DailyBudgetExhausted extends Error {
+ constructor(readonly waitMs: number) {
+ super(
+ 'the endpoint\'s free daily token budget for this model is used up; the run stopped rather than waiting for it to reset',
+ );
+ this.name = 'DailyBudgetExhausted';
+ }
+}
+
+/**
+ * Announce a pacing wait, serve it, and say when it is over.
+ *
+ * Hoisted out of the individual steps and into the generator on purpose. The
+ * pacer can only sleep; it cannot emit, because it is not the thing the route
+ * is iterating. Leaving the wait inside `reserve` is what made the trace panel
+ * stop dead for forty seconds at a time with nothing said — and a progress view
+ * that goes silent during the longest part of the run is worse than no progress
+ * view, because the reader's only available conclusion is that it broke.
+ */
+async function* awaitBudget(
+ pacer: TokenPacer,
+ tokens: number,
+ now: () => number,
+ stepId?: string,
+): AsyncGenerator {
+ const daily = pacer.dailyBlockMs();
+ if (daily > 0) {
+ yield { type: 'budget-wait', at: now(), waitMs: daily, scope: 'tokens-per-day', stepId };
+ throw new DailyBudgetExhausted(daily);
+ }
+
+ const wait = pacer.waitFor(tokens);
+ if (wait <= 0) return;
+
+ yield { type: 'budget-wait', at: now(), waitMs: wait, scope: 'tokens-per-minute', stepId };
+ await sleep(wait);
+ yield { type: 'budget-resumed', at: now(), stepId };
+}
+
+/* ------------------------------------------------------------------ *
+ * Engine
+ * ------------------------------------------------------------------ */
+
+export interface WorkflowRunInput {
+ workflow: WorkflowDefinition;
+ /** The user's country or topic line. Already PII-screened by the route. */
+ subject: string;
+ signal?: AbortSignal;
+ /**
+ * Epoch ms after which no further section may be started.
+ *
+ * The serverless function this runs inside has a hard lifetime, and a run
+ * killed at that boundary takes the stream with it: the document stops
+ * mid-sentence with no caveat and no way for the reader to tell an empty
+ * section from an unattempted one. Given a deadline the engine stops one
+ * section early and says so, which is the same amount of document and a great
+ * deal more information. Omitted by tests and by any caller without a
+ * lifetime to respect.
+ */
+ deadline?: number;
+}
+
+export interface EngineDeps {
+ model?: LanguageModel;
+ /**
+ * The tool registry. Defaults to `haiTools` and is injected only by tests —
+ * templates name tools by string and the engine resolves them here, so a tool
+ * added to the registry becomes available to every template that names it
+ * without this file changing.
+ */
+ tools?: ToolSet;
+ pacer?: TokenPacer;
+ now?: () => number;
+}
+
+interface RunState {
+ subject: string;
+ iso3?: string;
+ evidence: EvidenceItem[];
+ sourceErrors: SourceError[];
+ sectionMarkdown: Map;
+ flagged: number;
+ evidenceCounter: number;
+}
+
+export async function* runWorkflow(
+ input: WorkflowRunInput,
+ deps: EngineDeps = {},
+): AsyncGenerator {
+ const model = deps.model ?? getDeliverablesModel();
+ const registry = deps.tools ?? (haiTools as unknown as ToolSet);
+ const pacer = deps.pacer ?? new TokenPacer(getDeliverablesBudget());
+ const now = deps.now ?? Date.now;
+ const { workflow, subject, signal, deadline } = input;
+
+ const state: RunState = {
+ subject: subject.trim(),
+ evidence: [],
+ sourceErrors: [],
+ sectionMarkdown: new Map(),
+ flagged: 0,
+ evidenceCounter: 0,
+ };
+
+ const nextEvidenceId = () => `e${++state.evidenceCounter}`;
+ /** Set once the run runs out of time, so the caveat is recorded only once. */
+ let outOfTime = false;
+
+ try {
+ /* -------------------------------------------------------------- *
+ * Plan
+ * -------------------------------------------------------------- */
+ const planStepId = 'plan';
+ yield {
+ type: 'step-started',
+ at: now(),
+ stepId: planStepId,
+ kind: 'plan',
+ label: 'Resolve subject and plan sections',
+ };
+
+ if (workflow.subjectKind === 'country') {
+ const resolved = await resolveCountry(state.subject, { model, pacer, signal });
+ state.subject = resolved.display;
+ state.iso3 = resolved.iso3;
+ if (!resolved.iso3) {
+ // Not fatal. The country-scoped tools take an ISO3 and the gather steps
+ // pass the name through as well, so the model can still resolve it at
+ // call time — but the reader should know the run started uncertain.
+ state.sourceErrors.push({
+ source: 'subject',
+ message: `Could not resolve "${state.subject}" to an ISO 3166-1 alpha-3 country code; country-scoped datasets may return nothing.`,
+ });
+ yield {
+ type: 'source-error',
+ at: now(),
+ source: 'subject',
+ message: `Could not resolve "${state.subject}" to an ISO3 country code.`,
+ };
+ }
+ }
+
+ yield {
+ type: 'plan-created',
+ at: now(),
+ workflowId: workflow.id,
+ subject: state.subject,
+ iso3: state.iso3,
+ sections: workflow.sections.map((section) => ({
+ id: section.id,
+ heading: section.heading,
+ })),
+ };
+ yield { type: 'step-finished', at: now(), stepId: planStepId, ok: true };
+
+ /* -------------------------------------------------------------- *
+ * Sections
+ * -------------------------------------------------------------- */
+ for (const section of workflow.sections) {
+ throwIfAborted(signal);
+
+ // Checked per section rather than per call: a section is the smallest
+ // unit that is worth anything on its own, and abandoning one halfway
+ // leaves a heading with a truncated paragraph under it.
+ //
+ // Skipped rather than broken out of, so that the synthesised sources and
+ // caveats block still assembles. A truncated run is precisely the run
+ // whose reader most needs it: it is the only part of the document that
+ // says which sections are missing and why.
+ if (deadline !== undefined && now() >= deadline && !section.synthesised) {
+ if (!outOfTime) {
+ outOfTime = true;
+ const message =
+ 'the run reached the time limit for a single request; the sections after this one were not attempted';
+ state.sourceErrors.push({ source: 'run', message });
+ yield { type: 'source-error', at: now(), source: 'run', message };
+ }
+ continue;
+ }
+
+ if (section.synthesised === 'sources-and-caveats') {
+ // Assembled from the run's own bookkeeping rather than from the model:
+ // which sources answered, which failed, and when this was generated.
+ // Writing it with an LLM would be both a waste of budget and a way to
+ // get the error list paraphrased into something less true.
+ const markdown = renderSourcesAndCaveats({
+ evidence: state.evidence,
+ errors: state.sourceErrors,
+ generatedAt: new Date(now()).toISOString(),
+ flagged: state.flagged,
+ });
+ state.sectionMarkdown.set(section.id, markdown);
+ yield {
+ type: 'step-started',
+ at: now(),
+ stepId: `${section.id}:assemble`,
+ kind: 'draft',
+ label: `Assemble ${section.heading}`,
+ sectionId: section.id,
+ };
+ yield { type: 'draft-delta', at: now(), sectionId: section.id, delta: markdown };
+ yield {
+ type: 'draft-section',
+ at: now(),
+ sectionId: section.id,
+ heading: section.heading,
+ markdown,
+ };
+ yield {
+ type: 'step-finished',
+ at: now(),
+ stepId: `${section.id}:assemble`,
+ ok: true,
+ };
+ continue;
+ }
+
+ /* ---- gather ---- */
+ const gatherId = `${section.id}:gather`;
+ yield {
+ type: 'step-started',
+ at: now(),
+ stepId: gatherId,
+ kind: 'gather',
+ label: `Gather evidence — ${section.heading}`,
+ sectionId: section.id,
+ };
+
+ const sectionEvidence: EvidenceItem[] = [];
+ const activeTools = selectTools(registry, section.tools);
+
+ if (activeTools.names.length === 0) {
+ const message = `No registered tool matches ${section.tools.join(', ')}.`;
+ state.sourceErrors.push({ source: section.id, message });
+ yield { type: 'source-error', at: now(), source: section.id, message };
+ yield { type: 'step-finished', at: now(), stepId: gatherId, ok: false, note: message };
+ } else {
+ let gatherNote: string | undefined;
+ let calls = 0;
+ const gatherPromptText = gatherPrompt(workflow, section, state);
+
+ try {
+ yield* awaitBudget(pacer, gatherCost(gatherPromptText, activeTools.names.length), now, gatherId);
+ for await (const event of gather({
+ model,
+ pacer,
+ signal,
+ stepId: gatherId,
+ tools: activeTools.set,
+ prompt: gatherPromptText,
+ now,
+ onHarvest: (tool, output) => {
+ const result = harvest(tool, output, nextEvidenceId);
+ sectionEvidence.push(...result.items);
+ state.evidence.push(...result.items);
+ state.sourceErrors.push(...result.errors);
+ return result;
+ },
+ })) {
+ if (event.type === 'tool-called') calls += 1;
+ yield event;
+ }
+ } catch (error) {
+ // A gather that fails outright is one degraded section, not a dead
+ // run — the draft step below will write what the absence of evidence
+ // permits, which is a sentence saying the sources were unreachable.
+ //
+ // Unless the endpoint has closed for the day, which is not a property
+ // of this section and will not be different for the next one. Letting
+ // that degrade section by section would produce a document whose every
+ // section blamed its own sources for a single account-level fact.
+ if (isDailyLimit(error)) throw new DailyBudgetExhausted(0);
+ gatherNote = errorMessage(error);
+ state.sourceErrors.push({ source: section.id, message: gatherNote });
+ yield { type: 'source-error', at: now(), source: section.id, message: gatherNote };
+ }
+
+ if (calls === 0 && !gatherNote) {
+ gatherNote = 'the model answered without consulting a source';
+ }
+ yield {
+ type: 'step-finished',
+ at: now(),
+ stepId: gatherId,
+ ok: sectionEvidence.length > 0,
+ note: sectionEvidence.length > 0 ? undefined : gatherNote,
+ };
+ }
+
+ /* ---- draft ---- */
+ throwIfAborted(signal);
+ const draftId = `${section.id}:draft`;
+ yield {
+ type: 'step-started',
+ at: now(),
+ stepId: draftId,
+ kind: 'draft',
+ label: `Draft ${section.heading}`,
+ sectionId: section.id,
+ };
+
+ let raw = '';
+ let draftFailed = false;
+ const draftPromptText = draftPrompt(workflow, section, state, sectionEvidence);
+ try {
+ yield* awaitBudget(pacer, draftCost(draftPromptText), now, draftId);
+ for await (const delta of draft({
+ model,
+ pacer,
+ signal,
+ prompt: draftPromptText,
+ })) {
+ raw += delta;
+ yield { type: 'draft-delta', at: now(), sectionId: section.id, delta };
+ }
+ } catch (error) {
+ if (isDailyLimit(error)) throw new DailyBudgetExhausted(0);
+ const note = errorMessage(error);
+ draftFailed = true;
+ raw = `_This section could not be drafted: ${note}_`;
+ state.sourceErrors.push({ source: section.id, message: `draft failed — ${note}` });
+ yield { type: 'step-finished', at: now(), stepId: draftId, ok: false, note };
+ }
+
+ // A call that succeeded and returned no prose. Rare, and it used to pass
+ // in total silence: the section was written as an empty string, the step
+ // emitted no `step-finished` at all, and the finished document simply had
+ // a heading with nothing under it and nothing in the caveats to say why.
+ // The cause is above — an output ceiling consumed entirely by a reasoning
+ // model's hidden tokens — but any future cause deserves the same
+ // treatment, because a silently missing section is the one failure mode a
+ // document like this must never have.
+ if (!draftFailed && !raw.trim()) {
+ draftFailed = true;
+ const note = 'the model returned an empty completion for this section';
+ raw = `_This section could not be drafted: ${note}._`;
+ state.sourceErrors.push({ source: section.id, message: `draft failed — ${note}` });
+ yield { type: 'step-finished', at: now(), stepId: draftId, ok: false, note };
+ }
+
+ // Citation ids become source labels here, and an id the model invented
+ // becomes a visible marker rather than a dangling `[e9]`.
+ const rendered = renderSection(raw, sectionEvidence);
+ state.flagged += rendered.invented;
+ state.sectionMarkdown.set(section.id, rendered.markdown);
+
+ yield {
+ type: 'draft-section',
+ at: now(),
+ sectionId: section.id,
+ heading: section.heading,
+ markdown: rendered.markdown,
+ };
+ if (!draftFailed && raw) {
+ yield { type: 'step-finished', at: now(), stepId: draftId, ok: true };
+ }
+
+ /* ---- verify ---- */
+ // A section that failed to draft holds an error message, not prose.
+ // Verifying it produced the genuinely absurd output seen on a live run:
+ // the checker read "This section could not be drafted: Failed after 3
+ // attempts…" as a factual claim, found no evidence for it, and marked the
+ // failure notice itself **[unverified]**.
+ if (draftFailed || section.skipVerification || !rendered.markdown.trim()) continue;
+
+ throwIfAborted(signal);
+ const verifyId = `${section.id}:verify`;
+ yield {
+ type: 'step-started',
+ at: now(),
+ stepId: verifyId,
+ kind: 'verify',
+ label: `Check claims — ${section.heading}`,
+ sectionId: section.id,
+ };
+
+ try {
+ yield* awaitBudget(pacer, verifyCost(rendered.markdown, sectionEvidence), now, verifyId);
+ const checked = await verifySection({
+ sectionId: section.id,
+ markdown: rendered.markdown,
+ evidence: sectionEvidence,
+ model,
+ pacer,
+ signal,
+ });
+
+ for (const check of checked.checks) {
+ yield {
+ type: 'check-run',
+ at: now(),
+ sectionId: section.id,
+ claim: check.claim,
+ verdict: check.verdict,
+ source: check.source,
+ };
+ }
+
+ state.flagged += checked.flagged;
+ state.sectionMarkdown.set(section.id, checked.markdown);
+ yield {
+ type: 'section-verified',
+ at: now(),
+ sectionId: section.id,
+ markdown: checked.markdown,
+ flagged: checked.flagged,
+ };
+ yield { type: 'step-finished', at: now(), stepId: verifyId, ok: true };
+ } catch (error) {
+ // Verification failing is itself a caveat the reader must see: the
+ // section stands as drafted, unchecked, and says so.
+ const note = errorMessage(error);
+ state.sourceErrors.push({
+ source: `${section.id} verification`,
+ message: `claims in this section were not checked — ${note}`,
+ });
+ yield { type: 'step-finished', at: now(), stepId: verifyId, ok: false, note };
+ }
+ }
+
+ yield {
+ type: 'workflow-done',
+ at: now(),
+ generatedAt: new Date(now()).toISOString(),
+ sections: state.sectionMarkdown.size,
+ flagged: state.flagged,
+ };
+ } catch (error) {
+ yield { type: 'workflow-error', at: now(), message: errorMessage(error) };
+ }
+}
+
+/* ------------------------------------------------------------------ *
+ * Steps
+ * ------------------------------------------------------------------ */
+
+/**
+ * Resolve a free-text country line to a display name and an ISO3 code.
+ *
+ * A structured-output call would be the tidy way to do this, but structured
+ * output over an OpenAI-compatible endpoint is the least portable thing the SDK
+ * offers — it degrades differently on Ollama and on Groq, and this runs on
+ * both. A twelve-token completion parsed with a regex works identically
+ * everywhere and fails visibly when it fails.
+ */
+async function resolveCountry(
+ subject: string,
+ ctx: { model: LanguageModel; pacer: TokenPacer; signal?: AbortSignal },
+): Promise<{ display: string; iso3?: string }> {
+ const prompt = `Country or territory: "${subject}"
+
+Reply with exactly one line, nothing else:
+|
+
+If it is not a country or territory, or you are not certain of the code, reply:
+${subject}|NONE`;
+
+ const estimated = estimateTokens(prompt, RESOLVE_MAX_OUTPUT);
+ await ctx.pacer.reserve(estimated);
+
+ try {
+ const result = await withRateLimitRetry(() =>
+ generateText({
+ model: ctx.model,
+ providerOptions: getProviderOptions(),
+ prompt,
+ temperature: 0,
+ maxOutputTokens: RESOLVE_MAX_OUTPUT,
+ abortSignal: ctx.signal,
+ }),
+ );
+ ctx.pacer.settle(estimated, result.usage?.totalTokens ?? estimated);
+
+ const line = result.text.trim().split('\n')[0] ?? '';
+ const [name, code] = line.split('|').map((part) => part.trim());
+ const iso3 = code && /^[A-Z]{3}$/.test(code.toUpperCase()) && code.toUpperCase() !== 'NON'
+ ? code.toUpperCase()
+ : undefined;
+ return { display: name || subject, iso3 };
+ } catch {
+ // The subject line as typed is a perfectly usable document title, and the
+ // gather steps carry the country name to the tools regardless.
+ return { display: subject };
+ }
+}
+
+interface GatherContext {
+ model: LanguageModel;
+ pacer: TokenPacer;
+ signal?: AbortSignal;
+ stepId: string;
+ tools: ToolSet;
+ prompt: string;
+ now: () => number;
+ onHarvest: (tool: string, output: unknown) => { items: EvidenceItem[]; errors: SourceError[] };
+}
+
+/**
+ * One section's evidence gathering: a short tool-calling loop whose prose
+ * output is discarded. Only what the tools returned survives, which is the
+ * point — a gather step that could contribute text would contribute text it
+ * had not retrieved.
+ */
+async function* gather(ctx: GatherContext): AsyncGenerator {
+ // Reserved for the whole loop up front, accumulation included — see
+ // `estimateToolLoopTokens`. Reserving per call instead would let the pacer
+ // wave three cheap-looking requests through inside one minute and then watch
+ // the endpoint refuse the third.
+ const estimated =
+ estimateTokens(`${STEP_POLICY}${ctx.prompt}`, 0) +
+ estimateToolLoopTokens(Object.keys(ctx.tools).length, GATHER_STEP_CAP);
+ await ctx.pacer.reserve(estimated);
+
+ const run = () =>
+ streamText({
+ model: ctx.model,
+ providerOptions: getProviderOptions(),
+ system: STEP_POLICY,
+ prompt: ctx.prompt,
+ tools: ctx.tools,
+ temperature: 0,
+ maxOutputTokens: GATHER_MAX_OUTPUT,
+ // One call per step (see `getProviderOptions`) times this cap is the most
+ // evidence one section can pull. Three is enough to cross-check a figure
+ // against a second source and no more; a fourth mostly re-runs the first.
+ stopWhen: stepCountIs(GATHER_STEP_CAP),
+ abortSignal: ctx.signal,
+ });
+
+ let result = run();
+ let sawCall = false;
+
+ for await (const part of result.fullStream) {
+ if (part.type === 'tool-call') {
+ sawCall = true;
+ yield {
+ type: 'tool-called',
+ at: ctx.now(),
+ stepId: ctx.stepId,
+ callId: part.toolCallId,
+ tool: part.toolName,
+ args: summariseArgs(part.input),
+ };
+ } else if (part.type === 'tool-result') {
+ const harvested = ctx.onHarvest(part.toolName, part.output);
+ yield {
+ type: 'tool-result',
+ at: ctx.now(),
+ stepId: ctx.stepId,
+ callId: part.toolCallId,
+ tool: part.toolName,
+ ok: harvested.items.length > 0,
+ summary: summariseResult(harvested),
+ };
+ for (const error of harvested.errors) {
+ yield { type: 'source-error', at: ctx.now(), source: error.source, message: error.message };
+ }
+ } else if (part.type === 'tool-error') {
+ yield {
+ type: 'tool-result',
+ at: ctx.now(),
+ stepId: ctx.stepId,
+ callId: part.toolCallId,
+ tool: part.toolName,
+ ok: false,
+ summary: errorMessage(part.error),
+ };
+ } else if (part.type === 'error') {
+ throw part.error;
+ }
+ }
+
+ ctx.pacer.settle(estimated, (await result.totalUsage)?.totalTokens ?? estimated);
+
+ if (sawCall) return;
+
+ // The model decided it already knew the answer. That is the exact failure the
+ // whole grounding policy exists to prevent, so the step is run once more with
+ // the choice taken away rather than accepted as an empty gather.
+ const forcedEstimate =
+ estimateTokens(ctx.prompt, 0) + estimateToolLoopTokens(Object.keys(ctx.tools).length, 1);
+ await ctx.pacer.reserve(forcedEstimate);
+ result = streamText({
+ model: ctx.model,
+ providerOptions: getProviderOptions(),
+ system: STEP_POLICY,
+ prompt: ctx.prompt,
+ tools: ctx.tools,
+ toolChoice: 'required',
+ temperature: 0,
+ maxOutputTokens: GATHER_MAX_OUTPUT,
+ stopWhen: stepCountIs(1),
+ abortSignal: ctx.signal,
+ });
+
+ for await (const part of result.fullStream) {
+ if (part.type === 'tool-call') {
+ yield {
+ type: 'tool-called',
+ at: ctx.now(),
+ stepId: ctx.stepId,
+ callId: part.toolCallId,
+ tool: part.toolName,
+ args: summariseArgs(part.input),
+ };
+ } else if (part.type === 'tool-result') {
+ const harvested = ctx.onHarvest(part.toolName, part.output);
+ yield {
+ type: 'tool-result',
+ at: ctx.now(),
+ stepId: ctx.stepId,
+ callId: part.toolCallId,
+ tool: part.toolName,
+ ok: harvested.items.length > 0,
+ summary: summariseResult(harvested),
+ };
+ for (const error of harvested.errors) {
+ yield { type: 'source-error', at: ctx.now(), source: error.source, message: error.message };
+ }
+ } else if (part.type === 'error') {
+ throw part.error;
+ }
+ }
+ ctx.pacer.settle(forcedEstimate, (await result.totalUsage)?.totalTokens ?? forcedEstimate);
+}
+
+/** Write one section from its own evidence. No tools, by design. */
+async function* draft(ctx: {
+ model: LanguageModel;
+ pacer: TokenPacer;
+ signal?: AbortSignal;
+ prompt: string;
+}): AsyncGenerator {
+ const estimated = estimateTokens(`${STEP_POLICY}${ctx.prompt}`, DRAFT_MAX_WORDS * 2);
+ await ctx.pacer.reserve(estimated);
+
+ // Not wrapped in `withRateLimitRetry`, deliberately. `streamText` returns
+ // synchronously and reports failures as an `error` part in the stream, so a
+ // retry around the call can never fire — it would be a comforting no-op. The
+ // SDK's own `maxRetries` already backs off inside the call, the pacer is what
+ // stops a rate limit being reached in the first place, and a draft that still
+ // fails degrades to a named caveat, which is the honest outcome.
+ const result = streamText({
+ model: ctx.model,
+ providerOptions: getProviderOptions(),
+ system: STEP_POLICY,
+ prompt: ctx.prompt,
+ temperature: 0,
+ maxOutputTokens: DRAFT_MAX_OUTPUT,
+ abortSignal: ctx.signal,
+ });
+
+ for await (const part of result.fullStream) {
+ if (part.type === 'text-delta') yield part.text;
+ else if (part.type === 'error') throw part.error;
+ }
+ ctx.pacer.settle(estimated, (await result.totalUsage)?.totalTokens ?? estimated);
+}
+
+/* ------------------------------------------------------------------ *
+ * Prompts
+ * ------------------------------------------------------------------ */
+
+function subjectLine(workflow: WorkflowDefinition, state: RunState): string {
+ if (workflow.subjectKind !== 'country') return `Programme/topic: ${state.subject}`;
+ return state.iso3
+ ? `Country: ${state.subject} (ISO3 ${state.iso3})`
+ : `Country: ${state.subject} (ISO3 code unknown — supply it yourself if you are certain of it)`;
+}
+
+function gatherPrompt(
+ workflow: WorkflowDefinition,
+ section: SectionSpec,
+ state: RunState,
+): string {
+ return `${subjectLine(workflow, state)}
+Document: ${workflow.title}
+Section: ${section.heading}
+
+Retrieve the evidence this section needs, then stop. ${section.gatherBrief}
+
+Call the tools. Do not write the section, do not summarise what you found, and do not answer from your own knowledge — another step writes the prose from whatever you retrieve. At most ${GATHER_STEP_CAP} tool calls; make each one different from the last.`;
+}
+
+function draftPrompt(
+ workflow: WorkflowDefinition,
+ section: SectionSpec,
+ state: RunState,
+ evidence: EvidenceItem[],
+): string {
+ return `${subjectLine(workflow, state)}
+Document: ${workflow.title}
+Section to write: ${section.heading}
+
+${section.brief}
+
+EVIDENCE (the only facts you may use):
+${formatEvidence(evidence)}
+
+Write the section body in markdown. No heading — the heading is added for you.
+Under ${DRAFT_MAX_WORDS} words. Prefer a short list or a compact table over paragraphs.
+
+Cite every figure and every substantive claim with the bracketed evidence id it came from, like [e2], placed at the end of the sentence. A sentence with a figure and no id will be rejected. Give the reference period for every figure — it is in the evidence.
+If the evidence does not cover part of this section, write one sentence naming what is missing instead of filling the gap.`;
+}
+
+/* ------------------------------------------------------------------ *
+ * Cost projection
+ * ------------------------------------------------------------------ */
+
+/*
+ * What each step is about to cost, projected before it runs.
+ *
+ * These mirror the reservations the steps themselves make, and exist as
+ * separate functions for one reason: the wait has to be announced by the
+ * generator, and the generator therefore has to know the number before the step
+ * it belongs to is entered. Kept beside the prompt builders so that a prompt
+ * that grows and a projection that does not stay visibly out of step.
+ *
+ * A projection being somewhat wrong is now cheap. It decides only how long to
+ * pause before asking; what the call actually spent comes back from the
+ * endpoint's own headers a moment later — see `lib/llm/rate-limit.ts`.
+ */
+
+export function gatherCost(prompt: string, toolCount: number): number {
+ return (
+ estimateTokens(`${STEP_POLICY}${prompt}`, GATHER_MAX_OUTPUT) +
+ estimateToolLoopTokens(toolCount, GATHER_STEP_CAP)
+ );
+}
+
+export function draftCost(prompt: string): number {
+ return estimateTokens(`${STEP_POLICY}${prompt}`, DRAFT_MAX_OUTPUT);
+}
+
+export function verifyCost(markdown: string, evidence: EvidenceItem[]): number {
+ return estimateTokens(`${markdown}${formatEvidence(evidence, VERIFY_EVIDENCE_CHARS)}`, 300);
+}
+
+/* ------------------------------------------------------------------ *
+ * Helpers
+ * ------------------------------------------------------------------ */
+
+/**
+ * Resolve a template's tool names against the live registry.
+ *
+ * Names that are not registered are dropped rather than raising. That is what
+ * lets a template be written against a tool before it exists and lets a tool be
+ * retired without breaking every template that mentioned it — the section
+ * degrades to whatever else it named, and the trace says which sources answered.
+ */
+export function selectTools(
+ registry: ToolSet,
+ names: readonly string[],
+): { set: ToolSet; names: string[] } {
+ const set: ToolSet = {};
+ const resolved: string[] = [];
+ for (const name of names) {
+ if (Object.prototype.hasOwnProperty.call(registry, name)) {
+ set[name] = registry[name];
+ resolved.push(name);
+ }
+ }
+ return { set, names: resolved };
+}
+
+function throwIfAborted(signal?: AbortSignal): void {
+ if (signal?.aborted) throw new Error('The run was cancelled.');
+}
+
+/**
+ * An error as the document should describe it.
+ *
+ * Upstream messages are written for whoever is paying the bill, not for whoever
+ * is reading the brief. Groq's rate-limit text ends "Need more tokens? Upgrade
+ * to Dev Tier today at https://console.groq.com/settings/billing", and on a
+ * live run that sentence was rendered verbatim into a humanitarian situation
+ * brief's caveats section — a vendor upsell inside a document about a caseload
+ * of 33 million people. The reader needs the fact (a section is missing because
+ * the run hit the endpoint's per-minute budget), not the sales copy.
+ */
+function errorMessage(error: unknown): string {
+ const raw =
+ error instanceof Error ? error.message : typeof error === 'string' ? error : 'unknown error';
+
+ // Shared with the chat route, which had the same problem and no such guard:
+ // it names which ceiling was hit and strips the vendor's link. See
+ // `humaniseUpstreamError`.
+ return humaniseUpstreamError(raw);
+}
diff --git a/app/src/lib/agent/evidence.test.ts b/app/src/lib/agent/evidence.test.ts
new file mode 100644
index 0000000..9ca537b
--- /dev/null
+++ b/app/src/lib/agent/evidence.test.ts
@@ -0,0 +1,247 @@
+import { describe, expect, it } from 'vitest';
+
+import { formatEvidence, harvest, summariseArgs, summariseResult } from './evidence';
+
+function counter() {
+ let n = 0;
+ return () => `e${++n}`;
+}
+
+describe('harvest', () => {
+ it('reads retrieved passages into labelled, citable evidence', () => {
+ const result = harvest(
+ 'search_standards',
+ {
+ query: 'water quantity',
+ chunks: [
+ {
+ source: 'sphere',
+ section: 'Water supply standard 2.1',
+ text: 'The minimum water quantity is 15 litres per person per day.',
+ },
+ ],
+ },
+ counter(),
+ );
+
+ expect(result.errors).toEqual([]);
+ expect(result.items).toHaveLength(1);
+ expect(result.items[0]).toMatchObject({
+ id: 'e1',
+ tool: 'search_standards',
+ label: 'sphere · Water supply standard 2.1',
+ });
+ expect(result.items[0].text).toContain('15 litres');
+ });
+
+ /*
+ * The registry is under active extension, so the harvester probes shapes
+ * rather than switching on tool name. This is the guard on that: a tool the
+ * engine has never heard of, returning fields it has never seen, still
+ * produces usable evidence rather than being dropped.
+ */
+ it('reads a tool it has no knowledge of', () => {
+ const result = harvest(
+ 'some_future_tool',
+ {
+ readings: [
+ { title: 'River gauge', date: '2026-08-01', level_m: 7.4, station: 'Khartoum' },
+ ],
+ },
+ counter(),
+ );
+
+ expect(result.items).toHaveLength(1);
+ expect(result.items[0].label).toBe('River gauge · 2026-08-01');
+ expect(result.items[0].text).toContain('level_m: 7.4');
+ });
+
+ it('records a per-source errors list as degradation, not as evidence', () => {
+ const result = harvest(
+ 'hazards_context',
+ {
+ scope: 'country',
+ gdacsAlerts: [
+ { source: 'GDACS', eventType: 'Flood', alertLevel: 'orange', title: 'Flood in Sudan' },
+ ],
+ countryContext: [],
+ errors: ['worldbank: HTTP 503', 'hpc: no plan published for 2026'],
+ },
+ counter(),
+ );
+
+ expect(result.errors).toEqual([
+ { source: 'hazards_context · worldbank', message: 'HTTP 503' },
+ { source: 'hazards_context · hpc', message: 'no plan published for 2026' },
+ ]);
+ // The failures must not become citable facts; the alert that did arrive must.
+ expect(result.items).toHaveLength(1);
+ expect(result.items[0].text).toContain('Flood in Sudan');
+ expect(result.items.some((item) => item.text.includes('HTTP 503'))).toBe(false);
+ });
+
+ /*
+ * The real shapes from `tools/live-sources/*`, which map their upstream feeds
+ * into camelCase TypeScript interfaces rather than the snake_case the
+ * HDX-backed tools return. Before the qualifier list covered both, every GDACS
+ * alert in a brief labelled as the bare word "GDACS", so three different
+ * floods cited identically and the reader could not tell them apart.
+ */
+ it('labels a GDACS alert by its hazard, not just its publisher', () => {
+ const result = harvest(
+ 'hazards_context',
+ {
+ scope: 'country',
+ gdacsAlerts: [
+ {
+ source: 'GDACS',
+ eventType: 'Flood',
+ alertLevel: 'orange',
+ title: 'Flood in Sudan',
+ country: 'Sudan',
+ fromDate: '2026-08-20',
+ },
+ ],
+ errors: [],
+ },
+ counter(),
+ );
+
+ expect(result.items[0].label).toBe('GDACS · Flood');
+ expect(result.items[0].text).toContain('orange');
+ });
+
+ it('labels an OCHA response plan by its name', () => {
+ const result = harvest(
+ 'hazards_context',
+ {
+ scope: 'global',
+ responsePlans: [
+ { source: 'OCHA HPC', id: 1234, name: 'Sudan Humanitarian Response Plan 2026', code: 'HSDN26' },
+ ],
+ errors: [],
+ },
+ counter(),
+ );
+
+ expect(result.items[0].label).toBe('OCHA HPC · Sudan Humanitarian Response Plan 2026');
+ // The name is spent on the label, so it is not repeated inside the body.
+ expect(result.items[0].text).not.toContain('Sudan Humanitarian Response Plan 2026');
+ });
+
+ it('labels a World Bank indicator by its reference year', () => {
+ const result = harvest(
+ 'hazards_context',
+ {
+ scope: 'country',
+ countryContext: [
+ { source: 'World Bank', indicator: 'SP.POP.TOTL', name: 'Population, total', country: 'Sudan', year: '2025', value: 51662147 },
+ ],
+ errors: [],
+ },
+ counter(),
+ );
+
+ expect(result.items[0].label).toBe('World Bank · 2025');
+ expect(result.items[0].text).toContain('51662147');
+ });
+
+ /*
+ * Seen on the live Sudan brief: `hazards_context` was asked for `hpc` at
+ * country scope, where that source does not apply, and returned an envelope
+ * with no data and no errors. The single-record fallback turned it into a
+ * citable item reading "scope: country; generatedAt: …" — a fact about the
+ * request, handed to a model under instructions to cite what it is given.
+ */
+ it('does not turn an empty result envelope into evidence', () => {
+ const result = harvest(
+ 'hazards_context',
+ {
+ scope: 'country',
+ countryIso3: 'SDN',
+ generatedAt: '2026-09-01T13:50:00.000Z',
+ errors: [],
+ },
+ counter(),
+ );
+
+ expect(result.items).toEqual([]);
+ expect(result.errors).toEqual([]);
+ });
+
+ it('treats an unavailable dataset as a source error with no evidence', () => {
+ const result = harvest(
+ 'humanitarian_data',
+ { available: false, reason: 'no_coverage', detail: 'HAPI holds no funding data for SDN.' },
+ counter(),
+ );
+
+ expect(result.items).toEqual([]);
+ expect(result.errors[0].message).toBe('HAPI holds no funding data for SDN.');
+ });
+
+ it('treats an empty-retrieval notice as a source error', () => {
+ const result = harvest(
+ 'search_standards',
+ { query: 'blockchain', chunks: [], notice: 'No passages matched this query.' },
+ counter(),
+ );
+
+ expect(result.items).toEqual([]);
+ expect(result.errors[0].message).toBe('No passages matched this query.');
+ });
+});
+
+describe('summaries', () => {
+ it('describes a result by count and source, never by content', () => {
+ const harvested = harvest(
+ 'search_standards',
+ {
+ chunks: [
+ { source: 'sphere', section: '2.1', text: 'a' },
+ { source: 'chs', section: '4', text: 'b' },
+ ],
+ },
+ counter(),
+ );
+ expect(summariseResult(harvested)).toBe('2 records — sphere, chs');
+ });
+
+ it('reports the failure when a result carried only one', () => {
+ const harvested = harvest('crisis_updates', { updates: [], error: 'feed unreachable' }, counter());
+ expect(summariseResult(harvested)).toBe('feed unreachable');
+ });
+
+ it('flattens tool arguments to one readable line', () => {
+ expect(summariseArgs({ country_iso3: 'SDN', dataset: 'funding', unset: undefined })).toBe(
+ 'country_iso3=SDN dataset=funding',
+ );
+ });
+});
+
+describe('formatEvidence', () => {
+ it('numbers evidence so a draft can cite it', () => {
+ const { items } = harvest(
+ 'search_standards',
+ { chunks: [{ source: 'sphere', section: '2.1', text: '15 litres per person per day' }] },
+ counter(),
+ );
+ expect(formatEvidence(items)).toBe(
+ '[e1] sphere · 2.1: 15 litres per person per day',
+ );
+ });
+
+ it('says so rather than going blank when nothing was retrieved', () => {
+ expect(formatEvidence([])).toContain('no evidence');
+ });
+
+ it('caps the digest so one section cannot spend the whole token budget', () => {
+ const items = Array.from({ length: 40 }, (_, index) => ({
+ id: `e${index}`,
+ tool: 't',
+ label: 'source',
+ text: 'x'.repeat(200),
+ }));
+ expect(formatEvidence(items, 1_000).length).toBeLessThanOrEqual(1_000);
+ });
+});
diff --git a/app/src/lib/agent/evidence.ts b/app/src/lib/agent/evidence.ts
new file mode 100644
index 0000000..1492dda
--- /dev/null
+++ b/app/src/lib/agent/evidence.ts
@@ -0,0 +1,364 @@
+/**
+ * Turning whatever a tool returned into evidence the rest of the engine can
+ * use, without the engine knowing which tools exist.
+ *
+ * This file deliberately does not import from `@/lib/tools`. HAI's tool
+ * registry is under active extension — live hazard feeds, country context —
+ * and a deliverable engine that switches on tool name would need editing every
+ * time one is added, which is exactly the coupling that leaves a new data
+ * source wired up but invisible in the product. So the shapes are probed
+ * structurally: find the arrays of records, name each record from whichever
+ * labelling fields it happens to carry, and serialise the rest.
+ *
+ * The other half of the job is the `situation.py` pattern this was ported
+ * from: a source that is down, unconfigured, or simply has no coverage for the
+ * country asked about must degrade the section it feeds and nothing else. Every
+ * such failure is caught here, recorded, and surfaced in the finished
+ * document's caveats — never thrown, and never silently dropped.
+ */
+
+/** One retrieved fact, carrying the label a drafted claim must cite it by. */
+export interface EvidenceItem {
+ /** Stable within a run: `e1`, `e2`… The draft step cites these. */
+ id: string;
+ /** Registry key of the tool that produced it. */
+ tool: string;
+ /** Human source label — "Sphere Handbook · WASH 2.1", "HDX HAPI · funding". */
+ label: string;
+ /** The content itself, clipped. */
+ text: string;
+}
+
+/** A source that failed or reported no coverage, for the caveats section. */
+export interface SourceError {
+ source: string;
+ message: string;
+}
+
+/**
+ * Per-item and per-result clipping.
+ *
+ * These numbers are a token budget, not a display choice. Every piece of
+ * evidence a section gathers is re-sent in that section's draft call and again
+ * in its verify call, and the hosted deployment runs against Groq's free tier
+ * at 8,000 tokens per minute (see `docs/DEPLOY.md`). Eight items of 420
+ * characters is roughly 850 tokens of evidence per section, which leaves room
+ * for the prompt and the output inside a single minute's budget. Raising either
+ * number without raising the pacer's ceiling in `pacer.ts` is how a run starts
+ * stalling on 429s two sections in.
+ */
+const MAX_ITEM_CHARS = 420;
+const MAX_ITEMS_PER_RESULT = 8;
+
+/** Fields that, when present, name a record well enough to cite it by. */
+const LABEL_FIELDS = [
+ 'source',
+ 'title',
+ 'name',
+ 'label',
+ 'indicator',
+ 'dataset',
+ 'event_type',
+ 'organization',
+ 'organisation',
+] as const;
+
+/**
+ * Fields that qualify a label — a section reference, a period, a date, or the
+ * particular thing a record is about.
+ *
+ * Both cases of each name, because the tools do not agree and should not have
+ * to. The HDX-backed tools return snake_case straight off the wire
+ * (`reference_period`); the live-source connectors in `tools/live-sources/*`
+ * map into camelCase TypeScript interfaces (`alertLevel`, `eventType`). Without
+ * the camelCase forms every GDACS alert and USGS event labels as the bare
+ * source name, so a brief citing three different floods cites all of them as
+ * "GDACS" and the reader cannot tell which.
+ */
+const QUALIFIER_FIELDS = [
+ 'section',
+ 'reference_period',
+ 'period',
+ 'date',
+ 'year',
+ // Hazard type before severity: it is the discriminator a reader needs to tell
+ // two GDACS records apart ("GDACS · Flood" against "GDACS · Tropical
+ // Cyclone"), while the severity is a property of the event and travels in the
+ // record body anyway.
+ 'event_type',
+ 'eventType',
+ 'alert_level',
+ 'alertLevel',
+ 'place',
+ 'page',
+ // Last resort: a plan or indicator name, for records whose only other
+ // labelling field is the publisher (OCHA HPC's response plans).
+ 'name',
+] as const;
+
+/**
+ * Fields whose presence means "this did not work" rather than "here is a
+ * result". `notice` is HAI's own convention for an empty retrieval (see
+ * `search_standards`); `available: false` is `humanitarian_data`'s; `error` is
+ * everyone's. Checked structurally so a tool added later gets the same
+ * treatment by following the same convention.
+ */
+function extractFailures(tool: string, value: Record): SourceError[] {
+ const found: SourceError[] = [];
+
+ if (typeof value.error === 'string' && value.error) {
+ found.push({ source: tool, message: clip(value.error, 260) });
+ }
+ if (value.available === false) {
+ const reason = typeof value.reason === 'string' ? value.reason : undefined;
+ const detail = typeof value.detail === 'string' ? value.detail : undefined;
+ found.push({ source: tool, message: clip(detail ?? reason ?? 'no data available', 260) });
+ }
+ if (typeof value.notice === 'string' && value.notice) {
+ found.push({ source: tool, message: clip(value.notice, 260) });
+ }
+
+ // A plural `errors` list is the `situation.py` convention this engine was
+ // ported from, and `hazards_context` follows it: each entry is one upstream
+ // source that degraded while the others succeeded. Reported per source rather
+ // than collapsed, because "GDACS is down" and "World Bank has no 2024 figure"
+ // are different facts to a reader deciding whether to trust the section.
+ if (Array.isArray(value.errors)) {
+ for (const entry of value.errors) {
+ if (typeof entry === 'string' && entry.trim()) {
+ // Entries are conventionally "source: what went wrong".
+ const split = /^([a-z0-9_ -]{2,24}):\s*(.+)$/i.exec(entry.trim());
+ found.push(
+ split
+ ? { source: `${tool} · ${split[1]}`, message: clip(split[2], 260) }
+ : { source: tool, message: clip(entry, 260) },
+ );
+ } else if (isRecord(entry)) {
+ const source = stringField(entry, LABEL_FIELDS)?.value ?? tool;
+ const message =
+ stringField(entry, ['message', 'error', 'detail', 'reason'])?.value ?? 'failed';
+ found.push({ source: `${tool} · ${source}`, message: clip(message, 260) });
+ }
+ }
+ }
+
+ return found;
+}
+
+export function clip(text: string, max = MAX_ITEM_CHARS): string {
+ const flat = text.replace(/\s+/g, ' ').trim();
+ if (flat.length <= max) return flat;
+ const cut = flat.slice(0, max);
+ const lastSpace = cut.lastIndexOf(' ');
+ return `${cut.slice(0, lastSpace > 0 ? lastSpace : cut.length)}…`;
+}
+
+function isRecord(value: unknown): value is Record {
+ return typeof value === 'object' && value !== null && !Array.isArray(value);
+}
+
+/**
+ * The first of `keys` this record actually carries a usable value for, with the
+ * key it came from.
+ *
+ * Returning the key matters: the caller excludes it from the record's body so
+ * the label is not repeated inside it. An earlier version guessed by re-scanning
+ * for the first key merely `!== undefined`, which picks a different field
+ * whenever an earlier one is present but null — common in the live-source
+ * shapes, where `title` and `country` are `string | null`.
+ */
+function stringField(
+ record: Record,
+ keys: readonly string[],
+): { key: string; value: string } | undefined {
+ for (const key of keys) {
+ const value = record[key];
+ if (typeof value === 'string' && value.trim()) return { key, value: value.trim() };
+ if (typeof value === 'number') return { key, value: String(value) };
+ }
+ return undefined;
+}
+
+/** Everything about a record except the parts already used as its label. */
+function bodyOf(record: Record, usedKeys: Set): string {
+ const parts: string[] = [];
+ for (const [key, value] of Object.entries(record)) {
+ if (usedKeys.has(key)) continue;
+ if (value === null || value === undefined || value === '') continue;
+ if (typeof value === 'object') {
+ parts.push(`${key}: ${clip(JSON.stringify(value), 160)}`);
+ } else {
+ parts.push(`${key}: ${String(value)}`);
+ }
+ }
+ return parts.join('; ');
+}
+
+/**
+ * A record's citable label, built from whatever naming fields it carries.
+ * `text`-bearing records (retrieved passages) label as "source · section";
+ * figure records label as "source · indicator (period)".
+ */
+function labelOf(record: Record, tool: string, used: Set): string {
+ const base = stringField(record, LABEL_FIELDS);
+ if (base) used.add(base.key);
+
+ const qualifier = stringField(record, QUALIFIER_FIELDS);
+ // A field can appear in both lists (`name`), in which case it has already
+ // been spent on the head and must not be repeated as its own qualifier.
+ if (qualifier && qualifier.key !== base?.key) {
+ used.add(qualifier.key);
+ return `${base?.value ?? tool} · ${qualifier.value}`;
+ }
+
+ return base?.value ?? tool;
+}
+
+/** The main body text of a record, if it has one. */
+function textOf(record: Record, used: Set): string | undefined {
+ for (const key of ['text', 'excerpt', 'summary', 'body', 'description'] as const) {
+ const value = record[key];
+ if (typeof value === 'string' && value.trim()) {
+ used.add(key);
+ return value;
+ }
+ }
+ return undefined;
+}
+
+/**
+ * Fields every tool result carries to describe the request rather than the
+ * answer. A record made only of these is an empty envelope.
+ */
+const ENVELOPE_FIELDS = new Set([
+ 'scope',
+ 'generatedAt',
+ 'generated_at',
+ 'query',
+ 'source',
+ 'dataset',
+ 'type',
+ 'country',
+ 'country_iso3',
+ 'countryIso3',
+ 'available',
+ 'errors',
+]);
+
+function hasSubstance(record: Record): boolean {
+ return Object.entries(record).some(
+ ([key, value]) =>
+ !ENVELOPE_FIELDS.has(key) && value !== null && value !== undefined && value !== '',
+ );
+}
+
+export interface Harvest {
+ items: EvidenceItem[];
+ errors: SourceError[];
+}
+
+/**
+ * Read one tool result into evidence items and source errors.
+ *
+ * `nextId` is passed in rather than held in module state so a run's ids are
+ * sequential across sections and two concurrent runs cannot collide.
+ */
+export function harvest(tool: string, output: unknown, nextId: () => string): Harvest {
+ const items: EvidenceItem[] = [];
+ const errors: SourceError[] = [];
+
+ if (!isRecord(output)) {
+ // A scalar or array at the top level: keep it whole rather than guessing.
+ if (output !== undefined && output !== null) {
+ items.push({ id: nextId(), tool, label: tool, text: clip(JSON.stringify(output)) });
+ }
+ return { items, errors };
+ }
+
+ errors.push(...extractFailures(tool, output));
+
+ // Arrays of records are the payload in every tool shape HAI has: `chunks`
+ // for passages, `updates` for situation reports, `figures` for indicators,
+ // `gdacsAlerts` and `countryContext` for hazards. `errors` is excluded — it
+ // was just read as degradation above, and evidence a section cites must never
+ // be a description of a source that failed.
+ const arrays = Object.entries(output).filter(
+ (entry): entry is [string, unknown[]] =>
+ entry[0] !== 'errors' && Array.isArray(entry[1]) && entry[1].length > 0,
+ );
+
+ for (const [key, array] of arrays) {
+ for (const entry of array.slice(0, MAX_ITEMS_PER_RESULT)) {
+ if (!isRecord(entry)) {
+ items.push({ id: nextId(), tool, label: `${tool} · ${key}`, text: clip(String(entry)) });
+ continue;
+ }
+ const used = new Set();
+ const label = labelOf(entry, tool, used);
+ const body = textOf(entry, used);
+ const rest = bodyOf(entry, used);
+ const text = body ? clip(rest ? `${body} (${rest})` : body) : clip(rest);
+ if (text) items.push({ id: nextId(), tool, label, text });
+ }
+ }
+
+ // No arrays and no failure: a single-record result. Keep it as one item so a
+ // tool that returns a flat object is not silently dropped — but only if the
+ // object says something. A result that carries nothing except the envelope it
+ // came in is not evidence, and on the live Sudan brief one became a citable
+ // item reading "scope: country; generatedAt: 2026-09-01T…". That is a fact
+ // about the request, offered to a model under instructions to cite what it is
+ // given.
+ if (items.length === 0 && errors.length === 0 && hasSubstance(output)) {
+ const used = new Set();
+ const label = labelOf(output, tool, used);
+ const body = textOf(output, used);
+ const rest = bodyOf(output, used);
+ const text = body ? clip(rest ? `${body} (${rest})` : body) : clip(rest);
+ if (text) items.push({ id: nextId(), tool, label, text });
+ }
+
+ return { items, errors };
+}
+
+/** One line describing a tool result, for the `tool-result` trace event. */
+export function summariseResult(harvested: Harvest): string {
+ if (harvested.errors.length > 0 && harvested.items.length === 0) {
+ return harvested.errors[0].message;
+ }
+ if (harvested.items.length === 0) return 'no records returned';
+
+ const labels = [...new Set(harvested.items.map((item) => item.label.split(' · ')[0]))];
+ const shown = labels.slice(0, 3).join(', ');
+ const more = labels.length > 3 ? ` +${labels.length - 3}` : '';
+ return `${harvested.items.length} record${harvested.items.length === 1 ? '' : 's'} — ${shown}${more}`;
+}
+
+/** One line describing a tool call's arguments, for the `tool-called` event. */
+export function summariseArgs(input: unknown): string {
+ if (!isRecord(input)) return '';
+ return Object.entries(input)
+ .filter(([, value]) => value !== undefined && value !== null && value !== '')
+ .map(([key, value]) => `${key}=${typeof value === 'string' ? value : JSON.stringify(value)}`)
+ .join(' ')
+ .slice(0, 160);
+}
+
+/**
+ * Evidence as it is handed to a draft or verify call: numbered, labelled, and
+ * capped. The cap is on the digest rather than per item because a section that
+ * gathered from three tools should not get three times the budget.
+ */
+export function formatEvidence(items: EvidenceItem[], maxChars = 3_400): string {
+ if (items.length === 0) return '(no evidence was retrieved for this section)';
+
+ const lines: string[] = [];
+ let used = 0;
+ for (const item of items) {
+ const line = `[${item.id}] ${item.label}: ${item.text}`;
+ if (used + line.length > maxChars) break;
+ lines.push(line);
+ used += line.length + 1;
+ }
+ return lines.join('\n');
+}
diff --git a/app/src/lib/agent/pacer.test.ts b/app/src/lib/agent/pacer.test.ts
new file mode 100644
index 0000000..e53f823
--- /dev/null
+++ b/app/src/lib/agent/pacer.test.ts
@@ -0,0 +1,179 @@
+import { describe, expect, it } from 'vitest';
+
+import { TokenBudget, parseRateLimitError } from '@/lib/llm/rate-limit';
+
+import {
+ TokenPacer,
+ estimateToolLoopTokens,
+ estimateTokens,
+ isDailyLimit,
+ isRateLimitError,
+} from './pacer';
+
+/**
+ * A budget already told what the endpoint's ceiling is, without a live call.
+ * `declareLimit` is the same door `LLM_TOKENS_PER_MINUTE` comes through, so
+ * these exercise the production path rather than a test-only seam.
+ */
+function measured(limit = 8_000): TokenBudget {
+ const budget = new TokenBudget();
+ budget.declareLimit(limit);
+ return budget;
+}
+
+describe('TokenPacer', () => {
+ it('lets a run through while the endpoint has room', () => {
+ expect(new TokenPacer(measured()).waitFor(3_000)).toBe(0);
+ });
+
+ it('holds the next call until the bucket has refilled enough', async () => {
+ const pacer = new TokenPacer(measured());
+ await pacer.reserve(5_000);
+ await pacer.reserve(2_500);
+
+ // 7,500 of 8,000 spent. The bucket refills continuously rather than in
+ // steps, so the wait is the time to earn back what is short — not a full
+ // window. The old sliding-window pacer waited nearly a minute here, and
+ // that difference is most of why a six-section brief could not finish.
+ const wait = pacer.waitFor(1_000);
+ expect(wait).toBeGreaterThan(0);
+ expect(wait).toBeLessThan(30_000);
+ });
+
+ it('is silent about an endpoint that never claims a ceiling', async () => {
+ // No headers seen and nothing declared: local Ollama, vLLM, anything self
+ // hosted. Replaces the old `isLocalInference()` URL check.
+ const pacer = new TokenPacer(new TokenBudget());
+ await pacer.reserve(1_000_000);
+ expect(pacer.waitFor(1_000_000)).toBe(0);
+ expect(pacer.isPacing).toBe(false);
+ });
+
+ it('lets a call larger than the whole budget through rather than stalling forever', () => {
+ expect(new TokenPacer(measured()).waitFor(20_000)).toBe(0);
+ });
+
+ /*
+ * The failure this file was rewritten for. Groq refused a live Sudan brief on
+ * its per-day ceiling while its per-minute headers reported the bucket full,
+ * and the engine — which only knew about minutes — retried into it for twelve
+ * minutes. A daily block has to be legible as something other than pacing.
+ */
+ it('reports a daily ceiling separately from a per-minute one', () => {
+ const budget = measured();
+ budget.observeRefusal(
+ parseRateLimitError(
+ 'Rate limit reached for model `qwen/qwen3.8-27b` in organization `org_x` ' +
+ 'service tier `on_demand` on tokens per day (TPD): Limit 200000, Used 199754, ' +
+ 'Requested 2679. Please try again in 17m31.056s.',
+ ),
+ );
+ const pacer = new TokenPacer(budget);
+
+ expect(pacer.dailyBlockMs()).toBeGreaterThan(17 * 60_000);
+ // The per-minute bucket is untouched by this, which is exactly the trap:
+ // a pacer looking only here would conclude everything was fine.
+ expect(pacer.waitFor(1_000)).toBe(0);
+ expect(pacer.snapshot().dailyExhausted?.used).toBe(199_754);
+ });
+});
+
+describe('parseRateLimitError', () => {
+ it('reads the numbers out of a per-day refusal', () => {
+ const facts = parseRateLimitError(
+ 'on tokens per day (TPD): Limit 200000, Used 199754, Requested 2679. ' +
+ 'Please try again in 17m31.056s.',
+ );
+ expect(facts).toMatchObject({
+ scope: 'tokens-per-day',
+ limit: 200_000,
+ used: 199_754,
+ requested: 2_679,
+ });
+ expect(facts.retryAfterMs).toBeCloseTo(1_051_056, -3);
+ });
+
+ /*
+ * The other half of the live failure: a request refused not for the account
+ * being out of budget but for reserving a completion larger than the whole
+ * per-minute ceiling. `Requested 16395` against `Limit 8000` is an uncapped
+ * `maxOutputTokens`, not a busy endpoint — see the engine's output caps.
+ */
+ it('reads a per-minute refusal caused by an oversized reservation', () => {
+ expect(
+ parseRateLimitError(
+ 'on tokens per minute (TPM): Limit 8000, Requested 16395, please reduce your message size',
+ ),
+ ).toMatchObject({ scope: 'tokens-per-minute', limit: 8_000, requested: 16_395 });
+ });
+
+ it('degrades to unknown rather than inventing a scope', () => {
+ expect(parseRateLimitError('something went wrong').scope).toBe('unknown');
+ });
+});
+
+describe('isDailyLimit', () => {
+ it('separates the wall from the pause', () => {
+ expect(isDailyLimit(new Error('on tokens per day (TPD): Limit 200000'))).toBe(true);
+ expect(isDailyLimit(new Error('on tokens per minute (TPM): Limit 8000'))).toBe(false);
+ });
+});
+
+describe('estimateTokens', () => {
+ it('counts the prompt and the expected reply', () => {
+ expect(estimateTokens('x'.repeat(350), 100)).toBe(200);
+ });
+});
+
+describe('isRateLimitError', () => {
+ it('recognises the endpoint saying "too fast"', () => {
+ expect(isRateLimitError(new Error('Rate limit reached for model, try again in 8.5s'))).toBe(true);
+ expect(isRateLimitError({ statusCode: 429 })).toBe(true);
+ expect(isRateLimitError(new Error('Limit 8000, Used 7000, Requested 2000 tokens per min'))).toBe(true);
+ });
+
+ it('does not retry a request that was simply wrong', () => {
+ expect(isRateLimitError(new Error('property reasoning_content is unsupported'))).toBe(false);
+ expect(isRateLimitError({ statusCode: 400 })).toBe(false);
+ });
+});
+
+describe('estimateToolLoopTokens', () => {
+ /*
+ * The arithmetic that a live Sudan brief proved was missing. Three steps do
+ * not cost three tool results: results accumulate, so step two re-sends step
+ * one's and step three re-sends both. Reserving a flat per-call figure
+ * under-estimated a gather by roughly a factor of three and let enough
+ * requests through in one minute for Groq to refuse one.
+ */
+ it('charges for tool results being re-sent on every later step', () => {
+ const oneStep = estimateToolLoopTokens(2, 1, 0);
+ const threeSteps = estimateToolLoopTokens(2, 3, 0);
+
+ // 1 result vs 1+2+3 = 6, plus schemas re-sent each step.
+ expect(threeSteps).toBeGreaterThan(oneStep * 3);
+ });
+
+ it('charges for every tool in the subset, not just the one called', () => {
+ expect(estimateToolLoopTokens(4, 2, 0)).toBeGreaterThan(estimateToolLoopTokens(1, 2, 0));
+ });
+
+ /*
+ * The constraint that set `GATHER_STEP_CAP` in the engine. A gather has to
+ * leave room in the same minute for the draft and verify calls that follow
+ * it, or every section waits out a full window and a six-section brief takes
+ * ten minutes. At three steps this came to 6,630 of the 8,000 available —
+ * which is what sent it back to two.
+ */
+ it('leaves room in the minute for the draft and verify that follow a gather', () => {
+ const gather = estimateToolLoopTokens(3, 2);
+ expect(gather).toBeLessThan(6_000);
+
+ // Draft and verify are prompt-plus-output calls of roughly this size.
+ expect(gather + 2 * 1_000).toBeLessThan(8_000);
+ });
+
+ it('shows why a third step does not fit', () => {
+ expect(estimateToolLoopTokens(3, 3)).toBeGreaterThan(6_000);
+ });
+});
diff --git a/app/src/lib/agent/pacer.ts b/app/src/lib/agent/pacer.ts
new file mode 100644
index 0000000..8fb9e0e
--- /dev/null
+++ b/app/src/lib/agent/pacer.ts
@@ -0,0 +1,278 @@
+/**
+ * Keeping a multi-step workflow inside the endpoint's tokens-per-minute
+ * ceiling, and surviving it when the estimate is wrong.
+ *
+ * Chat gets away without this. One turn is a handful of calls and a person
+ * typing between them, so the rate limit is never the binding constraint. A
+ * deliverable is different in kind: the situation brief is six sections and
+ * roughly eighteen model calls fired back to back with no human in the loop.
+ * Against the deployed hosted configuration — Groq's free tier, 8,000 tokens a
+ * minute — that sequence will exhaust a minute's budget somewhere in section
+ * two and then fail, and it fails in the worst way: half a document, the trace
+ * stopped mid-tick, and a 429 the user cannot act on.
+ *
+ * So the engine paces itself. Before each call it projects a cost; the pacer
+ * holds it until the endpoint has room. The wait is real time the user spends
+ * watching the trace panel rather than an error page, which is the trade this
+ * whole feature is built around — a brief that takes three minutes and is
+ * auditable beats one that takes forty seconds and is not.
+ *
+ * # What changed, and why the first version was not enough
+ *
+ * This file used to be the whole accountant: a per-run sliding window of
+ * *estimated* spends, enabled by checking whether the base URL looked local.
+ * Both halves were wrong in production.
+ *
+ * The estimates were wrong because the quantity that matters is not knowable
+ * from the prompt string — the SDK adds tool schemas, accumulated tool results
+ * and its own internal retries on the way out, and a hosted brief spent well
+ * over what the arithmetic here predicted. The window was wrong because it was
+ * per run, while the quota is per account: a chat turn arriving mid-brief, or a
+ * second brief, spent from a bucket this pacer did not know existed.
+ *
+ * Both are now read from the endpoint's own rate-limit headers on every
+ * response — see `lib/llm/rate-limit.ts`. What is left in this file is the
+ * projection (how much is this next call likely to cost, so we can ask whether
+ * it fits) and the headroom policy. The `isLocalInference()` check is gone with
+ * them: an endpoint that never claims a ceiling is never paced, which covers
+ * local Ollama without naming it.
+ */
+
+import { getModelBudget } from '@/lib/llm/provider';
+import { RATE_LIMITED, parseRateLimitError, type TokenBudget } from '@/lib/llm/rate-limit';
+
+/**
+ * Characters per token. Deliberately pessimistic: the real ratio for English
+ * prose is nearer 4, and the cost of over-estimating is a few seconds of extra
+ * pacing, while the cost of under-estimating is the 429 this exists to avoid.
+ */
+const CHARS_PER_TOKEN = 3.5;
+
+/** What a call is assumed to produce when nothing better is known. */
+const DEFAULT_OUTPUT_TOKENS = 500;
+
+/**
+ * A registered tool's description and JSON schema, as sent on every request.
+ *
+ * HAI's tool descriptions are long on purpose — they carry the grounding policy
+ * ("Mandatory before stating any caseload…") that keeps the model calling them.
+ * That prose is not free: it is re-sent with every step of a tool loop, for
+ * every tool in the subset.
+ *
+ * Measured rather than guessed, by sending the system prompt with and without
+ * the registry and reading `prompt_tokens` back: the four deployed schemas add
+ * 1,483 tokens over the prompt alone, so roughly 370 each — `hazards_context`,
+ * the largest, is 335 on its own. The previous 320 was a little under.
+ */
+const TOOL_SCHEMA_TOKENS = 370;
+
+/** What one tool result adds to the context it is returned into. */
+const TOOL_RESULT_TOKENS = 500;
+
+/**
+ * Tool calls assumed per step.
+ *
+ * Not one. Whether the model may fan out within a single step is decided by
+ * `parallel_tool_calls` in `lib/llm/provider.ts`, which is a deployment
+ * concern rather than this file's — and on the live Sudan run the needs
+ * section issued three calls in one step. Budgeting for one would put the
+ * pacer back where it was: confident, low, and wrong in the direction that
+ * ends in a 429.
+ */
+const RESULTS_PER_STEP = 2;
+
+/**
+ * What a tool-calling loop of `steps` steps costs, beyond its prompt.
+ *
+ * The term that matters is the accumulation, and getting it wrong is what put a
+ * live Sudan brief over Groq's ceiling despite the pacer. Every tool result
+ * rides along in every *subsequent* request of the same loop, so three steps do
+ * not cost three results — they cost one, then two, then three, i.e. six. The
+ * first implementation reserved a flat per-call figure, under-estimated a
+ * three-step gather by about a factor of three, and let enough calls through in
+ * one minute for the endpoint to refuse one.
+ */
+export function estimateToolLoopTokens(
+ toolCount: number,
+ steps: number,
+ outputPerStep = 250,
+): number {
+ const schemas = toolCount * TOOL_SCHEMA_TOKENS * steps;
+ const accumulated = ((steps * (steps + 1)) / 2) * TOOL_RESULT_TOKENS * RESULTS_PER_STEP;
+ return schemas + accumulated + steps * outputPerStep;
+}
+
+/**
+ * `LLM_TOKENS_PER_MINUTE`, when a deployment sets one.
+ *
+ * Undefined is now the normal case rather than a fallback to a hard-coded 8,000:
+ * the endpoint reports its own ceiling on every response, so a paid tier or a
+ * different provider is paced correctly with no configuration. This is for the
+ * endpoint that meters silently — see `TokenBudget.declareLimit`.
+ */
+export function declaredTokensPerMinute(): number | undefined {
+ const parsed = Number.parseInt(process.env.LLM_TOKENS_PER_MINUTE ?? '', 10);
+ return Number.isFinite(parsed) && parsed > 0 ? parsed : undefined;
+}
+
+/** Rough token count for a prompt, from its character length. */
+export function estimateTokens(text: string, expectedOutput = DEFAULT_OUTPUT_TOKENS): number {
+ return Math.ceil(text.length / CHARS_PER_TOKEN) + expectedOutput;
+}
+
+/**
+ * Headroom kept below the endpoint's stated ceiling.
+ *
+ * The budget is read from headers that describe the moment a response was
+ * written, and by the time we act on it another request may already be in
+ * flight. Aiming at 100% of the bucket means every mis-timing is a 429; aiming
+ * at 85% costs a few seconds a run and makes the ceiling a wall we stop short
+ * of rather than one we discover by hitting it.
+ */
+const SAFETY_FRACTION = 0.85;
+
+/**
+ * The pacer, now a thin policy layer over a measured budget.
+ *
+ * Everything that used to live here — a per-run sliding window of estimated
+ * spends, an enabled flag derived from the base URL — is gone, because it was
+ * modelling something the endpoint reports directly. What remains is the two
+ * decisions that are genuinely ours: how much headroom to leave under the
+ * stated ceiling, and how to project the cost of a call that has not been made
+ * yet. See `lib/llm/rate-limit.ts` for why the reading is authoritative.
+ *
+ * The budget is shared per process and per model rather than per run. That is a
+ * deliberate reversal: two runs on one instance really do take from the same
+ * quota, and a pacer that gave each its own imaginary window was how a second
+ * request pushed the first into a 429 it had already paced around.
+ */
+export class TokenPacer {
+ private readonly budget: TokenBudget;
+
+ /**
+ * Tests inject a budget; production takes the shared one for the configured
+ * model. A `LLM_TOKENS_PER_MINUTE` override is declared onto the budget
+ * rather than held here, so that everything reading the budget — the engine's
+ * trace, the chat route's queue notice — sees one consistent ceiling.
+ */
+ constructor(budget: TokenBudget = getModelBudget()) {
+ this.budget = budget;
+ const declared = declaredTokensPerMinute();
+ if (declared !== undefined) this.budget.declareLimit(declared);
+ }
+
+ /** Whether the endpoint has claimed a ceiling this pacer must respect. */
+ get isPacing(): boolean {
+ return this.budget.isMeasured;
+ }
+
+ /**
+ * How long the endpoint's per-day ceiling stays shut, or 0.
+ *
+ * Kept separate from `waitFor` because the caller must be able to act
+ * differently: a per-minute wait is paced through, a per-day one is reported
+ * and the run stops. Folding them together is what let a run spend twelve
+ * minutes retrying into a limit that resets tomorrow.
+ */
+ dailyBlockMs(now = Date.now()): number {
+ return this.budget.dailyBlockFor(now);
+ }
+
+ snapshot() {
+ return this.budget.snapshot();
+ }
+
+ /**
+ * Milliseconds to wait before spending `tokens`, or 0. Exposed separately
+ * from `reserve` so the engine can announce a long wait in the trace before
+ * going quiet for it, and so tests can assert the arithmetic without sleeping.
+ */
+ waitFor(tokens: number, now = Date.now()): number {
+ return this.budget.waitFor(tokens / SAFETY_FRACTION, now);
+ }
+
+ /** Wait if needed, then record the spend against the in-flight ledger. */
+ async reserve(tokens: number): Promise {
+ const wait = this.waitFor(tokens);
+ if (wait > 0) await sleep(wait);
+ this.budget.debit(tokens);
+ }
+
+ /**
+ * Correct a reservation once real usage is known.
+ *
+ * Mostly a no-op now, and deliberately so: the response that carried the
+ * usage also carried `x-ratelimit-remaining-tokens`, which the provider's
+ * fetch has already folded in as ground truth. This only matters for an
+ * endpoint that meters but does not say so in headers, where the estimate
+ * remains the only ledger there is.
+ */
+ settle(estimated: number, actual: number): void {
+ if (!this.budget.isMeasured) return;
+ const correction = actual - estimated;
+ if (correction > 0) this.budget.debit(correction);
+ }
+}
+
+export function sleep(ms: number): Promise {
+ return new Promise((resolve) => setTimeout(resolve, ms));
+}
+
+/**
+ * Whether an error is the endpoint saying "too fast" rather than "wrong".
+ * Matched on text because the OpenAI-compatible provider surfaces upstream
+ * 429s as errors whose useful detail is in the message, and because a rate
+ * limit retried once is recoverable while a 400 retried once is just slower.
+ */
+export function isRateLimitError(error: unknown): boolean {
+ if (typeof error === 'object' && error !== null) {
+ const status = (error as { statusCode?: number; status?: number }).statusCode ??
+ (error as { status?: number }).status;
+ if (status === 429) return true;
+ }
+ const message = error instanceof Error ? error.message : String(error ?? '');
+ return RATE_LIMITED.test(message);
+}
+
+/**
+ * Retry a model call through a rate limit, and only through a rate limit.
+ *
+ * Two attempts, not more. A workflow that keeps retrying turns one slow section
+ * into a run that never visibly finishes, and the engine's own fallback — mark
+ * the step degraded, say so in the trace, keep the document going — is a better
+ * outcome for the reader than a longer silence.
+ */
+export async function withRateLimitRetry(
+ operation: () => Promise,
+ onWait?: (ms: number) => void,
+): Promise {
+ try {
+ return await operation();
+ } catch (error) {
+ if (!isRateLimitError(error)) throw error;
+ // A daily ceiling is not something to retry into. The retry that used to
+ // happen here waited out a capped 45 seconds against a limit whose own
+ // message said "try again in 17m31s", burned a request doing it, and failed
+ // again — which is how a single exhausted quota turned into a run that
+ // looked hung for twelve minutes before dying.
+ if (isDailyLimit(error)) throw error;
+ const wait = retryAfterMs(error) ?? 20_000;
+ onWait?.(wait);
+ await sleep(wait);
+ return operation();
+ }
+}
+
+/** Whether a refusal was about the per-day ceiling rather than the per-minute one. */
+export function isDailyLimit(error: unknown): boolean {
+ const message = error instanceof Error ? error.message : String(error ?? '');
+ return parseRateLimitError(message).scope === 'tokens-per-day';
+}
+
+/** The endpoint's own `retry-after`, when it sent one, capped at 45s. */
+function retryAfterMs(error: unknown): number | undefined {
+ const message = error instanceof Error ? error.message : String(error ?? '');
+ const parsed = parseRateLimitError(message).retryAfterMs;
+ if (parsed === undefined) return undefined;
+ return Math.min(45_000, parsed + 500);
+}
diff --git a/app/src/lib/agent/render.test.ts b/app/src/lib/agent/render.test.ts
new file mode 100644
index 0000000..833236b
--- /dev/null
+++ b/app/src/lib/agent/render.test.ts
@@ -0,0 +1,254 @@
+import { describe, expect, it } from 'vitest';
+
+import type { EvidenceItem } from './evidence';
+import {
+ INVENTED_CITATION_MARK,
+ UNVERIFIED_MARK,
+ assembleDocument,
+ documentFilename,
+ foldRun,
+ renderSection,
+ renderSourcesAndCaveats,
+} from './render';
+
+const evidence: EvidenceItem[] = [
+ { id: 'e1', tool: 'search_standards', label: 'sphere · WASH 2.1', text: '15 litres' },
+ { id: 'e2', tool: 'humanitarian_data', label: 'HDX HAPI · funding', text: '$1.2bn' },
+];
+
+describe('renderSection', () => {
+ it('turns citation ids into the source labels a reader can check', () => {
+ const { markdown, invented } = renderSection(
+ 'The minimum is 15 litres per person per day [e1].',
+ evidence,
+ );
+ expect(markdown).toBe('The minimum is 15 litres per person per day (sphere · WASH 2.1).');
+ expect(invented).toBe(0);
+ });
+
+ /*
+ * The failure this exists for: a model that produces a claim and then
+ * produces a provenance for it. `[e9]` where the evidence stopped at `[e2]`
+ * is a fabricated citation, and it is more dangerous than an uncited claim
+ * because it reads as more trustworthy. It must be visible in the prose.
+ */
+ it('marks a citation that refers to no evidence and counts it', () => {
+ const { markdown, invented } = renderSection('Funding reached $3bn [e9].', evidence);
+ expect(markdown).toContain(INVENTED_CITATION_MARK);
+ expect(markdown).not.toContain('[e9]');
+ expect(invented).toBe(1);
+ });
+
+ /*
+ * Seen on the first live Sudan brief: a sentence resting on two passages came
+ * back as `[e58, e59]`, which the single-id pattern did not match, so the raw
+ * ids shipped in the finished document.
+ */
+ it('resolves several ids in one bracket', () => {
+ const { markdown, invented } = renderSection('Minimum 15 litres [e1, e2].', evidence);
+ expect(markdown).toBe('Minimum 15 litres (sphere · WASH 2.1; HDX HAPI · funding).');
+ expect(invented).toBe(0);
+ });
+
+ it('still counts an unknown id inside a multi-id bracket', () => {
+ const { markdown, invented } = renderSection('Minimum 15 litres [e1, e9].', evidence);
+ expect(markdown).toBe('Minimum 15 litres (sphere · WASH 2.1).');
+ expect(invented).toBe(1);
+ });
+
+ it('collapses a source cited several times in a row', () => {
+ const { markdown } = renderSection('Requirements are $2bn [e2][e2][e2].', evidence);
+ expect(markdown).toBe('Requirements are $2bn (HDX HAPI · funding).');
+ });
+
+ it('drops a heading the draft step was told not to write', () => {
+ const { markdown } = renderSection('## Overview\n\nSudan faces a crisis.', evidence);
+ expect(markdown).toBe('Sudan faces a crisis.');
+ });
+});
+
+describe('renderSourcesAndCaveats', () => {
+ it('names every degraded source once, however many sections hit it', () => {
+ const markdown = renderSourcesAndCaveats({
+ evidence,
+ errors: [
+ { source: 'hazards_context · worldbank', message: 'HTTP 503' },
+ { source: 'hazards_context · worldbank', message: 'HTTP 503' },
+ ],
+ generatedAt: '2026-09-01T10:00:00.000Z',
+ flagged: 0,
+ });
+
+ expect(markdown.match(/HTTP 503/g)).toHaveLength(1);
+ expect(markdown).toContain('Source issues (degraded sections)');
+ expect(markdown).toContain('2026-09-01T10:00:00.000Z');
+ });
+
+ it('states plainly when nothing degraded', () => {
+ const markdown = renderSourcesAndCaveats({
+ evidence,
+ errors: [],
+ generatedAt: '2026-09-01T10:00:00.000Z',
+ flagged: 0,
+ });
+ expect(markdown).toContain('Every source consulted returned data');
+ expect(markdown).toContain('Every claim in this document matched retrieved evidence');
+ });
+
+ it('tells the reader how many claims to check before sending the document on', () => {
+ const markdown = renderSourcesAndCaveats({
+ evidence,
+ errors: [],
+ generatedAt: '2026-09-01T10:00:00.000Z',
+ flagged: 2,
+ });
+ expect(markdown).toContain('2 claims');
+ expect(markdown).toContain(UNVERIFIED_MARK);
+ });
+
+ it('counts each source once per record, not once per section', () => {
+ const markdown = renderSourcesAndCaveats({
+ evidence: [...evidence, { ...evidence[0], id: 'e3' }],
+ errors: [],
+ generatedAt: '2026-09-01T10:00:00.000Z',
+ flagged: 0,
+ });
+ expect(markdown).toContain('sphere — 2 records');
+ expect(markdown).toContain('HDX HAPI — 1 record');
+ });
+});
+
+describe('assembleDocument', () => {
+ it('builds the exported markdown from the same bodies the page shows', () => {
+ const markdown = assembleDocument('Situation brief — Sudan', [
+ { id: 'overview', heading: 'Overview', markdown: 'Body one.' },
+ { id: 'empty', heading: 'Hazards', markdown: ' ' },
+ { id: 'needs', heading: 'Needs', markdown: 'Body two.' },
+ ]);
+
+ expect(markdown).toBe(
+ '# Situation brief — Sudan\n\n## Overview\n\nBody one.\n\n## Needs\n\nBody two.\n',
+ );
+ });
+});
+
+describe('documentFilename', () => {
+ it('sorts by date and survives every filesystem', () => {
+ expect(documentFilename('Situation brief — Sudan', new Date('2026-09-01T10:00:00Z'))).toBe(
+ 'situation-brief-sudan-2026-09-01.md',
+ );
+ });
+});
+
+describe('foldRun', () => {
+ const at = 1;
+
+ it('assembles the document in the order the plan declared', () => {
+ const run = foldRun(
+ [
+ {
+ type: 'plan-created',
+ at,
+ workflowId: 'situation-brief',
+ subject: 'Sudan',
+ iso3: 'SDN',
+ sections: [
+ { id: 'overview', heading: 'Overview' },
+ { id: 'needs', heading: 'Needs' },
+ ],
+ },
+ // Arrives second; must still render under Overview.
+ { type: 'draft-section', at, sectionId: 'needs', heading: 'Needs', markdown: 'N.' },
+ { type: 'draft-section', at, sectionId: 'overview', heading: 'Overview', markdown: 'O.' },
+ ],
+ 'Situation brief',
+ );
+
+ expect(run.title).toBe('Situation brief — Sudan');
+ expect(run.sections.map((section) => section.id)).toEqual(['overview', 'needs']);
+ });
+
+ it('shows a section filling in before it is finished', () => {
+ const run = foldRun(
+ [
+ { type: 'draft-delta', at, sectionId: 'overview', delta: 'Sudan faces ' },
+ { type: 'draft-delta', at, sectionId: 'overview', delta: 'a crisis.' },
+ ],
+ 'Situation brief',
+ );
+ expect(run.sections[0].markdown).toBe('Sudan faces a crisis.');
+ });
+
+ /*
+ * The ordering that matters most: the verified body must win over the drafted
+ * one, or the flags this whole feature exists to show would be overwritten by
+ * the unflagged text they were added to.
+ */
+ it('replaces a drafted section with its verified version', () => {
+ const run = foldRun(
+ [
+ { type: 'draft-delta', at, sectionId: 'needs', delta: '2,000,000 in need.' },
+ { type: 'draft-section', at, sectionId: 'needs', heading: 'Needs', markdown: '2,000,000 in need.' },
+ {
+ type: 'section-verified',
+ at,
+ sectionId: 'needs',
+ markdown: `2,000,000 in need. ${UNVERIFIED_MARK}`,
+ flagged: 1,
+ },
+ // A straggling delta must not append underneath the flag.
+ { type: 'draft-delta', at, sectionId: 'needs', delta: ' extra' },
+ ],
+ 'Situation brief',
+ );
+
+ expect(run.sections[0].markdown).toBe(`2,000,000 in need. ${UNVERIFIED_MARK}`);
+ });
+
+ it('reports a finished run and its flag count', () => {
+ const run = foldRun(
+ [{ type: 'workflow-done', at, generatedAt: '2026-09-01T10:00:00.000Z', sections: 6, flagged: 2 }],
+ 'Situation brief',
+ );
+ expect(run.finished).toBe(true);
+ expect(run.flagged).toBe(2);
+ expect(run.failed).toBeNull();
+ });
+
+ it('surfaces a failed run rather than rendering an empty document', () => {
+ const run = foldRun(
+ [{ type: 'workflow-error', at, message: 'endpoint unreachable' }],
+ 'Situation brief',
+ );
+ expect(run.failed).toBe('endpoint unreachable');
+ expect(run.finished).toBe(true);
+ });
+
+ /*
+ * A run cut short by the serverless timeout produces no terminal event at
+ * all. What it produced must still be there, and `finished` must stay false so
+ * the page can say the document is partial.
+ */
+ it('keeps what a truncated run produced, and does not call it finished', () => {
+ const run = foldRun(
+ [
+ {
+ type: 'plan-created',
+ at,
+ workflowId: 'situation-brief',
+ subject: 'Sudan',
+ sections: [
+ { id: 'overview', heading: 'Overview' },
+ { id: 'needs', heading: 'Needs' },
+ ],
+ },
+ { type: 'draft-section', at, sectionId: 'overview', heading: 'Overview', markdown: 'O.' },
+ ],
+ 'Situation brief',
+ );
+
+ expect(run.finished).toBe(false);
+ expect(run.sections[0].markdown).toBe('O.');
+ expect(run.sections[1].markdown).toBe('');
+ });
+});
diff --git a/app/src/lib/agent/render.ts b/app/src/lib/agent/render.ts
new file mode 100644
index 0000000..bfc67a0
--- /dev/null
+++ b/app/src/lib/agent/render.ts
@@ -0,0 +1,317 @@
+/**
+ * Turning drafted section text into the document a reader gets.
+ *
+ * Pure functions with no server dependencies, because the client imports
+ * `assembleDocument` to build the markdown export from exactly the same section
+ * bodies it is displaying. An export path that re-derives the document a second
+ * way is an export path that will eventually disagree with what was on screen,
+ * which for a document someone forwards to a donor is the one bug that must not
+ * exist.
+ */
+
+import type { EvidenceItem, SourceError } from './evidence';
+import type { TraceEvent } from './types';
+
+/**
+ * How an unverified claim is marked, and why it is marked in the markdown
+ * itself rather than styled by the UI.
+ *
+ * The document is copied, downloaded, and pasted into other people's reports —
+ * that is what it is for. A flag that lives only in the web page's CSS survives
+ * none of that, so the sentence would arrive in a donor report looking exactly
+ * as authoritative as the sourced ones around it. Bold literal text survives
+ * copy-paste into Word, plain-text email, and the .md file alike.
+ */
+export const UNVERIFIED_MARK = '**[unverified]**';
+export const INVENTED_CITATION_MARK = '**[citation not in evidence]**';
+
+/**
+ * `[e12]`, and also `[e12, e13]`.
+ *
+ * The prompt asks for one id per bracket and the model mostly complies, but a
+ * sentence resting on two passages comes back as `[e58, e59]` often enough to
+ * matter — and a pattern that only matched the single form left those ids
+ * sitting raw in the finished document, which is the exact failure the
+ * invented-citation path exists to make impossible. Seen on the first live
+ * Sudan brief.
+ */
+const CITATION_PATTERN = /\[(e\d+(?:\s*,\s*e\d+)*)\]/g;
+
+/**
+ * Two or more identical adjacent citations, collapsed to one.
+ *
+ * A model citing every clause of a three-clause sentence to the same source
+ * emits the same label three times in a row, which on the funding section
+ * produced a paragraph ending in eight consecutive parentheticals. The
+ * provenance is unchanged; only the noise goes.
+ */
+const REPEATED_CITATION = /(\([^()]+\))\1+/g;
+
+export interface RenderedSection {
+ markdown: string;
+ /** Citation ids the model produced that no evidence item carries. */
+ invented: number;
+}
+
+/**
+ * Replace evidence ids with their source labels, and expose the ones that
+ * refer to nothing.
+ *
+ * An invented citation is not a formatting problem. `[e9]` in a section whose
+ * evidence stopped at `[e6]` means the model produced a claim and then produced
+ * a provenance for it, which is the most dangerous single failure mode this
+ * whole feature is built to catch — it looks more trustworthy than an uncited
+ * sentence, not less. So it is marked in the prose and counted as a flag.
+ */
+export function renderSection(raw: string, evidence: EvidenceItem[]): RenderedSection {
+ const labels = new Map(evidence.map((item) => [item.id, item.label]));
+ let invented = 0;
+
+ const body = stripLeadingHeading(raw.trim())
+ .replace(CITATION_PATTERN, (match, group: string) => {
+ const resolved: string[] = [];
+ for (const id of group.split(',').map((part) => part.trim())) {
+ const label = labels.get(id);
+ if (label) {
+ // De-duplicated within one bracket as well as across adjacent ones:
+ // `[e4, e4]` is one source, not two.
+ if (!resolved.includes(label)) resolved.push(label);
+ } else {
+ invented += 1;
+ }
+ }
+ if (resolved.length === 0) return INVENTED_CITATION_MARK;
+ return `(${resolved.join('; ')})`;
+ })
+ .replace(REPEATED_CITATION, '$1');
+
+ return { markdown: body, invented };
+}
+
+/**
+ * The draft prompt says not to write a heading, and the model writes one
+ * roughly one time in five. Left in, the document gets two headings per
+ * section — so it is removed here rather than by asking the prompt harder.
+ */
+function stripLeadingHeading(text: string): string {
+ return text.replace(/^#{1,6}\s+.*\n+/, '').trim();
+}
+
+/* ------------------------------------------------------------------ *
+ * Sources and caveats
+ * ------------------------------------------------------------------ */
+
+export interface CaveatsInput {
+ evidence: EvidenceItem[];
+ errors: SourceError[];
+ /** ISO-8601 UTC. */
+ generatedAt: string;
+ flagged: number;
+}
+
+/**
+ * The section that makes the rest of the document auditable: what answered,
+ * what did not, and when.
+ *
+ * Ported from `situation.py`'s `errors` list, which appended a "Source issues
+ * (degraded sections)" block to every report rather than letting a dead
+ * connector fail the run. The reasoning holds exactly: a brief assembled while
+ * one source was unreachable is still worth having, and is only safe to use if
+ * it says so on its face. A reader who cannot see that the funding figures are
+ * missing will assume there were none.
+ */
+export function renderSourcesAndCaveats(input: CaveatsInput): string {
+ const lines: string[] = [];
+
+ const bySource = new Map();
+ for (const item of input.evidence) {
+ const base = item.label.split(' · ')[0];
+ bySource.set(base, (bySource.get(base) ?? 0) + 1);
+ }
+
+ if (bySource.size > 0) {
+ lines.push('**Sources consulted**');
+ lines.push('');
+ for (const [source, count] of [...bySource].sort((a, b) => b[1] - a[1])) {
+ lines.push(`- ${source} — ${count} record${count === 1 ? '' : 's'}`);
+ }
+ } else {
+ lines.push('**Sources consulted** — none. No source answered for this brief.');
+ }
+
+ lines.push('');
+ if (input.errors.length > 0) {
+ lines.push('**Source issues (degraded sections)**');
+ lines.push('');
+ // De-duplicated: one unreachable API fails once per section that asked it,
+ // and six identical lines reads as six problems rather than one.
+ const seen = new Set();
+ for (const error of input.errors) {
+ const key = `${error.source}::${error.message}`;
+ if (seen.has(key)) continue;
+ seen.add(key);
+ lines.push(`- \`${error.source}\` — ${error.message}`);
+ }
+ } else {
+ lines.push('**Source issues** — none. Every source consulted returned data.');
+ }
+
+ lines.push('');
+ lines.push('**Caveats**');
+ lines.push('');
+ lines.push(
+ input.flagged > 0
+ ? `- ${input.flagged} claim${input.flagged === 1 ? '' : 's'} in this document could not be matched to retrieved evidence and ${input.flagged === 1 ? 'is' : 'are'} marked ${UNVERIFIED_MARK} in place. Verify ${input.flagged === 1 ? 'it' : 'them'} against the primary source before using this document externally.`
+ : '- Every claim in this document matched retrieved evidence at generation time.',
+ );
+ lines.push(
+ '- Figures carry the reference period of the dataset that supplied them, not the date below. A figure can be current in its source and months out of date in the field.',
+ );
+ lines.push(
+ '- This is a machine-assembled starting point for a human analyst, not a cleared product. It has not been reviewed by anyone.',
+ );
+ lines.push('');
+ lines.push(`_Generated ${input.generatedAt} · HAI_`);
+
+ return lines.join('\n');
+}
+
+/* ------------------------------------------------------------------ *
+ * Document assembly
+ * ------------------------------------------------------------------ */
+
+export interface DocumentSection {
+ id: string;
+ heading: string;
+ markdown: string;
+}
+
+/**
+ * The whole deliverable as one markdown string — what the copy button copies
+ * and the download button writes. Sections with no body are omitted rather than
+ * emitted as an empty heading, so a run stopped half way exports what it has.
+ */
+export function assembleDocument(title: string, sections: DocumentSection[]): string {
+ const parts = [`# ${title}`, ''];
+ for (const section of sections) {
+ if (!section.markdown.trim()) continue;
+ parts.push(`## ${section.heading}`, '', section.markdown.trim(), '');
+ }
+ return parts.join('\n').trimEnd() + '\n';
+}
+
+/** A filename that sorts by date and survives every filesystem. */
+export function documentFilename(title: string, generatedAt = new Date()): string {
+ const slug = title
+ .toLowerCase()
+ .replace(/[^a-z0-9]+/g, '-')
+ .replace(/^-|-$/g, '')
+ .slice(0, 60);
+ const date = generatedAt.toISOString().slice(0, 10);
+ return `${slug || 'hai-deliverable'}-${date}.md`;
+}
+
+/* ------------------------------------------------------------------ *
+ * Deriving the document from the trace
+ * ------------------------------------------------------------------ */
+
+export interface RunState {
+ title: string;
+ sections: DocumentSection[];
+ flagged: number;
+ finished: boolean;
+ failed: string | null;
+}
+
+/**
+ * Fold the event stream into the document.
+ *
+ * Everything on screen comes from here, which is deliberate: the document is a
+ * projection of the trace rather than a parallel stream. A section cannot
+ * appear that the trace does not explain, and the `.md` export is built from
+ * these same bodies — so what someone pastes into a report is byte-for-byte
+ * what they read on the page, flags included.
+ *
+ * `draft-delta` fills a section in as it is written; `draft-section` replaces it
+ * with the rendered version once citations have been resolved; and
+ * `section-verified` replaces it again with the annotated one. Later events win,
+ * which is what makes the flags appear in place rather than as an afterthought.
+ */
+export function foldRun(events: TraceEvent[], fallbackTitle: string): RunState {
+ const order: string[] = [];
+ const headings = new Map();
+ const bodies = new Map();
+ const settled = new Set();
+
+ let title = fallbackTitle;
+ let flagged = 0;
+ let finished = false;
+ let failed: string | null = null;
+
+ for (const event of events) {
+ switch (event.type) {
+ case 'plan-created':
+ title = `${fallbackTitle} — ${event.subject}`;
+ for (const section of event.sections) {
+ if (!headings.has(section.id)) order.push(section.id);
+ headings.set(section.id, section.heading);
+ }
+ break;
+
+ case 'draft-delta':
+ // Ignored once the section has been published whole: a late delta from
+ // a section already replaced by its verified version would append
+ // duplicate prose under the flags.
+ if (settled.has(event.sectionId)) break;
+ // Registered here as well as on `plan-created`, so prose is never
+ // dropped for want of a plan. A run whose plan step failed still
+ // streams sections, and silently discarding them would be the worst
+ // possible response to a degraded run.
+ if (!headings.has(event.sectionId)) {
+ order.push(event.sectionId);
+ headings.set(event.sectionId, event.sectionId);
+ }
+ bodies.set(event.sectionId, (bodies.get(event.sectionId) ?? '') + event.delta);
+ break;
+
+ case 'draft-section':
+ bodies.set(event.sectionId, event.markdown);
+ settled.add(event.sectionId);
+ if (!headings.has(event.sectionId)) {
+ order.push(event.sectionId);
+ headings.set(event.sectionId, event.heading);
+ }
+ break;
+
+ case 'section-verified':
+ bodies.set(event.sectionId, event.markdown);
+ settled.add(event.sectionId);
+ break;
+
+ case 'workflow-done':
+ flagged = event.flagged;
+ finished = true;
+ break;
+
+ case 'workflow-error':
+ failed = event.message;
+ finished = true;
+ break;
+
+ default:
+ break;
+ }
+ }
+
+ return {
+ title,
+ sections: order.map((id) => ({
+ id,
+ heading: headings.get(id) ?? id,
+ markdown: bodies.get(id) ?? '',
+ })),
+ flagged,
+ finished,
+ failed,
+ };
+}
diff --git a/app/src/lib/agent/types.ts b/app/src/lib/agent/types.ts
new file mode 100644
index 0000000..359a3e9
--- /dev/null
+++ b/app/src/lib/agent/types.ts
@@ -0,0 +1,265 @@
+/**
+ * The vocabulary of HAI's agentic workflows: what a deliverable is made of,
+ * and what the machine says about itself while it makes one.
+ *
+ * Two ideas live here and nothing else, so both the server engine and the
+ * client trace panel can import them without dragging the AI SDK into the
+ * browser bundle:
+ *
+ * 1. `WorkflowDefinition` — a deliverable expressed as an ordered list of
+ * bounded model calls. Not a free-running agent loop: every step names its
+ * own prompt and its own allowed tools, and the sequence is fixed in advance
+ * by a template. That is deliberate. A loop that decides its own next move
+ * is impossible to show honestly to a reader and impossible to keep inside a
+ * token budget; a declared sequence is both.
+ *
+ * 2. `TraceEvent` — the record of what actually happened. Every claim the
+ * finished document makes should be traceable back to an event in this
+ * stream, which is the whole point: a humanitarian brief that cannot be
+ * audited is not usable, however well written.
+ */
+
+/* ------------------------------------------------------------------ *
+ * Workflow definition
+ * ------------------------------------------------------------------ */
+
+/**
+ * What a step does. Each maps to a different shape of model call, and the
+ * engine treats them differently — see `engine.ts`.
+ *
+ * - `plan` resolves the subject (country name → ISO3, display name) and
+ * publishes the section checklist. One tiny structured call.
+ * - `gather` runs a tool-calling loop with a named tool subset and turns what
+ * comes back into evidence. No prose is produced.
+ * - `draft` writes one section from the evidence gathered for it. No tools —
+ * a drafting step that can still call tools is a drafting step that
+ * will pad a thin section with a fresh search instead of admitting
+ * the gap.
+ * - `verify` extracts the factual claims from a drafted section and checks each
+ * one against the evidence that step was given.
+ */
+export type StepKind = 'plan' | 'gather' | 'draft' | 'verify';
+
+/** A verdict on one extracted claim. */
+export type Verdict = 'supported' | 'unsupported' | 'unverifiable';
+
+/** A section of the finished deliverable, as declared by the template. */
+export interface SectionSpec {
+ id: string;
+ /** English heading. Rendered into the document itself, so it is not i18n chrome. */
+ heading: string;
+ /**
+ * What this section must contain, in the second person, handed to the draft
+ * step verbatim. Keep it to a few lines: it is re-sent on every draft call,
+ * and the hosted deployment is metered by the minute.
+ */
+ brief: string;
+ /**
+ * Tool names this section's gather step may call, matched against the keys of
+ * the `haiTools` registry. Names that are not registered are skipped rather
+ * than erroring, so a tool added later flows in by being registered and a
+ * tool removed does not break every template that mentioned it.
+ */
+ tools: readonly string[];
+ /**
+ * What to search for, as instructions to the gather step. Written in the
+ * vocabulary of the standards and the datasets rather than the user's.
+ */
+ gatherBrief: string;
+ /**
+ * Sections that assemble from the run's own bookkeeping (sources, per-source
+ * errors, timestamp) rather than from a model call. They still appear in the
+ * plan and still tick through the trace; they just have no LLM step.
+ */
+ synthesised?: 'sources-and-caveats';
+ /** Skip verification for a section that makes no factual claims of its own. */
+ skipVerification?: boolean;
+}
+
+export interface WorkflowDefinition {
+ id: string;
+ /** English title of the deliverable, used in the exported markdown. */
+ title: string;
+ /**
+ * How the subject line is interpreted — a country for the situation brief, a
+ * free-text programme topic for the donor report. Drives the plan step and
+ * the document title.
+ */
+ subjectKind: 'country' | 'topic';
+ /** One line describing the deliverable, shown on the template card. */
+ description: string;
+ sections: readonly SectionSpec[];
+}
+
+/* ------------------------------------------------------------------ *
+ * Trace events
+ * ------------------------------------------------------------------ */
+
+/** A section as announced by the plan step, before any of it exists. */
+export interface PlannedSection {
+ id: string;
+ heading: string;
+}
+
+/**
+ * Everything the engine says about itself, in the order it happens.
+ *
+ * `at` is milliseconds since the epoch on the server. The client renders
+ * durations from it rather than running its own timers, so what the reader sees
+ * is when the work actually happened rather than when the bytes arrived.
+ */
+export type TraceEvent =
+ | {
+ type: 'plan-created';
+ at: number;
+ workflowId: string;
+ /** Resolved display subject, e.g. "Sudan" for an input of "sudan". */
+ subject: string;
+ /** ISO3 where the subject resolved to a country, else undefined. */
+ iso3?: string;
+ sections: PlannedSection[];
+ }
+ | {
+ type: 'step-started';
+ at: number;
+ stepId: string;
+ kind: StepKind;
+ /** English label; the client translates by `kind`, not by this string. */
+ label: string;
+ sectionId?: string;
+ }
+ | {
+ type: 'step-finished';
+ at: number;
+ stepId: string;
+ ok: boolean;
+ /** Present when the step degraded — the reason, in one clause. */
+ note?: string;
+ }
+ | {
+ type: 'tool-called';
+ at: number;
+ stepId: string;
+ callId: string;
+ tool: string;
+ /** Arguments flattened to one readable line — never the raw JSON blob. */
+ args: string;
+ }
+ | {
+ type: 'tool-result';
+ at: number;
+ stepId: string;
+ callId: string;
+ tool: string;
+ ok: boolean;
+ /** What came back, in one line: counts and source labels, not content. */
+ summary: string;
+ }
+ | {
+ /**
+ * A slice of section prose as it is written. Separate from
+ * `draft-section` so the document assembles in front of the reader
+ * instead of appearing in finished blocks — the same reason chat streams.
+ */
+ type: 'draft-delta';
+ at: number;
+ sectionId: string;
+ delta: string;
+ }
+ | {
+ type: 'draft-section';
+ at: number;
+ sectionId: string;
+ heading: string;
+ /** The section as drafted, before verification annotates it. */
+ markdown: string;
+ }
+ | {
+ type: 'check-run';
+ at: number;
+ sectionId: string;
+ claim: string;
+ verdict: Verdict;
+ /** The evidence label that supports the claim, when one does. */
+ source?: string;
+ }
+ | {
+ /**
+ * The section after verification, with unsupported and unverifiable
+ * claims marked in the prose. Replaces the `draft-section` body in the
+ * document view. Always emitted for a verified section, even when every
+ * claim held, so the client never has to guess whether checking finished.
+ */
+ type: 'section-verified';
+ at: number;
+ sectionId: string;
+ markdown: string;
+ flagged: number;
+ }
+ | {
+ /**
+ * A source that failed or reported no coverage. Collected rather than
+ * thrown — a brief with five of six sources is worth having, provided it
+ * says which one is missing. Ported from `situation.py`'s `errors` list.
+ */
+ type: 'source-error';
+ at: number;
+ source: string;
+ message: string;
+ }
+ | {
+ /**
+ * The run is deliberately idle, waiting for the endpoint's token budget
+ * to refill before the next step.
+ *
+ * This exists because the alternative is a lie by omission. Pacing means
+ * a brief spends a large fraction of its wall clock doing nothing on
+ * purpose, and a trace panel that simply stops updating for forty seconds
+ * is indistinguishable from one that has crashed — which is what QA
+ * reported on the live deployment. Naming the wait, with the time it will
+ * take, turns the most alarming part of the run into the most legible.
+ *
+ * `scope` decides what the reader is told, and the two cases are not the
+ * same news. A per-minute wait resolves itself; a per-day one does not,
+ * and dressing it up as "resuming shortly" would leave someone watching a
+ * counter that will still be there tomorrow.
+ */
+ type: 'budget-wait';
+ at: number;
+ /** Milliseconds from `at` until the run intends to continue. */
+ waitMs: number;
+ scope: 'tokens-per-minute' | 'tokens-per-day';
+ /** The step the run is about to take, when it is waiting to take one. */
+ stepId?: string;
+ }
+ | {
+ /** The wait announced by the preceding `budget-wait` is over. */
+ type: 'budget-resumed';
+ at: number;
+ stepId?: string;
+ }
+ | {
+ type: 'workflow-done';
+ at: number;
+ /** ISO-8601 UTC, rendered into the document's own caveats section. */
+ generatedAt: string;
+ sections: number;
+ flagged: number;
+ }
+ | {
+ type: 'workflow-error';
+ at: number;
+ message: string;
+ };
+
+export type TraceEventType = TraceEvent['type'];
+
+/** Narrowing helper for the client, which receives these as opaque data parts. */
+export function isTraceEvent(value: unknown): value is TraceEvent {
+ return (
+ typeof value === 'object' &&
+ value !== null &&
+ typeof (value as { type?: unknown }).type === 'string' &&
+ typeof (value as { at?: unknown }).at === 'number'
+ );
+}
diff --git a/app/src/lib/agent/verify.test.ts b/app/src/lib/agent/verify.test.ts
new file mode 100644
index 0000000..cc6268a
--- /dev/null
+++ b/app/src/lib/agent/verify.test.ts
@@ -0,0 +1,218 @@
+import { describe, expect, it } from 'vitest';
+
+import { scriptedModel, text } from './__testing__/mock-model';
+import type { EvidenceItem } from './evidence';
+import { TokenPacer } from './pacer';
+import { TokenBudget } from '@/lib/llm/rate-limit';
+import { UNVERIFIED_MARK } from './render';
+import { annotate, extractClaims, verifySection } from './verify';
+
+const pacer = new TokenPacer(new TokenBudget());
+
+const evidence: EvidenceItem[] = [
+ {
+ id: 'e1',
+ tool: 'humanitarian_data',
+ label: 'HDX HAPI · food_security',
+ text: 'Sudan IPC acute food insecurity: 8,500,000 people in Phase 3 or above, reference period 2026-06 to 2026-09.',
+ },
+ {
+ id: 'e2',
+ tool: 'search_standards',
+ label: 'sphere · WASH standard 2.1',
+ text: 'The minimum water quantity for drinking, cooking and personal hygiene is 15 litres per person per day.',
+ },
+];
+
+describe('extractClaims', () => {
+ it('pulls the sentences that assert something checkable', () => {
+ const claims = extractClaims(
+ 'Sudan faces a deepening crisis affecting many communities.\n' +
+ '- 8,500,000 people are in IPC Phase 3 or above (HDX HAPI · food_security).\n' +
+ '- The minimum water standard is 15 litres per person per day (sphere · WASH standard 2.1).',
+ );
+
+ expect(claims).toHaveLength(2);
+ expect(claims[0]).toContain('8,500,000');
+ expect(claims[1]).toContain('15 litres');
+ });
+
+ it('ignores framing prose that asserts no figure or standard', () => {
+ expect(extractClaims('This section sets out the operating context for the response.')).toEqual([]);
+ });
+
+ /*
+ * From the first live Sudan brief. A rendered citation contains
+ * "> 2. Water supply", and splitting on that full stop ended the claim inside
+ * its own source label — so `annotate` then inserted the unverified mark into
+ * the middle of the citation, destroying the provenance a reader needs in
+ * order to check the flag.
+ */
+ it('does not split a sentence inside its own citation', () => {
+ const claims = extractClaims(
+ '- Sphere standard 2.1 limits users to 250 people per tap (sphere · Promotion > 2. Water supply > standard 2.1 (4th edition, 2018), p106).',
+ );
+
+ expect(claims).toHaveLength(1);
+ expect(claims[0]).toContain('p106');
+ });
+
+ it('still splits ordinary sentences', () => {
+ const claims = extractClaims(
+ 'The appeal requires USD 2,866,228,593. Funding received is USD 1,172,022,676 to date.',
+ );
+ expect(claims).toHaveLength(2);
+ });
+
+ it('does not run away on a long section', () => {
+ const line = '- 1,000,000 people were reached in the period (source).\n';
+ expect(extractClaims(line.repeat(30)).length).toBeLessThanOrEqual(6);
+ });
+});
+
+describe('verifySection', () => {
+ /*
+ * The cheap path. A figure that is literally in the evidence, in a sentence
+ * about the same subject, must be settled without a model call at all — both
+ * because it is free and because string equality is a stronger guarantee than
+ * a model's opinion about string equality.
+ */
+ it('settles a figure that is in the evidence without asking the model', async () => {
+ let modelCalled = false;
+ const model = scriptedModel(() => {
+ modelCalled = true;
+ return [text('1|unverifiable|-')];
+ });
+
+ const result = await verifySection({
+ sectionId: 'needs',
+ markdown:
+ 'Some 8,500,000 people in Sudan are in IPC Phase 3 acute food insecurity or above, reference period 2026-06 (HDX HAPI · food_security).',
+ evidence,
+ model,
+ pacer,
+ });
+
+ expect(modelCalled).toBe(false);
+ expect(result.checks[0].verdict).toBe('supported');
+ expect(result.checks[0].source).toBe('HDX HAPI · food_security');
+ expect(result.flagged).toBe(0);
+ expect(result.markdown).not.toContain(UNVERIFIED_MARK);
+ });
+
+ /*
+ * The failure this whole feature exists to catch: a fluent, plausible,
+ * correctly-formatted figure that appears in none of the retrieved evidence.
+ * It must end up marked in the prose, not quietly shipped.
+ */
+ it('flags a figure that appears nowhere in the evidence', async () => {
+ const model = scriptedModel(() => [text('1|unsupported|-')]);
+
+ const result = await verifySection({
+ sectionId: 'needs',
+ markdown: 'An estimated 12,300,000 people are displaced across the country.',
+ evidence,
+ model,
+ pacer,
+ });
+
+ expect(result.checks[0].verdict).toBe('unsupported');
+ expect(result.flagged).toBe(1);
+ expect(result.markdown).toBe(
+ `An estimated 12,300,000 people are displaced across the country. ${UNVERIFIED_MARK}`,
+ );
+ });
+
+ it('accepts a paraphrase the string pass could not settle', async () => {
+ const model = scriptedModel(() => [text('1|supported|sphere · WASH standard 2.1')]);
+
+ const result = await verifySection({
+ sectionId: 'guidance',
+ markdown: 'Each person needs at least fifteen litres of water daily for drinking and hygiene.',
+ evidence,
+ model,
+ pacer,
+ });
+
+ expect(result.checks[0].verdict).toBe('supported');
+ expect(result.flagged).toBe(0);
+ });
+
+ /*
+ * A verifier that fails open verifies nothing. If the check itself breaks, or
+ * the model replies with something unparseable, every deferred claim must end
+ * up flagged rather than waved through.
+ */
+ it('flags rather than passes when the check itself fails', async () => {
+ const model = scriptedModel(() => {
+ throw new Error('endpoint unreachable');
+ });
+
+ const result = await verifySection({
+ sectionId: 'needs',
+ markdown: 'An estimated 12,300,000 people are displaced across the country.',
+ evidence,
+ model,
+ pacer,
+ });
+
+ expect(result.checks[0].verdict).toBe('unverifiable');
+ expect(result.markdown).toContain(UNVERIFIED_MARK);
+ });
+
+ it('flags rather than passes when the model replies with nonsense', async () => {
+ const model = scriptedModel(() => [text('I think claim one is probably fine.')]);
+
+ const result = await verifySection({
+ sectionId: 'needs',
+ markdown: 'An estimated 12,300,000 people are displaced across the country.',
+ evidence,
+ model,
+ pacer,
+ });
+
+ expect(result.checks[0].verdict).toBe('unverifiable');
+ expect(result.flagged).toBe(1);
+ });
+
+ it('flags every claim in a section drafted with no evidence at all', async () => {
+ let modelCalled = false;
+ const model = scriptedModel(() => {
+ modelCalled = true;
+ return [text('1|supported|-')];
+ });
+
+ const result = await verifySection({
+ sectionId: 'funding',
+ markdown: 'The appeal requires 2,600,000,000 USD for the current year.',
+ evidence: [],
+ model,
+ pacer,
+ });
+
+ // No evidence means nothing can support anything; spending a model call to
+ // be told so would be a waste of the token budget.
+ expect(modelCalled).toBe(false);
+ expect(result.flagged).toBe(1);
+ expect(result.markdown).toContain(UNVERIFIED_MARK);
+ });
+});
+
+describe('annotate', () => {
+ it('marks the claim in place, not in a footnote at the end', () => {
+ const markdown = '- 1,000,000 people in need (source).\n- 2,000,000 targeted (source).';
+ const output = annotate(markdown, [
+ { claim: '1,000,000 people in need (source).', verdict: 'supported' },
+ { claim: '2,000,000 targeted (source).', verdict: 'unsupported' },
+ ]);
+
+ expect(output).toBe(
+ `- 1,000,000 people in need (source).\n- 2,000,000 targeted (source). ${UNVERIFIED_MARK}`,
+ );
+ });
+
+ it('leaves the section alone rather than mangling it when a claim cannot be located', () => {
+ const markdown = 'Body text.';
+ expect(annotate(markdown, [{ claim: 'not present here', verdict: 'unsupported' }])).toBe(markdown);
+ });
+});
diff --git a/app/src/lib/agent/verify.ts b/app/src/lib/agent/verify.ts
new file mode 100644
index 0000000..7c7ef99
--- /dev/null
+++ b/app/src/lib/agent/verify.ts
@@ -0,0 +1,430 @@
+/**
+ * The self-check: does the section actually say what the evidence supports?
+ *
+ * This is the step that makes the rest of the feature defensible. Everything
+ * before it — the tools, the per-section evidence, the citation ids — makes
+ * grounding *likely*. None of it makes grounding *checked*. A model handed six
+ * passages and asked for 200 words will occasionally produce a figure that is
+ * in none of them, and it will produce that figure in exactly the same
+ * confident register as the five that are, because a language model has no
+ * privileged access to which of its own sentences were retrieved.
+ *
+ * So every factual sentence is pulled back out of the draft and matched against
+ * the evidence that section was given. Three verdicts:
+ *
+ * - `supported` — the evidence contains it.
+ * - `unsupported` — the evidence covers the topic and does not say this, or
+ * says something else.
+ * - `unverifiable` — the evidence is silent; nothing here can settle it.
+ *
+ * The last two are marked in the prose, not removed. Deleting an unverified
+ * claim would produce a shorter document that reads as entirely verified, which
+ * is worse than the problem: the reader loses both the claim and the warning.
+ * A flagged sentence tells an analyst precisely where to look.
+ *
+ * # Why two passes
+ *
+ * The cheap pass is string and figure matching, and it settles most sentences
+ * for free. It cannot settle paraphrase — "roughly a quarter of the population"
+ * against "24.6%" — so what it cannot settle goes to one batched model call per
+ * section. One call, not one per claim: six sections times six claims would be
+ * thirty-six extra requests, which neither the token budget nor anyone's
+ * patience survives.
+ *
+ * The default when the check itself fails or returns nonsense is
+ * `unverifiable`, i.e. flag it. A verifier that fails open verifies nothing.
+ */
+
+import { generateText, type LanguageModel } from 'ai';
+
+import { getProviderOptions } from '@/lib/llm/provider';
+
+import { formatEvidence, type EvidenceItem } from './evidence';
+import { TokenPacer, estimateTokens, withRateLimitRetry } from './pacer';
+import { UNVERIFIED_MARK } from './render';
+import type { Verdict } from './types';
+
+export interface ClaimCheck {
+ claim: string;
+ verdict: Verdict;
+ /** Evidence label that supports it, where one does. */
+ source?: string;
+}
+
+export interface VerificationResult {
+ checks: ClaimCheck[];
+ /** The section with unsupported/unverifiable claims marked in place. */
+ markdown: string;
+ flagged: number;
+}
+
+/**
+ * Claims checked per section.
+ *
+ * Six covers a 200-word section's factual load with room over, and caps what
+ * the batched check costs. Sentences beyond the cap are left unmarked rather
+ * than marked unverified — claiming to have checked something that was never
+ * looked at is the one outcome worse than not checking.
+ */
+const MAX_CLAIMS = 6;
+
+/** Lexical overlap at which a claim and an evidence item are about one thing. */
+const TOPIC_OVERLAP = 0.34;
+
+const STOPWORDS = new Set([
+ 'about', 'above', 'after', 'against', 'among', 'around', 'because', 'been', 'before',
+ 'being', 'between', 'both', 'during', 'each', 'from', 'have', 'having', 'into', 'more',
+ 'most', 'other', 'over', 'same', 'some', 'such', 'than', 'that', 'their', 'them',
+ 'these', 'they', 'this', 'those', 'through', 'under', 'until', 'were', 'what', 'when',
+ 'where', 'which', 'while', 'with', 'within', 'would', 'there', 'also', 'across',
+]);
+
+/* ------------------------------------------------------------------ *
+ * Claim extraction
+ * ------------------------------------------------------------------ */
+
+/** Words that make a sentence a factual assertion rather than framing. */
+const FACTUAL_PATTERN =
+ /\b(sphere|chs|iasc|ipc|standard|indicator|commitment|threshold|minimum|litres|liters|per\s+person|per\s+day|per\s+capita|phase\s+\d|cluster|appeal|funded|funding|requirement|caseload|displaced|refugees?|idps?|malnutrition|mortality|coverage)\b/i;
+
+/**
+ * Pull the checkable sentences out of a section.
+ *
+ * Split on sentence boundaries and on list-item boundaries, since a brief's
+ * factual load is mostly in bullets and table rows rather than prose. Anything
+ * carrying a figure is checkable by definition; anything naming a standard or a
+ * humanitarian indicator is checkable too, because a wrong section number is as
+ * damaging as a wrong number and much harder to spot.
+ */
+/**
+ * Sentence boundaries, ignoring any that fall inside brackets.
+ *
+ * The naive split — a full stop, whitespace, then a capital — is wrong here in a
+ * way that showed up only on a real brief. Citations are rendered inline as
+ * `(sphere · Promotion > 2. Water supply > standard 2.1)`, and that label
+ * contains a full stop followed by a capital. Splitting there ends the "claim"
+ * halfway through its own source label, so `annotate` then inserted the
+ * unverified mark *inside* the citation, producing:
+ *
+ * …500 per hand pump (sphere · Promotion > 2. **[unverified]** Water supply…
+ *
+ * which is worse than not flagging at all: it corrupts the provenance the
+ * reader needs in order to check the flag. Tracking bracket depth costs one
+ * loop and removes the whole class.
+ */
+function splitSentences(text: string): string[] {
+ const sentences: string[] = [];
+ let depth = 0;
+ let start = 0;
+
+ for (let index = 0; index < text.length; index += 1) {
+ const char = text[index];
+ if (char === '(' || char === '[') depth += 1;
+ else if (char === ')' || char === ']') depth = Math.max(0, depth - 1);
+ else if (depth === 0 && (char === '.' || char === '!' || char === '?')) {
+ const rest = text.slice(index + 1);
+ const boundary = /^\s+[A-Z(]/.exec(rest);
+ if (boundary) {
+ sentences.push(text.slice(start, index + 1));
+ start = index + boundary[0].length;
+ index = start - 1;
+ }
+ }
+ }
+
+ sentences.push(text.slice(start));
+ return sentences;
+}
+
+export function extractClaims(markdown: string): string[] {
+ const claims: string[] = [];
+
+ for (const line of markdown.split('\n')) {
+ const text = line
+ .replace(/^\s*([-*+]|\d+[.)])\s+/, '') // list markers
+ .replace(/^\s*\|?\s*/, '') // table cell leader
+ .replace(/^#{1,6}\s+/, '') // stray heading
+ .trim();
+ if (!text || /^[-|:\s]+$/.test(text)) continue;
+
+ for (const sentence of splitSentences(text)) {
+ const claim = sentence.trim();
+ if (claim.length < 25) continue;
+ const hasFigure = /\d/.test(claim);
+ if (!hasFigure && !FACTUAL_PATTERN.test(claim)) continue;
+ claims.push(claim);
+ if (claims.length >= MAX_CLAIMS) return claims;
+ }
+ }
+
+ return claims;
+}
+
+/* ------------------------------------------------------------------ *
+ * Deterministic pass
+ * ------------------------------------------------------------------ */
+
+/**
+ * Digit runs with thousands separators removed, so a claim's "1,200,000"
+ * matches an evidence item's "1 200 000" or "1200000".
+ *
+ * The group pattern is deliberately narrow — a separator is consumed only when
+ * exactly three digits follow it. A looser `[\d,\s]*` reads "IPC Phase 3 4" as
+ * the number 34, which would then "match" evidence containing 34 of anything.
+ * Single digits are dropped for the same reason: they occur everywhere and so
+ * discriminate nothing.
+ */
+function figuresIn(text: string): string[] {
+ const found = text.match(/\d+(?:[,\u00a0\u202f ]\d{3})*(?:\.\d+)?/g) ?? [];
+ return found
+ .map((figure) => figure.replace(/[,\u00a0\u202f ]/g, ''))
+ .filter((figure) => figure.length >= 2);
+}
+
+function contentWords(text: string): Set {
+ const words = text.toLowerCase().match(/[a-z]{4,}/g) ?? [];
+ return new Set(words.filter((word) => !STOPWORDS.has(word)));
+}
+
+function overlap(claim: Set, evidence: Set): number {
+ if (claim.size === 0) return 0;
+ let shared = 0;
+ for (const word of claim) if (evidence.has(word)) shared += 1;
+ return shared / claim.size;
+}
+
+interface Prepared {
+ item: EvidenceItem;
+ figures: Set;
+ words: Set;
+}
+
+/**
+ * Settle a claim on string evidence alone, or return undefined to defer.
+ *
+ * A claim with figures is supported only by an item that contains all of them
+ * *and* is about the same subject — the conjunction matters, because "3" and
+ * "2018" turn up in unrelated passages constantly, and matching on figures
+ * alone would wave through a caseload attributed to the wrong dataset.
+ */
+function checkDeterministically(claim: string, prepared: Prepared[]): ClaimCheck | undefined {
+ const claimFigures = figuresIn(claim);
+ const claimWords = contentWords(claim);
+
+ // The draft cites by source label in parentheses; when that label names a
+ // real evidence item, check that item first and hardest.
+ for (const candidate of prepared) {
+ if (!claim.includes(`(${candidate.item.label})`)) continue;
+ const figuresHeld = claimFigures.every((figure) => candidate.figures.has(figure));
+ if (figuresHeld && overlap(claimWords, candidate.words) >= TOPIC_OVERLAP) {
+ return { claim, verdict: 'supported', source: candidate.item.label };
+ }
+ }
+
+ for (const candidate of prepared) {
+ if (claimFigures.length === 0) continue;
+ const figuresHeld = claimFigures.every((figure) => candidate.figures.has(figure));
+ if (figuresHeld && overlap(claimWords, candidate.words) >= TOPIC_OVERLAP) {
+ return { claim, verdict: 'supported', source: candidate.item.label };
+ }
+ }
+
+ return undefined;
+}
+
+/* ------------------------------------------------------------------ *
+ * Model pass
+ * ------------------------------------------------------------------ */
+
+const VERDICTS: Record = {
+ supported: 'supported',
+ unsupported: 'unsupported',
+ unverifiable: 'unverifiable',
+};
+
+/**
+ * Evidence budget for the batched check.
+ *
+ * Deliberately well under the draft step's, because the check does not need
+ * what the draft needed. The draft was given the section's whole evidence set
+ * because it had to decide what to write from it; the check is asking a narrow
+ * question about a handful of sentences that the cheap pass could not settle,
+ * and every item unrelated to those sentences is budget spent to no effect.
+ *
+ * That mattered more than it looks. Verify ran once per section on the same
+ * evidence the draft had just been sent, so a brief was paying for its evidence
+ * twice — and the second time was the call most likely to be the one that tipped
+ * a section over the endpoint's ceiling, because it landed immediately after the
+ * draft rather than after a pause.
+ */
+export const VERIFY_EVIDENCE_CHARS = 1_800;
+
+/**
+ * The evidence items that bear on the claims still in question.
+ *
+ * Ranked by the same lexical overlap the deterministic pass uses, so an item is
+ * "relevant" here on exactly the criterion that would have let it settle a claim
+ * there. Falls back to the original order when nothing overlaps — a claim whose
+ * subject appears nowhere in the evidence is precisely the one that should come
+ * back `unverifiable`, and it needs some evidence in front of the model to be
+ * judged that rather than none.
+ */
+function relevantEvidence(claims: string[], evidence: EvidenceItem[]): EvidenceItem[] {
+ if (evidence.length === 0) return evidence;
+ const wanted = contentWords(claims.join(' '));
+
+ const scored = evidence.map((item, index) => ({
+ item,
+ index,
+ score: overlap(contentWords(`${item.label} ${item.text}`), wanted),
+ }));
+
+ scored.sort((left, right) => right.score - left.score || left.index - right.index);
+ return scored.map((entry) => entry.item);
+}
+
+async function checkWithModel(
+ claims: string[],
+ evidence: EvidenceItem[],
+ ctx: { model: LanguageModel; pacer: TokenPacer; signal?: AbortSignal },
+): Promise