diff --git a/benchmarks/genui-formats/.gitignore b/benchmarks/genui-formats/.gitignore index e1eddf9..6df95fa 100644 --- a/benchmarks/genui-formats/.gitignore +++ b/benchmarks/genui-formats/.gitignore @@ -2,3 +2,5 @@ node_modules results/raw/ samples/ +vendor/openui-prompt-*.txt +vendor/openui-schema-0.13.5.json diff --git a/benchmarks/genui-formats/README.md b/benchmarks/genui-formats/README.md index 3a768a7..cb56c15 100644 --- a/benchmarks/genui-formats/README.md +++ b/benchmarks/genui-formats/README.md @@ -6,7 +6,7 @@ and **validated by its own validator**, across a ladder of models from flagship | Format | Wire format | Prompt source | | --- | --- | --- | | [MDMA](https://github.com/MobileReality/mr-mdma) | Markdown + YAML in ` ```mdma ` blocks | `buildSystemPrompt()` + per-model author variant | -| [OpenUI Lang](https://github.com/thesysdev/openui) | line-oriented DSL | `benchmarks/system-prompt.txt` (their own published artifact) | +| [OpenUI Lang](https://github.com/thesysdev/openui) | line-oriented DSL | `openuiLibrary.prompt()` from `@openuidev/react-ui@0.13.5` | | [json-render](https://github.com/vercel-labs/json-render) | JSONL of RFC-6902 patches | `catalog.prompt()` | | [AGenUI](https://github.com/AGenUI/AGenUI) | A2UI v0.9 JSON messages | `skills/a2ui-generation/SKILL.md` + required refs, flattened | | [A2UI](https://github.com/a2ui-project/a2ui) | A2UI v0.9 JSON messages (transport) | `a2ui_agent` Python prompt generator, standard v0.9 `basic` catalog | @@ -77,6 +77,19 @@ vendor/ pinned upstream artifacts + PINS.txt (commit SHAs) results/ ledger, raw generations, results.json, REPORT.md ``` +## A2UI arm setup (one-off) + +The A2UI arm validates through A2UI's own Python SDK, which is not published to +PyPI. Install it from the pinned commit before running or scoring that arm: + +```bash +./a2ui-python/setup.sh # clones the pinned commit, builds a venv +``` + +The checkout and venv default to `/tmp`, so the OS will eventually delete them; +re-run the script when `pnpm verify` reports a missing interpreter. Set +`A2UI_SRC`, `A2UI_VENV` and `A2UI_PYTHON` to keep them somewhere durable. + ## Running ```bash diff --git a/benchmarks/genui-formats/a2ui-python/setup.sh b/benchmarks/genui-formats/a2ui-python/setup.sh new file mode 100755 index 0000000..e543381 --- /dev/null +++ b/benchmarks/genui-formats/a2ui-python/setup.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Rebuild the Python environment the A2UI arm needs. +# +# The A2UI adapter shells out to a2ui_tool.py, which imports the A2UI project's +# own SDK. That SDK is not on PyPI, so it is installed from a pinned commit of +# a2ui-project/a2ui. Both the checkout and the venv live under /tmp by default, +# which means they disappear whenever the OS cleans temporary files. Re-run this +# script when `pnpm verify` reports a spawn failure for the A2UI adapter. +# +# ./a2ui-python/setup.sh +# +# Override locations with A2UI_SRC / A2UI_VENV if you want them somewhere +# durable, and set A2UI_PYTHON to the resulting interpreter so the adapter +# finds it: +# +# A2UI_SRC=~/.cache/a2ui-src A2UI_VENV=~/.cache/a2ui-venv ./a2ui-python/setup.sh +# export A2UI_PYTHON=~/.cache/a2ui-venv/bin/python + +set -euo pipefail + +# The commit the published results were generated against. Later commits added +# an ANTLR build dependency that this arm does not need, so the pin is +# deliberate rather than incidental. +A2UI_COMMIT="${A2UI_COMMIT:-349c97909572c85111b6201c4e46169ae0277d9b}" +A2UI_SRC="${A2UI_SRC:-/tmp/a2ui-src}" +A2UI_VENV="${A2UI_VENV:-/tmp/a2ui-venv}" + +echo "A2UI commit : $A2UI_COMMIT" +echo "checkout : $A2UI_SRC" +echo "venv : $A2UI_VENV" + +rm -rf "$A2UI_SRC" "$A2UI_VENV" +mkdir -p "$A2UI_SRC" + +# Shallow single-commit fetch: the full history is large and unnecessary. +git -C "$A2UI_SRC" init -q +git -C "$A2UI_SRC" remote add origin https://github.com/a2ui-project/a2ui.git +git -C "$A2UI_SRC" fetch -q --depth 1 origin "$A2UI_COMMIT" +git -C "$A2UI_SRC" checkout -q FETCH_HEAD + +# a2ui_agent's build reads ../../specification, so both packages install from +# the repo layout rather than in isolation. +python3 -m venv "$A2UI_VENV" +"$A2UI_VENV/bin/pip" install -q \ + "$A2UI_SRC/agent_sdks/python/a2ui_core" \ + "$A2UI_SRC/agent_sdks/python/a2ui_agent" + +"$A2UI_VENV/bin/python" -c "import a2ui" && echo "OK: A2UI SDK installed" + +# The adapter also reads the standard v0.9 catalog straight from the checkout. +test -f "$A2UI_SRC/specification/v0_9/catalogs/basic/catalog.json" \ + && echo "OK: v0.9 basic catalog present" diff --git a/benchmarks/genui-formats/package.json b/benchmarks/genui-formats/package.json index 2a71bbe..1dc3335 100644 --- a/benchmarks/genui-formats/package.json +++ b/benchmarks/genui-formats/package.json @@ -20,6 +20,7 @@ "@mobile-reality/mdma-prompt-pack": "workspace:*", "@mobile-reality/mdma-validator": "workspace:*", "@openuidev/lang-core": "0.2.10", + "@openuidev/react-ui": "0.13.5", "ajv": "^8.17.1", "dotenv": "^16.4.5", "openai": "^4.77.0", diff --git a/benchmarks/genui-formats/results/REPORT.md b/benchmarks/genui-formats/results/REPORT.md index 40c252b..2d0d3f7 100644 --- a/benchmarks/genui-formats/results/REPORT.md +++ b/benchmarks/genui-formats/results/REPORT.md @@ -1,6 +1,6 @@ # Generative-UI format reliability benchmark -_Generated 2026-07-28T07:26:52.650Z · 1377 generations · 0 API errors_ +_Generated 2026-08-07T14:19:12.249Z · 1647 generations · 0 API errors_ Five open-source generative-UI formats, each generated natively from its own published prompt and validated by its own validator, across a ladder of models. @@ -21,9 +21,9 @@ They score very differently, which is why they are separate columns. | Rung | Model | Provider id | Generations | | --- | --- | --- | --- | -| flagship | Opus 5 | `anthropic/claude-opus-5` | 450 | -| mid | GPT-5.6-terra | `openai/gpt-5.6-terra` | 450 | -| small | Gemma-4-26B-A4B | `google/gemma-4-26b-a4b-it` | 450 | +| flagship | Opus 5 | `anthropic/claude-opus-5` | 540 | +| mid | GPT-5.6-terra | `openai/gpt-5.6-terra` | 540 | +| small | Gemma-4-26B-A4B | `google/gemma-4-26b-a4b-it` | 540 | All reached through OpenRouter, 5 repeats per scenario, temperature 0.7, max_tokens 8192 — identical settings for every format and model. @@ -40,7 +40,7 @@ rendered — because a format that works four times in five is not something you | Format | Opus 5 (flagship) | Gemma-4-26B (open weights) | Drop | | --- | --- | --- | --- | | MDMA | 94.4% | 94.4% | 0.0pp | -| OpenUI Lang | 83.3% | 55.6% | 27.8pp | +| OpenUI Lang | 94.4% | 50.0% | 44.4pp | | json-render | 83.3% | 38.9% | 44.4pp | | A2UI (transport) | 83.3% | 44.4% | 38.9pp | | AGenUI (A2UI SDK) | 38.9% \* | 77.8% | N/A | @@ -91,8 +91,8 @@ Prompt size matters here — you pay it on every request: | Format | Prompt tokens ↓ | | --- | --- | -| OpenUI Lang | 5172 | | MDMA | 5910 | +| OpenUI Lang | 6978 | | json-render | 8466 | | A2UI (transport) | 16085 | | AGenUI (A2UI SDK) | 19689 | @@ -125,7 +125,7 @@ Every metric below is marked **↑ higher is better** or **↓ lower is better** | Formats | 5 | | Models | 3 | | Repeats per scenario | 5 | -| Total generations | 1377 | +| Total generations | 1647 | | Temperature | 0.7, identical for every format and model | | Max tokens | 8192, identical | @@ -137,7 +137,7 @@ artifact or prompt-generation API. | Format | Source | Prompt tokens (↓ lower is better) | | --- | --- | --- | | MDMA | buildSystemPrompt() + per-model author variant (packages/prompt-pack) | 5910 | -| OpenUI Lang | vendor/openui-system-prompt.txt (thesysdev/openui @ 65b5f93) | 5172 | +| OpenUI Lang | openuiLibrary.prompt() from @openuidev/react-ui@0.13.5 | 6978 | | json-render | catalog.prompt() — @json-render/core 0.19.0, shadcn catalog + Chart | 8466 | | A2UI (transport) | a2ui_agent TransportFormat.prompt_generator.generate() — A2UI's own Python SDK | 16085 | | AGenUI (A2UI SDK) | flattened AGenUI skills/a2ui-generation SKILL.md + refs (AGenUI @ 3e79bea) | 19689 | @@ -149,11 +149,11 @@ Share of generations that parse and validate — i.e. that a renderer could rend | Model | MDMA | OpenUI Lang | json-render | A2UI (transport) | AGenUI (A2UI SDK) | | --- | --- | --- | --- | --- | --- | | **flagship** | | | | | | -| Opus 5 | 100.0% | 98.8% | 100.0% | 96.4% | 100.0% | +| Opus 5 | 100.0% | 100.0% | 100.0% | 96.4% | 100.0% | | **mid** | | | | | | -| GPT-5.6-terra | 91.1% | 100.0% | 93.3% | 86.7% | 81.1% | +| GPT-5.6-terra | 91.1% | 98.9% | 93.3% | 86.7% | 81.1% | | **small** | | | | | | -| Gemma-4-26B-A4B | 98.9% | 83.3% | 83.3% | 81.1% | 93.3% | +| Gemma-4-26B-A4B | 98.9% | 78.9% | 83.3% | 81.1% | 93.3% | > ⚠️ **AGenUI's column is measured on a looser standard than AGenUI's own tooling applies.** Every > format here is checked for structural renderability, but AGenUI additionally ships an 889-line @@ -173,7 +173,7 @@ It is still a real cost. Truncation here means: at 8k output tokens, that format | Model | MDMA | OpenUI Lang | json-render | A2UI (transport) | AGenUI (A2UI SDK) | | --- | --- | --- | --- | --- | --- | | **flagship** | | | | | | -| Opus 5 | 4.4% | 8.9% | 12.2% | 7.8% | 52.2% | +| Opus 5 | 4.4% | 5.6% | 12.2% | 7.8% | 52.2% | | **mid** | | | | | | | GPT-5.6-terra | 0.0% | 0.0% | 0.0% | 0.0% | 0.0% | | **small** | | | | | | @@ -187,11 +187,11 @@ you are shipping a product: a format that works 4 times in 5 still breaks in pro | Model | MDMA | OpenUI Lang | json-render | A2UI (transport) | AGenUI (A2UI SDK) | | --- | --- | --- | --- | --- | --- | | **flagship** | | | | | | -| Opus 5 | 94.4% | 83.3% | 83.3% | 83.3% | 38.9% | +| Opus 5 | 94.4% | 94.4% | 83.3% | 83.3% | 38.9% | | **mid** | | | | | | -| GPT-5.6-terra | 83.3% | 100.0% | 72.2% | 61.1% | 72.2% | +| GPT-5.6-terra | 83.3% | 94.4% | 72.2% | 61.1% | 72.2% | | **small** | | | | | | -| Gemma-4-26B-A4B | 94.4% | 55.6% | 38.9% | 44.4% | 77.8% | +| Gemma-4-26B-A4B | 94.4% | 50.0% | 38.9% | 44.4% | 77.8% | ## 3. Shape stability — ↑ higher is better (diagnostic only — NOT comparable across formats) @@ -212,7 +212,7 @@ Share of scenarios where all 5 repeats produced the same component structure. | **flagship** | | | | | | | Opus 5 | 16.7% | 5.6% | 5.6% | 5.6% | 5.6% | | **mid** | | | | | | -| GPT-5.6-terra | 72.2% | 5.6% | 5.6% | 11.1% | 5.6% | +| GPT-5.6-terra | 72.2% | 22.2% | 5.6% | 11.1% | 5.6% | | **small** | | | | | | | Gemma-4-26B-A4B | 72.2% | 5.6% | 0.0% | 5.6% | 5.6% | @@ -226,29 +226,29 @@ Cheap output nobody can render is not cheap. | Model | MDMA | OpenUI Lang | json-render | A2UI (transport) | AGenUI (A2UI SDK) | | --- | --- | --- | --- | --- | --- | | **flagship** | | | | | | -| Opus 5 | 1666 | 1814 | 4019 | 3892 | 7209 | +| Opus 5 | 1666 | 2125 | 4019 | 3892 | 7209 | | **mid** | | | | | | -| GPT-5.6-terra | 554 | 567 | 1020 | 1081 | 1674 | +| GPT-5.6-terra | 554 | 624 | 1020 | 1081 | 1674 | | **small** | | | | | | -| Gemma-4-26B-A4B | 448 | 417 | 836 | 1300 | 1485 | +| Gemma-4-26B-A4B | 448 | 525 | 836 | 1300 | 1485 | **Efficiency — ↑ higher is better (renderable output per 1k output tokens):** | Model | MDMA | OpenUI Lang | json-render | A2UI (transport) | AGenUI (A2UI SDK) | | --- | --- | --- | --- | --- | --- | | **flagship** | | | | | | -| Opus 5 | 0.60 | 0.54 | 0.25 | 0.25 | 0.14 | +| Opus 5 | 0.60 | 0.47 | 0.25 | 0.25 | 0.14 | | **mid** | | | | | | -| GPT-5.6-terra | 1.64 | 1.77 | 0.91 | 0.80 | 0.48 | +| GPT-5.6-terra | 1.64 | 1.59 | 0.91 | 0.80 | 0.48 | | **small** | | | | | | -| Gemma-4-26B-A4B | 2.21 | 2.00 | 1.00 | 0.62 | 0.63 | +| Gemma-4-26B-A4B | 2.21 | 1.50 | 1.00 | 0.62 | 0.63 | ## 5. Failure taxonomy — ↓ lower is better (0 is perfect) | Format | broken-reference | no-structured-output | off-task | parse-error | prose-leakage | schema-error | truncated | unknown-component | | --- | --- | --- | --- | --- | --- | --- | --- | --- | | MDMA | 0 | 6 | 2 | 1 | 0 | 0 | 4 | 0 | -| OpenUI Lang | 142 | 0 | 0 | 0 | 0 | 57 | 8 | 1 | +| OpenUI Lang | 38 | 0 | 0 | 0 | 0 | 55 | 7 | 5 | | json-render | 27 | 0 | 0 | 0 | 0 | 26 | 11 | 0 | | A2UI (transport) | 0 | 0 | 0 | 32 | 0 | 0 | 7 | 0 | | AGenUI (A2UI SDK) | 3 | 13 | 7 | 0 | 2 | 0 | 47 | 0 | @@ -257,19 +257,19 @@ Cheap output nobody can render is not cheap. | Family | MDMA | OpenUI Lang | json-render | A2UI (transport) | AGenUI (A2UI SDK) | | --- | --- | --- | --- | --- | --- | -| contact-form | 100.0% | 100.0% | 93.3% | 82.2% | 77.8% | -| data-table | 82.2% | 86.7% | 82.2% | 66.7% | 66.7% | -| chart | 97.8% | 93.3% | 95.6% | 95.6% | 80.0% | -| callout | 100.0% | 91.1% | 95.6% | 95.6% | 93.3% | +| contact-form | 100.0% | 91.1% | 93.3% | 82.2% | 77.8% | +| data-table | 82.2% | 82.2% | 82.2% | 66.7% | 66.7% | +| chart | 97.8% | 91.1% | 95.6% | 95.6% | 80.0% | +| callout | 100.0% | 97.8% | 95.6% | 95.6% | 93.3% | | button | 100.0% | 97.8% | 88.9% | 100.0% | 88.9% | -| tasklist | 91.1% | 77.8% | 73.3% | 73.3% | 37.8% | +| tasklist | 91.1% | 84.4% | 73.3% | 73.3% | 37.8% | **By variant** — the adversarial variant is where formats separate: | Variant | MDMA | OpenUI Lang | json-render | A2UI (transport) | AGenUI (A2UI SDK) | | --- | --- | --- | --- | --- | --- | -| minimal | 100.0% | 93.3% | 95.6% | 92.2% | 87.8% | -| realistic | 98.9% | 97.8% | 90.0% | 87.8% | 71.1% | +| minimal | 100.0% | 96.7% | 95.6% | 92.2% | 87.8% | +| realistic | 98.9% | 93.3% | 90.0% | 87.8% | 71.1% | | adversarial | 86.7% | 82.2% | 78.9% | 76.7% | 63.3% | ## 7. Does adding few-shot examples help a weak model? — no @@ -289,6 +289,41 @@ asymmetry disadvantaged A2UI, we re-ran the same 18 scenarios x 5 repeats on Gem `` block at all — the model answered in prose — and parse errors rose from 17 to 27. At this model size the instruction gets lost in the context rather than reinforced. +## 8. OpenUI's published prompt is stale against the library it ships + +thesysdev publish `benchmarks/system-prompt.txt` and use it in their own benchmark. It is +byte-identical to their repository today, but it no longer matches the library they ship on +npm: `openuiLibrary.prompt()` from `@openuidev/react-ui@0.13.5` produces a different prompt. + +| | Published artifact | Shipped library v0.13.5 | +| --- | --- | --- | +| Size | 13,080 chars | 17,431 chars | +| `## Examples` | present | **removed** | +| `## Action - Button Behavior` | absent | **added** | +| `## Final Verification` | absent | **added** | +| Components | base set | **+Modal, +Action** | + +Anyone installing the package today gets the second one, so **the scored OpenUI arm uses the +shipped library prompt**, with the schema regenerated from the same version so parser and +prompt agree. We re-ran all 270 generations rather than adjusting anything after the fact. + +The superseded run is kept here because the comparison is informative: + +| Model | Published artifact | Shipped library | Delta | +| --- | --- | --- | --- | +| Opus 5 | 83.3% | 94.4% | +11.1pp | +| GPT-5.6-terra | 100.0% | 94.4% | -5.6pp | +| Gemma-4-26B | 55.6% | 50.0% | -5.6pp | + +"Every time" rate, five repeats per scenario. The correction gained 2 scenarios on the +flagship and lost 1 on each cheaper model, averaging out to no change across the ladder. +Correcting it was therefore about accuracy, not about advantage: the published artifact was +not systematically unfair to OpenUI, and scored better on two rungs of three. + +The direction is consistent with section 7. The shipped prompt trades worked examples for +prose rules; a strong model applies rules, a weak one needs the examples. On Gemma the +largest failure category became `missing-required`, which is what losing examples predicts. + ## Method and fairness - Every format uses **its own published prompt, unmodified**, from its own official artifact diff --git a/benchmarks/genui-formats/results/generations.jsonl b/benchmarks/genui-formats/results/generations.jsonl index 822d974..4043ddf 100644 --- a/benchmarks/genui-formats/results/generations.jsonl +++ b/benchmarks/genui-formats/results/generations.jsonl @@ -1375,3 +1375,273 @@ {"model": "google/gemma-4-26b-a4b-it", "format": "a2ui", "scenario": "tasklist/adversarial", "repeat": 4, "output": "I will create a structured deployment guide interface for you. This UI will use a vertical timeline-style layout where each step is clearly delineated, providing the command, the \"Why it matters\" explanation, and the \"Consequences of skipping\" warning.\n\n\n{\n \"version\": \"v0.9\",\n \"createSurface\": {\n \"surfaceId\": \"deployment_guide_surface\",\n \"catalogId\": \"https://example.com/deployment-docs\",\n \"theme\": {\n \"primaryColor\": \"#2563eb\"\n }\n }\n}\n\n\n\n{\n \"version\": \"v0.9\",\n \"updateComponents\": {\n \"surfaceId\": \"deployment_guide_surface\",\n \"components\": [\n {\n \"id\": \"root\",\n \"component\": \"Column\",\n \"children\": [\n \"header_container\",\n \"steps_list\"\n ],\n \"align\": \"stretch\"\n },\n {\n \"id\": \"header_container\",\n \"component\": \"Column\",\n \"children\": [\n \"title\",\n \"subtitle\"\n ],\n \"align\": \"start\"\n },\n {\n \"id\": \"title\",\n \"component\": \"Text\",\n \"text\": \"# Deployment Protocol\",\n \"variant\": \"h1\"\n },\n {\n \"id\": \"subtitle\",\n \"component\": \"Text\",\n \"text\": \"Follow these steps precisely to ensure a stable production environment.\",\n \"variant\": \"body\"\n },\n {\n \"id\": \"steps_list\",\n \"component\": \"Column\",\n \"children\": [\n \"step_1_card\",\n \"step_2_card\",\n \"step_3_card\"\n ],\n \"align\": \"stretch\"\n },\n {\n \"id\": \"step_1_card\",\n \"component\": \"Card\",\n \"child\": \"step_1_content\"\n },\n {\n \"id\": \"step_1_content\",\n \"component\": \"Column\",\n \"children\": [\n \"step_1_title\",\n \"step_1_cmd\",\n \"step_1_divider\",\n \"step_1_why\",\n \"step_1_skip\"\n ],\n \"align\": \"stretch\"\n },\n {\n \"id\": \"step_1_title\",\n \"component\": \"Text\",\n \"text\": \"Step 1: Environment Configuration\",\n \"variant\": \"h3\"\n },\n {\n \"id\": \"step_1_cmd\",\n \"component\": \"TextField\",\n \"label\": \"Command to run\",\n \"value\": \"npm install && cp .env.example .env\",\n \"variant\": \"shortText\"\n },\n {\n \"id\": \"step_1_divider\",\n \"component\": \"Divider\",\n \"axis\": \"horizontal\"\n },\n {\n \"id\": \"step_1_why\",\n \"component\": \"Text\",\n \"text\": \"**Why it matters:** Ensures all necessary dependencies are present and your application has the correct secrets/keys to communicate with databases and APIs.\\n\\n**If you skip it:** The build will fail immediately due to missing modules, or the app will crash on startup because it cannot find sensitive credentials.\",\n \"variant\": \"body\"\n },\n {\n \"id\": \"step_1_skip\",\n \"component\": \"Text\",\n \"text\": \"\u26a0\ufe0f Critical Step\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"step_2_card\",\n \"component\": \"Card\",\n \"child\": \"step_2_content\"\n },\n {\n \"id\": \"step_2_content\",\n \"component\": \"Column\",\n \"children\": [\n \"step_2_title\",\n \"step_2_cmd\",\n \"step_2_divider\",\n \"step_2_why\",\n \"step_2_skip\"\n ],\n \"align\": \"stretch\"\n },\n {\n \"id\": \"step_2_title\",\n \"component\": \"Text\",\n \"text\": \"Step 2: Build Artifact Generation\",\n \"variant\": \"h3\"\n },\n {\n \"id\": \"step_2_cmd\",\n \"component\": \"TextField\",\n \"label\": \"Command to run\",\n \"value\": \"npm run build\",\n \"variant\": \"shortText\"\n },\n {\n \"id\": \"step_2_divider\",\n \"component\": \"Divider\",\n \"axis\": \"horizontal\"\n },\n {\n \"id\": \"step_2_why\",\n \"component\": \"Text\",\n \"text\": \"**Why it matters:** Compiles, minifies, and optimizes your source code into highly efficient production-ready files.\\n\\n**If you skip it:** You will be attempting to serve raw development code, which is significantly slower, insecure, and likely contains debugging tools that should never reach production.\",\n \"variant\": \"body\"\n },\n {\n \"id\": \"step_2_skip\",\n \"component\": \"Text\",\n \"text\": \"\u26a0\ufe0f Critical Step\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"step_3_card\",\n \"component\": \"Card\",\n \"child\": \"step_3_content\"\n },\n {\n \"id\": \"step_3_content\",\n \"component\": \"Column\",\n \"children\": [\n \"step_3_title\",\n \"step_3_cmd\",\n \"step_3_divider\",\n \"step_3_why\",\n \"step_3_skip\"\n ],\n \"align\": \"stretch\"\n },\n {\n \"id\": \"step_3_title\",\n \"component\": \"Text\",\n \"text\": \"Step 3: Deployment to Server\",\n \"variant\": \"h3\"\n },\n {\n \"id\": \"step_3_cmd\",\n \"component\": \"TextField\",\n \"label\": \"Command to run\",\n \"value\": \"docker compose up -d\",\n \"variant\": \"shortText\"\n },\n {\n \"id\": \"step_3_divider\",\n \"component\": \"Divider\",\n \"axis\": \"horizontal\"\n },\n {\n \"id\": \"step_3_why\",\n \"component\": \"Text\",\n \"text\": \"**Why it matters:** This is the actual launch. It spins up the containers and makes your application accessible to the world.\\n\\n**If you skip it:** Your code is sitting idle on the server but is not actually running or reachable by users.\",\n \"variant\": \"body\"\n },\n {\n \"id\": \"step_3_skip\",\n \"component\": \"Text\",\n \"text\": \"\u26a0\ufe0f Final Step\",\n \"variant\": \"caption\"\n }\n ]\n }\n}\n", "promptTokens": 10278, "completionTokens": 1688, "latencyMs": 163306, "finishReason": "stop"} {"model": "google/gemma-4-26b-a4b-it", "format": "a2ui", "scenario": "chart/minimal", "repeat": 0, "output": "I cannot directly render a graphical canvas-based bar chart (like a `` or SVG chart) using the current A2UI component set, as the catalog focuses on standard UI primitives (Layouts, Inputs, Text, etc.). \n\nHowever, I can create a **highly effective visual representation** of your monthly revenue using a combination of `Row`, `Column`, and `Slider` components to simulate a bar chart. This approach provides a clean, structured, and responsive \"data visualization\" dashboard.\n\n\n{\n \"version\": \"v0.9\",\n \"createSurface\": {\n \"surfaceId\": \"revenue_chart_surface\",\n \"catalogId\": \"https://a2ui.org/specification/v0_9/catalogs/basic/catalog.json\",\n \"theme\": {\n \"primaryColor\": \"#3b82f6\"\n }\n }\n}\n\n\n\n{\n \"version\": \"v0.9\",\n \"updateComponents\": {\n \"surfaceId\": \"revenue_chart_surface\",\n \"components\": [\n {\n \"id\": \"root\",\n \"component\": \"Column\",\n \"children\": [\n \"header_section\",\n \"chart_container\",\n \"legend_section\"\n ],\n \"align\": \"stretch\"\n },\n {\n \"id\": \"header_section\",\n \"component\": \"Column\",\n \"children\": [\n \"title_text\",\n \"subtitle_text\"\n ],\n \"align\": \"start\"\n },\n {\n \"id\": \"title_text\",\n \"component\": \"Text\",\n \"text\": \"Monthly Revenue\",\n \"variant\": \"h1\"\n },\n {\n \"id\": \"subtitle_text\",\n \"component\": \"Text\",\n \"text\": \"Last 6 months performance overview\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"chart_container\",\n \"component\": \"Card\",\n \"child\": \"chart_row\"\n },\n {\n \"id\": \"chart_row\",\n \"component\": \"Row\",\n \"justify\": \"spaceEvenly\",\n \"align\": \"end\",\n \"children\": [\n \"bar_month_1\",\n \"bar_month_2\",\n \"bar_month_3\",\n \"bar_month_4\",\n \"bar_month_5\",\n \"bar_month_6\"\n ]\n },\n {\n \"id\": \"bar_month_1\",\n \"component\": \"Column\",\n \"align\": \"center\",\n \"children\": [\n \"val_1\",\n \"bar_1\",\n \"label_1\"\n ]\n },\n {\n \"id\": \"val_1\",\n \"component\": \"Text\",\n \"text\": \"$12,400\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_1\",\n \"component\": \"Slider\",\n \"value\": 60,\n \"max\": 100\n },\n {\n \"id\": \"label_1\",\n \"component\": \"Text\",\n \"text\": \"Jan\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_month_2\",\n \"component\": \"Column\",\n \"align\": \"center\",\n \"children\": [\n \"val_2\",\n \"bar_2\",\n \"label_2\"\n ]\n },\n {\n \"id\": \"val_2\",\n \"component\": \"Text\",\n \"text\": \"$15,200\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_2\",\n \"component\": \"Slider\",\n \"value\": 75,\n \"max\": 100\n },\n {\n \"id\": \"label_2\",\n \"component\": \"Text\",\n \"text\": \"Feb\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_month_3\",\n \"component\": \"Column\",\n \"align\": \"center\",\n \"children\": [\n \"val_3\",\n \"bar_3\",\n \"label_3\"\n ]\n },\n {\n \"id\": \"val_3\",\n \"component\": \"Text\",\n \"text\": \"$11,000\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_3\",\n \"component\": \"Slider\",\n \"value\": 55,\n \"max\": 100\n },\n {\n \"id\": \"label_3\",\n \"component\": \"Text\",\n \"text\": \"Mar\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_month_4\",\n \"component\": \"Column\",\n \"align\": \"center\",\n \"children\": [\n \"val_4\",\n \"bar_4\",\n \"label_4\"\n ]\n },\n {\n \"id\": \"val_4\",\n \"component\": \"Text\",\n \"text\": \"$18,900\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_4\",\n \"component\": \"Slider\",\n \"value\": 90,\n \"max\": 100\n },\n {\n \"id\": \"label_4\",\n \"component\": \"Text\",\n \"text\": \"Apr\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_month_5\",\n \"component\": \"Column\",\n \"align\": \"center\",\n \"children\": [\n \"val_5\",\n \"bar_5\",\n \"label_5\"\n ]\n },\n {\n \"id\": \"val_5\",\n \"component\": \"Text\",\n \"text\": \"$22,000\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_5\",\n \"component\": \"Slider\",\n \"value\": 100,\n \"max\": 100\n },\n {\n \"id\": \"label_5\",\n \"component\": \"Text\",\n \"text\": \"May\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_month_6\",\n \"component\": \"Column\",\n \"align\": \"center\",\n \"children\": [\n \"val_6\",\n \"bar_6\",\n \"label_6\"\n ]\n },\n {\n \"id\": \"val_6\",\n \"component\": \"Text\",\n \"text\": \"$19,500\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"bar_6\",\n \"component\": \"Slider\",\n \"value\": 85,\n \"max\": 100\n },\n {\n \"id\": \"label_6\",\n \"component\": \"Text\",\n \"text\": \"Jun\",\n \"variant\": \"caption\"\n },\n {\n \"id\": \"legend_section\",\n \"component\": \"Row\",\n \"justify\": \"center\",\n \"children\": [\n \"legend_icon\",\n \"legend_text\"\n ]\n },\n {\n \"id\": \"legend_icon\",\n \"component\": \"Icon\",\n \"name\": \"info\"\n },\n {\n \"id\": \"legend_text\",\n \"component\": \"Text\",\n \"text\": \"Data updated as of today\",\n \"variant\": \"caption\"\n }\n ]\n }\n}\n", "promptTokens": 10262, "completionTokens": 1894, "latencyMs": 30435, "finishReason": "stop"} {"model": "google/gemma-4-26b-a4b-it", "format": "a2ui", "scenario": "chart/realistic", "repeat": 0, "output": "I cannot render a functional pie chart directly as the current A2UI catalog provides layout and input components rather than specialized data visualization (charting) components. However, I can create a highly structured, visually clear **Dashboard Summary** using a list of progress-style components that represent the traffic breakdown.\n\nHere is a breakdown of your website traffic for last month:\n\n\n{\n \"version\": \"v0.9\",\n \"createSurface\": {\n \"surfaceId\": \"traffic_breakdown_dashboard\",\n \"catalogId\": \"https://a2ui.org/specification/v0_9/catalogs/basic/catalog.json\",\n \"theme\": {\n \"primaryColor\": \"#4F46E5\"\n }\n }\n}\n\n\n\n{\n \"version\": \"v0.9\",\n \"updateComponents\": {\n \"surfaceId\": \"traffic_breakdown_dashboard\",\n \"components\": [\n {\n \"id\": \"root\",\n \"component\": \"Column\",\n \"children\": [\n \"header_section\",\n \"stats_container\"\n ],\n \"align\": \"stretch\"\n },\n {\n \"id\": \"header_section\",\n \"component\": \"Column\",\n \"children\": [\n \"title_text\",\n \"subtitle_text\"\n ],\n \"align\": \"start\"\n },\n {\n \"id\": \"title_text\",\n \"component\": \"Text\",\n \"text\": \"Website Traffic Breakdown\",\n \"variant\": \"h1\"\n },\n {\n \"id\": \"subtitle_text\",\n \"component\": \"Text\",\n \"text\": \"Last Month's Acquisition Channels\",\n \"variant\": \"body\"\n },\n {\n \"id\": \"stats_container\",\n \"component\": \"Column\",\n \"children\": [\n \"organic_row\",\n \"paid_row\",\n \"social_row\",\n \"direct_row\",\n \"referral_row\"\n ],\n \"align\": \"stretch\"\n },\n {\n \"id\": \"organic_row\",\n \"component\": \"Row\",\n \"children\": [\n \"organic_label\",\n \"organic_bar\"\n ],\n \"justify\": \"spaceBetween\"\n },\n {\n \"id\": \"organic_label\",\n \"component\": \"Text\",\n \"text\": \"\ud83d\udd0d Organic Search (45%)\"\n },\n {\n \"id\": \"organic_bar\",\n \"component\": \"Slider\",\n \"label\": \"Organic\",\n \"value\": 45,\n \"max\": 100\n },\n {\n \"id\": \"paid_row\",\n \"component\": \"Row\",\n \"children\": [\n \"paid_label\",\n \"paid_bar\"\n ],\n \"justify\": \"spaceBetween\"\n },\n {\n \"id\": \"paid_label\",\n \"component\": \"Text\",\n \"text\": \"\ud83d\udcb0 Paid Ads (25%)\"\n },\n {\n \"id\": \"paid_bar\",\n \"component\": \"Slider\",\n \"label\": \"Paid\",\n \"value\": 25,\n \"max\": 100\n },\n {\n \"id\": \"social_row\",\n \"component\": \"Row\",\n \"children\": [\n \"social_label\",\n \"social_bar\"\n ],\n \"justify\": \"spaceBetween\"\n },\n {\n \"id\": \"social_label\",\n \"component\": \"Text\",\n \"text\": \"\ud83d\udcf1 Social Media (15%)\"\n },\n {\n \"id\": \"social_bar\",\n \"component\": \"Slider\",\n \"label\": \"Social\",\n \"value\": 15,\n \"max\": 100\n },\n {\n \"id\": \"direct_row\",\n \"component\": \"Row\",\n \"children\": [\n \"direct_label\",\n \"direct_bar\"\n ],\n \"justify\": \"spaceBetween\"\n },\n {\n \"id\": \"direct_label\",\n \"component\": \"Text\",\n \"text\": \"\ud83c\udfe0 Direct (10%)\"\n },\n {\n \"id\": \"direct_bar\",\n \"component\": \"Slider\",\n \"label\": \"Direct\",\n \"value\": 10,\n \"max\": 100\n },\n {\n \"id\": \"referral_row\",\n \"component\": \"Row\",\n \"children\": [\n \"referral_label\",\n \"referral_bar\"\n ],\n \"justify\": \"spaceBetween\"\n },\n {\n \"id\": \"referral_label\",\n \"component\": \"Text\",\n \"text\": \"\ud83d\udd17 Referrals (5%)\"\n },\n {\n \"id\": \"referral_bar\",\n \"component\": \"Slider\",\n \"label\": \"Referral\",\n \"value\": 5,\n \"max\": 100\n }\n ]\n }\n}\n", "promptTokens": 10284, "completionTokens": 1238, "latencyMs": 42571, "finishReason": "stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/minimal","repeat":4,"output":"root = Stack([headerCard, contactCard])\n\nheaderCard = CardHeader(\"Get in Touch\", \"Fill out the form below and we'll get back to you within one business day.\")\n\ncontactCard = Card([formHeader, contactForm])\n\nformHeader = CardHeader(\"Contact Us\")\n\ncontactForm = Form(\"contact\", formButtons, [nameField, emailField, messageField])\n\nnameField = FormControl(\"Name\", nameInput, \"Your full name\")\nemailField = FormControl(\"Email\", emailInput, \"We'll only use this to reply\")\nmessageField = FormControl(\"Message\", messageInput, \"Tell us what's on your mind\")\n\nnameInput = Input(\"name\", \"Jane Doe\", \"text\", nameRules)\nemailInput = Input(\"email\", \"jane@example.com\", \"email\", emailRules)\nmessageInput = TextArea(\"message\", \"How can we help?\", 5, messageRules)\n\nnameRules = {required: true, minLength: 2, maxLength: 80}\nemailRules = {required: true, email: true}\nmessageRules = {required: true, minLength: 10, maxLength: 1000}\n\nformButtons = Buttons([submitBtn, clearBtn])\n\nsubmitBtn = Button(\"Send Message\", Action([@ToAssistant(\"Submit the contact form\")]), \"primary\", \"normal\", \"medium\")\nclearBtn = Button(\"Clear\", Action([@ToAssistant(\"Clear the contact form\")]), \"tertiary\", \"normal\", \"medium\")","promptTokens":6942,"completionTokens":488,"latencyMs":10294,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/minimal","repeat":1,"output":"root = Stack([formCard])\n\nformCard = Card([header, contactForm])\n\nheader = CardHeader(\"Get in Touch\", \"Fill out the form below and we'll get back to you within 1–2 business days.\")\n\ncontactForm = Form(\"contact\", formButtons, [nameField, emailField, subjectField, messageField])\n\nnameField = FormControl(\"Full name\", nameInput)\nnameInput = Input(\"name\", \"Jane Doe\", \"text\", {required: true, minLength: 2, maxLength: 80})\n\nemailField = FormControl(\"Email address\", emailInput, \"We'll only use this to reply to you.\")\nemailInput = Input(\"email\", \"jane@example.com\", \"email\", {required: true, email: true})\n\nsubjectField = FormControl(\"Topic\", subjectSelect)\nsubjectSelect = Select(\"subject\", subjectItems, \"Choose a topic\", {required: true})\nsubjectItems = [subjGeneral, subjSupport, subjSales, subjFeedback]\nsubjGeneral = SelectItem(\"general\", \"General inquiry\")\nsubjSupport = SelectItem(\"support\", \"Technical support\")\nsubjSales = SelectItem(\"sales\", \"Sales question\")\nsubjFeedback = SelectItem(\"feedback\", \"Product feedback\")\n\nmessageField = FormControl(\"Message\", messageArea, \"Up to 1000 characters.\")\nmessageArea = TextArea(\"message\", \"Tell us how we can help…\", 6, {required: true, minLength: 10, maxLength: 1000})\n\nformButtons = Buttons([submitBtn, clearBtn])\nsubmitBtn = Button(\"Send message\", Action([@ToAssistant(\"Submit the contact form\")]), \"primary\", \"normal\", \"medium\")\nclearBtn = Button(\"Clear\", Action([@ToAssistant(\"Clear the contact form\")]), \"tertiary\", \"normal\", \"medium\")","promptTokens":6942,"completionTokens":604,"latencyMs":10830,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/minimal","repeat":0,"output":"root = Stack([formCard])\n\nformCard = Card([header, contactForm])\n\nheader = CardHeader(\"Get in Touch\", \"Fill out the form below and we'll get back to you within 1–2 business days.\")\n\ncontactForm = Form(\"contact\", formButtons, [nameField, emailField, messageField])\n\nnameField = FormControl(\"Full name\", nameInput)\nnameInput = Input(\"name\", \"Jane Doe\", \"text\", {required: true, minLength: 2, maxLength: 80})\n\nemailField = FormControl(\"Email address\", emailInput, \"We'll only use this to reply to you.\")\nemailInput = Input(\"email\", \"jane@example.com\", \"email\", {required: true, email: true})\n\nmessageField = FormControl(\"Message\", messageInput, \"Please include as much detail as you can.\")\nmessageInput = TextArea(\"message\", \"How can we help?\", 5, {required: true, minLength: 10, maxLength: 1000})\n\nformButtons = Buttons([submitBtn, clearBtn])\nsubmitBtn = Button(\"Send message\", Action([@ToAssistant(\"Submit the contact form\")]), \"primary\")\nclearBtn = Button(\"Clear\", Action([@ToAssistant(\"Clear the contact form\")]), \"tertiary\")","promptTokens":6942,"completionTokens":402,"latencyMs":11760,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/minimal","repeat":2,"output":"root = Stack([headerCard, formCard])\n\nheaderCard = Card([header])\nheader = CardHeader(\"Get in Touch\", \"Send us a message and we'll get back to you within 1–2 business days.\")\n\nformCard = Card([contactForm])\n\ncontactForm = Form(\"contact\", formButtons, [nameField, emailField, messageField])\n\nnameField = FormControl(\"Full name\", nameInput, \"First and last name\")\nnameInput = Input(\"name\", \"Jane Doe\", \"text\", {required: true, minLength: 2, maxLength: 80})\n\nemailField = FormControl(\"Email address\", emailInput, \"We'll only use this to reply to you\")\nemailInput = Input(\"email\", \"jane@example.com\", \"email\", {required: true, email: true})\n\nmessageField = FormControl(\"Message\", messageInput, \"Up to 1000 characters\")\nmessageInput = TextArea(\"message\", \"Tell us how we can help…\", 6, {required: true, minLength: 10, maxLength: 1000})\n\nformButtons = Buttons([submitBtn, clearBtn])\nsubmitBtn = Button(\"Send message\", Action([@ToAssistant(\"Submit the contact form\")]), \"primary\", \"normal\", \"medium\")\nclearBtn = Button(\"Clear\", Action([@ToAssistant(\"Clear the contact form\")]), \"tertiary\", \"normal\", \"medium\")","promptTokens":6942,"completionTokens":446,"latencyMs":11831,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/minimal","repeat":3,"output":"root = Stack([header, contactCard], \"column\", \"l\")\n\nheader = CardHeader(\"Get in Touch\", \"Fill out the form below and we'll get back to you within 1–2 business days.\")\n\ncontactCard = Card([contactForm], \"card\")\n\ncontactForm = Form(\"contact\", formButtons, [nameField, emailField, messageField])\n\nnameField = FormControl(\"Name\", nameInput, \"Your full name\")\nnameInput = Input(\"name\", \"Jane Doe\", \"text\", {required: true, minLength: 2, maxLength: 60})\n\nemailField = FormControl(\"Email\", emailInput, \"We'll only use this to reply to you\")\nemailInput = Input(\"email\", \"jane@example.com\", \"email\", {required: true, email: true})\n\nmessageField = FormControl(\"Message\", messageInput, \"Tell us a bit about your request\")\nmessageInput = TextArea(\"message\", \"How can we help?\", 5, {required: true, minLength: 10, maxLength: 1000})\n\nformButtons = Buttons([submitBtn, clearBtn], \"row\")\nsubmitBtn = Button(\"Send message\", Action([@ToAssistant(\"Submit the contact form\")]), \"primary\", \"normal\", \"medium\")\nclearBtn = Button(\"Clear\", Action([@ToAssistant(\"Clear the contact form\")]), \"tertiary\", \"normal\", \"medium\")","promptTokens":6942,"completionTokens":454,"latencyMs":12660,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/realistic","repeat":0,"output":"root = Stack([header, infoCallout, formCard, footerNote])\n\nheader = CardHeader(\"Request an Appointment\", \"Tell us a little about you and we'll confirm your visit within one business day.\")\n\ninfoCallout = TextCallout(\"info\", \"Before you book\", \"Fields marked required help us reserve the right chair and clinician. Everything else is optional.\")\n\nformCard = Card([appointmentForm])\n\nappointmentForm = Form(\"appointment-request\", formButtons, [nameField, phoneField, emailField, treatmentField, dateField, notesField])\n\nnameField = FormControl(\"Full name\", nameInput, \"As it appears on your ID or insurance card.\")\nnameInput = Input(\"fullName\", \"Jane Doe\", \"text\", {\"required\": true, \"minLength\": 2, \"maxLength\": 80})\n\nphoneField = FormControl(\"Phone number\", phoneInput, \"We'll text or call to confirm your slot.\")\nphoneInput = Input(\"phone\", \"(555) 123-4567\", \"text\", {\"required\": true, \"minLength\": 7, \"maxLength\": 20})\n\nemailField = FormControl(\"Email address\", emailInput, \"Optional — for appointment reminders and receipts.\")\nemailInput = Input(\"email\", \"jane@example.com\", \"email\", {\"email\": true})\n\ntreatmentField = FormControl(\"Treatment requested\", treatmentSelect, \"Not sure? Choose checkup and we'll advise.\")\ntreatmentSelect = Select(\"treatment\", treatmentOptions, \"Select a treatment\", {\"required\": true})\ntreatmentOptions = [checkupOption, cleaningOption, whiteningOption, rootCanalOption]\ncheckupOption = SelectItem(\"checkup\", \"Routine checkup\")\ncleaningOption = SelectItem(\"cleaning\", \"Hygiene cleaning\")\nwhiteningOption = SelectItem(\"whitening\", \"Teeth whitening\")\nrootCanalOption = SelectItem(\"root-canal\", \"Root canal treatment\")\n\ndateField = FormControl(\"Preferred date\", datePicker, \"Optional — we'll offer the closest available time.\")\ndatePicker = DatePicker(\"preferredDate\", \"single\")\n\nnotesField = FormControl(\"Anything else we should know?\", notesArea, \"Dental anxiety, medications, ongoing pain, accessibility needs — anything helps.\")\nnotesArea = TextArea(\"notes\", \"Tell us about symptoms, past treatments, or scheduling preferences...\", 4, {\"maxLength\": 800})\n\nformButtons = Buttons([submitButton, callButton])\nsubmitButton = Button(\"Submit request\", Action([@ToAssistant(\"Submit my dental appointment request\")]), \"primary\", \"normal\", \"medium\")\ncallButton = Button(\"Prefer to call us?\", Action([@ToAssistant(\"Show the clinic phone number and opening hours\")]), \"tertiary\", \"normal\", \"medium\")\n\nfooterNote = TextContent(\"Requests are not confirmed until our front desk replies. For dental emergencies such as severe pain, swelling, or trauma, please call the clinic directly instead of using this form.\", \"small\")","promptTokens":7028,"completionTokens":1027,"latencyMs":17658,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/realistic","repeat":1,"output":"root = Stack([header, intro, formCard, footerNote])\n\nheader = CardHeader(\"Request an Appointment\", \"Tell us a few details and our front desk will confirm your slot within one business day.\")\n\nintro = TextCallout(\"info\", \"Fields marked required\", \"We need your name, phone number and treatment type to book you in. Everything else helps us prepare.\")\n\nformCard = Card([appointmentForm])\n\nappointmentForm = Form(\"appointmentRequest\", formButtons, [nameField, phoneField, emailField, treatmentField, dateField, notesField])\n\nnameField = FormControl(\"Full name\", nameInput, \"As it appears on your ID or insurance card.\")\nnameInput = Input(\"fullName\", \"Jane Doe\", \"text\", {required: true, minLength: 2, maxLength: 80})\n\nphoneField = FormControl(\"Phone number\", phoneInput, \"We'll text or call to confirm your appointment.\")\nphoneInput = Input(\"phone\", \"(555) 123-4567\", \"text\", {required: true, minLength: 7, maxLength: 20})\n\nemailField = FormControl(\"Email address\", emailInput, \"Optional — used for reminders and receipts.\")\nemailInput = Input(\"email\", \"jane@example.com\", \"email\", {email: true})\n\ntreatmentField = FormControl(\"Treatment requested\", treatmentSelect, \"Not sure? Choose a check-up and we'll advise.\")\ntreatmentSelect = Select(\"treatment\", treatmentOptions, \"Select a treatment\", {required: true})\ntreatmentOptions = [checkupOpt, cleaningOpt, whiteningOpt, rootCanalOpt]\ncheckupOpt = SelectItem(\"checkup\", \"Check-up (30 min)\")\ncleaningOpt = SelectItem(\"cleaning\", \"Hygiene cleaning (45 min)\")\nwhiteningOpt = SelectItem(\"whitening\", \"Teeth whitening (60 min)\")\nrootCanalOpt = SelectItem(\"root-canal\", \"Root canal treatment (90 min)\")\n\ndateField = FormControl(\"Preferred date\", datePicker, \"Optional — we'll offer the closest available slot.\")\ndatePicker = DatePicker(\"preferredDate\", \"single\")\n\nnotesField = FormControl(\"Anything else we should know?\", notesArea, \"Pain, dental anxiety, medications, insurance details or accessibility needs.\")\nnotesArea = TextArea(\"notes\", \"e.g. I have a sensitive upper-left molar and prefer morning appointments.\", 4, {maxLength: 600})\n\nformButtons = Buttons([submitBtn, clearBtn])\nsubmitBtn = Button(\"Request appointment\", null, \"primary\", \"normal\", \"medium\")\nclearBtn = Button(\"Clear form\", null, \"tertiary\", \"normal\", \"medium\")\n\nfooterNote = TextContent(\"Need care today? For dental emergencies please call the clinic directly on **(555) 010-2200** instead of using this form.\", \"small\")","promptTokens":7028,"completionTokens":995,"latencyMs":14330,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/adversarial","repeat":0,"output":"root = Stack([intro, feedbackCard], \"column\", \"l\")\n\nintro = CardHeader(\"How did we do?\", \"Three quick questions — takes about 30 seconds.\")\n\nfeedbackCard = Card([feedbackForm])\n\nfeedbackForm = Form(\"feedback\", formButtons, [happinessField, issueField, detailsField, contactField])\n\nhappinessField = FormControl(\"Overall, how happy are you?\", happinessInput)\nhappinessInput = RadioGroup(\"happiness\", happinessItems, \"good\")\nhappinessItems = [happyGreat, happyGood, happyMeh, happyBad]\nhappyGreat = RadioItem(\"Great\", \"Everything worked\", \"great\")\nhappyGood = RadioItem(\"Good\", \"Minor bumps\", \"good\")\nhappyMeh = RadioItem(\"Okay\", \"Could be better\", \"okay\")\nhappyBad = RadioItem(\"Bad\", \"Something went wrong\", \"bad\")\n\nissueField = FormControl(\"Anything go wrong?\", issueInput, \"Optional — pick the closest match.\")\nissueInput = Select(\"issue\", issueItems, \"Nothing, all good\")\nissueItems = [issueNone, issueSlow, issueConfusing, issueBroken, issueSupport, issueOther]\nissueNone = SelectItem(\"none\", \"Nothing — all good\")\nissueSlow = SelectItem(\"slow\", \"Too slow\")\nissueConfusing = SelectItem(\"confusing\", \"Confusing to use\")\nissueBroken = SelectItem(\"broken\", \"Something was broken\")\nissueSupport = SelectItem(\"support\", \"Support experience\")\nissueOther = SelectItem(\"other\", \"Something else\")\n\ndetailsField = FormControl(\"Tell us more\", detailsInput, \"A sentence is plenty.\")\ndetailsInput = TextArea(\"details\", \"What happened?\", 3, detailsRules)\ndetailsRules = {maxLength: 400}\n\ncontactField = FormControl(\"Email (only if you'd like a reply)\", contactInput)\ncontactInput = Input(\"email\", \"you@example.com\", \"email\", contactRules)\ncontactRules = {email: true}\n\nformButtons = Buttons([submitBtn, skipBtn], \"column\")\nsubmitBtn = Button(\"Send feedback\", Action([@ToAssistant(\"Submit my feedback\")]), \"primary\", \"normal\", \"medium\")\nskipBtn = Button(\"Not now\", Action([@ToAssistant(\"Skip the feedback form\")]), \"tertiary\", \"normal\", \"medium\")","promptTokens":7009,"completionTokens":886,"latencyMs":12012,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/realistic","repeat":3,"output":"root = Stack([headerCard, formCard, footerNote], \"column\", \"l\")\n\nheaderCard = Card([header, intro])\nheader = CardHeader(\"Request an Appointment\", \"Tell us what you need and we'll confirm your slot within one business day\")\nintro = TextContent(\"Fields marked **required** are the minimum we need to reserve a time for you. We'll call or email you back to confirm.\", \"small\")\n\nformCard = Card([appointmentForm])\nappointmentForm = Form(\"appointment_request\", formButtons, [nameField, phoneField, emailField, treatmentField, dateField, notesField])\n\nnameField = FormControl(\"Full name\", nameInput, \"As it appears on your ID or insurance card\")\nnameInput = Input(\"full_name\", \"Jane Alexandra Doe\", \"text\", {required: true, minLength: 2, maxLength: 80})\n\nphoneField = FormControl(\"Phone number\", phoneInput, \"We'll text a reminder the day before\")\nphoneInput = Input(\"phone\", \"(555) 123-4567\", \"text\", {required: true, minLength: 7, maxLength: 20})\n\nemailField = FormControl(\"Email address\", emailInput, \"Optional — used to send your confirmation\")\nemailInput = Input(\"email\", \"jane@example.com\", \"email\", {email: true})\n\ntreatmentField = FormControl(\"Treatment requested\", treatmentSelect, \"Not sure? Choose Check-up and we'll advise\")\ntreatmentSelect = Select(\"treatment\", treatmentItems, \"Select a treatment\", {required: true})\ntreatmentItems = [checkupItem, cleaningItem, whiteningItem, rootCanalItem]\ncheckupItem = SelectItem(\"checkup\", \"Check-up (30 min)\")\ncleaningItem = SelectItem(\"cleaning\", \"Cleaning / Hygiene (45 min)\")\nwhiteningItem = SelectItem(\"whitening\", \"Teeth Whitening (60 min)\")\nrootCanalItem = SelectItem(\"root_canal\", \"Root Canal (90 min)\")\n\ndateField = FormControl(\"Preferred date\", datePicker, \"Optional — we'll offer the closest available slot\")\ndatePicker = DatePicker(\"preferred_date\", \"single\")\n\nnotesField = FormControl(\"Anything else we should know?\", notesArea, \"Pain level, nervousness, insurance, accessibility needs, preferred time of day\")\nnotesArea = TextArea(\"notes\", \"e.g. I'd prefer a morning appointment and I get anxious at the dentist.\", 4, {maxLength: 600})\n\nformButtons = Buttons([submitBtn, callBtn])\nsubmitBtn = Button(\"Submit request\", Action([@ToAssistant(\"Submit my dental appointment request\")]), \"primary\", \"normal\", \"medium\")\ncallBtn = Button(\"Call the clinic instead\", Action([@ToAssistant(\"Show me the clinic phone number and opening hours\")]), \"tertiary\", \"normal\", \"medium\")\n\nfooterNote = Callout(\"info\", \"This is a request, not a confirmed booking\", \"A team member reviews every request. For dental emergencies or severe pain, please call us directly instead of using this form.\")","promptTokens":7028,"completionTokens":1045,"latencyMs":13919,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/realistic","repeat":4,"output":"root = Stack([header, intro, formCard, footerNote])\n\nheader = CardHeader(\"Request an Appointment\", \"Fill in your details and we'll confirm your visit within one business day\")\n\nintro = TextCallout(\"info\", \"How it works\", \"Submit the form below and a member of our reception team will contact you to confirm the exact time. Fields marked required are the minimum we need to book you in.\")\n\nformCard = Card([appointmentForm])\n\nappointmentForm = Form(\"appointment-request\", formButtons, [nameField, phoneField, emailField, treatmentField, dateField, notesField])\n\nnameField = FormControl(\"Full name\", nameInput, \"As it appears on your ID or insurance card\")\nnameInput = Input(\"fullName\", \"e.g. Jane Marie Doe\", \"text\", {required: true, minLength: 2, maxLength: 80})\n\nphoneField = FormControl(\"Phone number\", phoneInput, \"We'll text or call you to confirm your slot\")\nphoneInput = Input(\"phone\", \"e.g. 07700 900123\", \"text\", {required: true, minLength: 7, maxLength: 20})\n\nemailField = FormControl(\"Email address\", emailInput, \"Optional — used only to send your appointment confirmation\")\nemailInput = Input(\"email\", \"you@example.com\", \"email\", {email: true})\n\ntreatmentField = FormControl(\"Treatment requested\", treatmentSelect, \"Not sure? Choose a check-up and we'll advise you at the visit\")\ntreatmentSelect = Select(\"treatment\", treatmentOptions, \"Select a treatment\", {required: true})\ntreatmentOptions = [checkupOption, cleaningOption, whiteningOption, rootCanalOption]\ncheckupOption = SelectItem(\"checkup\", \"Check-up (30 min)\")\ncleaningOption = SelectItem(\"cleaning\", \"Hygiene cleaning (45 min)\")\nwhiteningOption = SelectItem(\"whitening\", \"Teeth whitening (60 min)\")\nrootCanalOption = SelectItem(\"root-canal\", \"Root canal treatment (90 min)\")\n\ndateField = FormControl(\"Preferred date\", datePicker, \"Optional — we'll offer the closest available slot\")\ndatePicker = DatePicker(\"preferredDate\", \"single\")\n\nnotesField = FormControl(\"Anything else we should know?\", notesArea, \"Pain, anxiety, medication, accessibility needs or preferred time of day\")\nnotesArea = TextArea(\"notes\", \"e.g. I get nervous at the dentist and prefer morning appointments. I'm also taking blood thinners.\", 4, {maxLength: 600})\n\nformButtons = Buttons([submitBtn, callBtn])\nsubmitBtn = Button(\"Submit request\", Action([@ToAssistant(\"Submit my dental appointment request\")]), \"primary\", \"normal\", \"medium\")\ncallBtn = Button(\"Prefer to call?\", Action([@ToAssistant(\"Show me the clinic's phone number and opening hours\")]), \"tertiary\", \"normal\", \"medium\")\n\nfooterNote = TextContent(\"**Dental emergency?** Please call the clinic directly instead of using this form — we keep same-day slots for urgent pain and trauma. Your details are stored securely and never shared with third parties.\", \"small\")","promptTokens":7028,"completionTokens":1088,"latencyMs":14545,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/realistic","repeat":2,"output":"root = Stack([header, intro, formCard, footerNote], \"column\", \"l\")\n\nheader = CardHeader(\"Request an Appointment\", \"Tell us what you need and we'll confirm your slot within one business day\")\n\nintro = TextCallout(\"info\", \"Before you start\", \"Fields marked required are the minimum we need to reserve a slot. For dental emergencies, please call us directly at (555) 014-2280.\")\n\nformCard = Card([header, appointmentForm], \"card\")\n\nappointmentForm = Form(\"appointment-request\", formButtons, [nameField, phoneField, emailField, treatmentField, dateField, notesField])\n\nnameField = FormControl(\"Full name\", nameInput, \"As it appears on your ID or insurance card\")\nnameInput = Input(\"fullName\", \"Jane Doe\", \"text\", {required: true, minLength: 2, maxLength: 80})\n\nphoneField = FormControl(\"Phone number\", phoneInput, \"We'll text or call to confirm your appointment\")\nphoneInput = Input(\"phone\", \"(555) 123-4567\", \"text\", {required: true, minLength: 7, maxLength: 20})\n\nemailField = FormControl(\"Email address\", emailInput, \"Optional — for your appointment reminder and receipts\")\nemailInput = Input(\"email\", \"jane@example.com\", \"email\", {email: true})\n\ntreatmentField = FormControl(\"Treatment requested\", treatmentSelect, \"Not sure? Choose Check-up and we'll advise at your visit\")\ntreatmentSelect = Select(\"treatment\", treatmentOptions, \"Select a treatment\", {required: true})\ntreatmentOptions = [optCheckup, optCleaning, optWhitening, optRootCanal]\noptCheckup = SelectItem(\"checkup\", \"Check-up (30 min)\")\noptCleaning = SelectItem(\"cleaning\", \"Cleaning / Hygiene (45 min)\")\noptWhitening = SelectItem(\"whitening\", \"Teeth Whitening (60 min)\")\noptRootCanal = SelectItem(\"root-canal\", \"Root Canal (90 min)\")\n\ndateField = FormControl(\"Preferred date\", datePicker, \"Optional — we'll offer the closest available slot\")\ndatePicker = DatePicker(\"preferredDate\", \"single\")\n\nnotesField = FormControl(\"Anything else we should know?\", notesArea, \"Pain level, past treatments, allergies, anxiety, accessibility needs, insurance provider\")\nnotesArea = TextArea(\"notes\", \"e.g. I have a sensitive upper left molar and prefer morning appointments.\", 5, {maxLength: 1000})\n\nformButtons = Buttons([submitBtn, callBtn], \"row\")\nsubmitBtn = Button(\"Submit request\", Action([@ToAssistant(\"Submit the dental appointment request\")]), \"primary\", \"normal\", \"medium\")\ncallBtn = Button(\"Call the clinic instead\", Action([@OpenUrl(\"tel:+15550142280\")]), \"secondary\", \"normal\", \"medium\")\n\nfooterNote = TextContent(\"_Submitting a request does not confirm a booking. Our front desk will contact you to finalise the date and time. Your details are stored securely and used only for your dental care._\", \"small\")","promptTokens":7028,"completionTokens":1066,"latencyMs":16174,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/adversarial","repeat":1,"output":"root = Stack([shell], \"column\", \"m\")\n\nshell = Card([header, form, footnote], \"card\", \"column\", \"m\")\n\nheader = CardHeader(\"Quick feedback\", \"Three questions, about 30 seconds.\")\n\nform = Form(\"feedback\", formButtons, [happinessField, issueField, detailsField, contactField])\n\nhappinessField = FormControl(\"How happy were you?\", happinessInput, \"Tap the one that fits best.\")\nhappinessInput = RadioGroup(\"satisfaction\", happinessItems, \"ok\", {required: true})\nhappinessItems = [happyGreat, happyOk, happyBad]\nhappyGreat = RadioItem(\"Great\", \"Worked the way I hoped\", \"great\")\nhappyOk = RadioItem(\"Okay\", \"Got there, but with friction\", \"ok\")\nhappyBad = RadioItem(\"Frustrating\", \"Something clearly went wrong\", \"bad\")\n\nissueField = FormControl(\"Anything go wrong?\", issueInput)\nissueInput = Select(\"issue\", issueItems, \"Pick the closest match\", null, $issue, \"large\")\nissueItems = [issueNone, issueConfusing, issueSlow, issueBroken, issueMissing, issueOther]\nissueNone = SelectItem(\"none\", \"Nothing — all good\")\nissueConfusing = SelectItem(\"confusing\", \"Confusing to use\")\nissueSlow = SelectItem(\"slow\", \"Too slow\")\nissueBroken = SelectItem(\"broken\", \"Something was broken\")\nissueMissing = SelectItem(\"missing\", \"Missing something I needed\")\nissueOther = SelectItem(\"other\", \"Something else\")\n\ndetailsField = $issue != \"\" ? detailsControl : null\ndetailsControl = FormControl(\"Want to say more?\", detailsInput, \"Optional — one line is plenty.\")\ndetailsInput = TextArea(\"details\", \"What happened?\", 3, {maxLength: 300})\n\ncontactField = FormControl(\"Email for a reply\", contactInput, \"Optional. Only used to follow up on this.\")\ncontactInput = Input(\"email\", \"you@example.com\", \"email\", {email: true})\n\nformButtons = Buttons([submitBtn, skipBtn], \"column\")\nsubmitBtn = Button(\"Send feedback\", sendAction, \"primary\", \"normal\", \"large\")\nskipBtn = Button(\"Not right now\", skipAction, \"tertiary\", \"normal\", \"medium\")\n\nsendAction = Action([@ToAssistant(\"Submit my feedback\")])\nskipAction = Action([@ToAssistant(\"Dismiss the feedback form\")])\n\nfootnote = TextContent(\"We read every response. No account or sign-in needed.\", \"small\")","promptTokens":7009,"completionTokens":962,"latencyMs":15063,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/minimal","repeat":0,"output":"root = Stack([header, table, note])\n\nheader = CardHeader(\"Three Largest Countries by Area\", \"Total area including land and water, in square kilometers\")\n\ntable = Table([colRank, colCountry, colContinent, colArea, colShare])\n\ncolRank = Col(\"Rank\", [\"1\", \"2\", \"3\"])\ncolCountry = Col(\"Country\", [\"Russia\", \"Canada\", \"China\"])\ncolContinent = Col(\"Continent\", [\"Europe / Asia\", \"North America\", \"Asia\"])\ncolArea = Col(\"Area (km²)\", [17098246, 9984670, 9596961], \"number\")\ncolShare = Col(\"% of World Land\", [11.5, 6.7, 6.4], \"number\")\n\nnote = TextContent(\"Figures are approximate. The United States is sometimes ranked third instead of China depending on whether coastal and territorial waters are included.\", \"small\")","promptTokens":6943,"completionTokens":287,"latencyMs":6704,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/minimal","repeat":1,"output":"root = Stack([header, areaTable, chartCard, note])\n\nheader = CardHeader(\"Three Largest Countries by Area\", \"Total area including land and water, in square kilometers\")\n\nareaTable = Table([rankCol, countryCol, continentCol, areaCol, shareCol])\n\nrankCol = Col(\"Rank\", [1, 2, 3], \"number\")\ncountryCol = Col(\"Country\", [\"Russia\", \"Canada\", \"China\"])\ncontinentCol = Col(\"Region\", [europeAsiaTag, northAmericaTag, asiaTag])\nareaCol = Col(\"Area (km²)\", [17098246, 9984670, 9596961], \"number\")\nshareCol = Col(\"Share of Land\", [\"11.0%\", \"6.1%\", \"6.3%\"])\n\neuropeAsiaTag = Tag(\"Europe / Asia\", null, \"sm\", \"info\")\nnorthAmericaTag = Tag(\"North America\", null, \"sm\", \"success\")\nasiaTag = Tag(\"Asia\", null, \"sm\", \"warning\")\n\nchartCard = Card([chartHeader, areaChart])\nchartHeader = CardHeader(\"Area Comparison\", \"In millions of km²\")\nareaChart = HorizontalBarChart([\"Russia\", \"Canada\", \"China\"], [areaSeries], \"grouped\", \"Area (million km²)\", \"Country\")\nareaSeries = Series(\"Area\", [17.1, 9.98, 9.6])\n\nnote = TextContent(\"Russia is roughly **1.7×** larger than Canada and spans 11 time zones. Canada and China are very close in size — rankings can flip depending on whether disputed territories and coastal waters are included.\", \"small\")","promptTokens":6943,"completionTokens":546,"latencyMs":8743,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/adversarial","repeat":3,"output":"root = Stack([header, formCard, footNote], \"column\", \"l\")\n\nheader = CardHeader(\"How did we do?\", \"Three quick questions — takes about 20 seconds.\")\n\nformCard = Card([feedbackForm], \"card\")\n\nfeedbackForm = Form(\"feedback\", formButtons, [happinessField, issueField, detailsField, contactField])\n\nhappinessField = FormControl(\"How happy were you?\", happinessRadio)\nhappinessRadio = RadioGroup(\"happiness\", [happyItem, okItem, sadItem], \"ok\", ratingRules)\nhappyItem = RadioItem(\"Great\", \"Everything worked the way I hoped\", \"great\")\nokItem = RadioItem(\"Okay\", \"Fine, but something could be better\", \"ok\")\nsadItem = RadioItem(\"Not good\", \"I hit a real problem\", \"bad\")\nratingRules = {required: true}\n\nissueField = FormControl(\"Anything go wrong?\", issueSelect, \"Optional — pick the closest match.\")\nissueSelect = Select(\"issue\", issueOptions, \"Nothing, all good\")\nissueOptions = [optNone, optSlow, optConfusing, optBroken, optMissing, optOther]\noptNone = SelectItem(\"none\", \"Nothing went wrong\")\noptSlow = SelectItem(\"slow\", \"Too slow\")\noptConfusing = SelectItem(\"confusing\", \"Confusing or hard to find\")\noptBroken = SelectItem(\"broken\", \"Something was broken\")\noptMissing = SelectItem(\"missing\", \"Missing a feature I needed\")\noptOther = SelectItem(\"other\", \"Something else\")\n\ndetailsField = FormControl(\"Want to add a detail?\", detailsArea, \"One sentence is plenty.\")\ndetailsArea = TextArea(\"details\", \"e.g. the export button did nothing\", 3, detailsRules)\ndetailsRules = {maxLength: 300}\n\ncontactField = FormControl(\"Email for a reply\", contactInput, \"Leave blank if you don't want a response.\")\ncontactInput = Input(\"email\", \"you@example.com\", \"email\", contactRules)\ncontactRules = {email: true}\n\nformButtons = Buttons([submitBtn, skipBtn], \"column\")\nsubmitBtn = Button(\"Send feedback\", Action([@ToAssistant(\"Submit my feedback\")]), \"primary\", \"normal\", \"medium\")\nskipBtn = Button(\"Skip for now\", Action([@ToAssistant(\"Skip the feedback form\")]), \"tertiary\", \"normal\", \"medium\")\n\nfootNote = TextContent(\"We read every response. No follow-up unless you leave an email.\", \"small\")","promptTokens":7009,"completionTokens":914,"latencyMs":12016,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/adversarial","repeat":2,"output":"root = Stack([feedbackCard])\n\nfeedbackCard = Card([header, feedbackForm, footerNote])\n\nheader = CardHeader(\"How did we do?\", \"Takes about 20 seconds — three questions, all optional except the first.\")\n\nfeedbackForm = Form(\"feedback\", formButtons, [happinessField, issueField, detailsField, contactField])\n\nhappinessField = FormControl(\"How happy were you?\", happinessRadio)\nhappinessRadio = RadioGroup(\"happiness\", happinessItems, \"good\", happinessRules)\nhappinessItems = [happyItem, okItem, sadItem]\nhappyItem = RadioItem(\"Great\", \"Everything worked the way I expected\", \"great\")\nokItem = RadioItem(\"Okay\", \"It worked, but something felt off\", \"good\")\nsadItem = RadioItem(\"Not good\", \"I had a real problem\", \"bad\")\nhappinessRules = {required: true}\n\nissueField = FormControl(\"Anything go wrong?\", issueSelect, \"Skip this if it all went fine.\")\nissueSelect = Select(\"issue\", issueItems, \"Pick the closest thing\")\nissueItems = [issueNone, issueSlow, issueConfusing, issueBroken, issueMissing, issueOther]\nissueNone = SelectItem(\"none\", \"Nothing — it was fine\")\nissueSlow = SelectItem(\"slow\", \"Too slow\")\nissueConfusing = SelectItem(\"confusing\", \"Confusing to use\")\nissueBroken = SelectItem(\"broken\", \"Something was broken\")\nissueMissing = SelectItem(\"missing\", \"Missing something I needed\")\nissueOther = SelectItem(\"other\", \"Something else\")\n\ndetailsField = FormControl(\"Want to add a line?\", detailsArea, \"Optional. One or two sentences is plenty.\")\ndetailsArea = TextArea(\"details\", \"What happened?\", 3, detailsRules)\ndetailsRules = {maxLength: 400}\n\ncontactField = FormControl(\"Email, if you'd like a reply\", contactInput, \"Leave blank to stay anonymous.\")\ncontactInput = Input(\"email\", \"you@example.com\", \"email\", contactRules)\ncontactRules = {email: true}\n\nformButtons = Buttons([submitBtn, skipBtn], \"column\")\nsubmitBtn = Button(\"Send feedback\", Action([@ToAssistant(\"Submit my feedback\")]), \"primary\", \"normal\", \"large\")\nskipBtn = Button(\"Not right now\", Action([@ToAssistant(\"Dismiss the feedback form\")]), \"tertiary\", \"normal\", \"large\")\n\nfooterNote = TextContent(\"We read every response. No follow-up unless you leave an email.\", \"small\")","promptTokens":7009,"completionTokens":972,"latencyMs":13456,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/minimal","repeat":2,"output":"root = Stack([header, areaTable, areaChart, note])\n\nheader = CardHeader(\"Three Largest Countries by Area\", \"Total area in millions of square kilometers\")\n\nareaTable = Table([colRank, colCountry, colContinent, colArea, colShare])\n\ncolRank = Col(\"Rank\", [\"1\", \"2\", \"3\"])\ncolCountry = Col(\"Country\", [\"Russia\", \"Canada\", \"China\"])\ncolContinent = Col(\"Region\", [\"Europe / Asia\", \"North America\", \"Asia\"])\ncolArea = Col(\"Area (km²)\", [17098246, 9984670, 9596961], \"number\")\ncolShare = Col(\"% of World Land\", [11.5, 6.7, 6.4], \"number\")\n\nareaChart = Card([chartHeader, barChart])\nchartHeader = CardHeader(\"Area Comparison\", \"Millions of km²\")\nbarChart = HorizontalBarChart([\"Russia\", \"Canada\", \"China\"], [areaSeries], \"grouped\", \"Area (million km²)\", \"Country\")\nareaSeries = Series(\"Area\", [17.1, 9.98, 9.6])\n\nnote = TextContent(\"Russia is roughly **1.7×** larger than Canada and spans 11 time zones. Canada and China are nearly tied — rankings can shift slightly depending on whether disputed territories and inland water are included.\", \"small\")","promptTokens":6943,"completionTokens":454,"latencyMs":7107,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"contact-form/adversarial","repeat":4,"output":"root = Stack([intro, feedbackCard, footerNote], \"column\", \"l\")\n\nintro = CardHeader(\"How did we do?\", \"Three quick questions — under a minute, promise.\")\n\nfeedbackCard = Card([feedbackForm], \"card\")\n\nfeedbackForm = Form(\"feedback\", formButtons, [happinessField, issueField, detailsField, contactField])\n\nhappinessField = FormControl(\"How happy were you?\", happinessRadio)\nhappinessRadio = RadioGroup(\"satisfaction\", [happyItem, okItem, sadItem], \"ok\", {required: true})\nhappyItem = RadioItem(\"Great\", \"Everything worked the way I hoped\", \"great\")\nokItem = RadioItem(\"Fine\", \"It did the job, nothing special\", \"fine\")\nsadItem = RadioItem(\"Not good\", \"Something got in my way\", \"bad\")\n\nissueField = FormControl(\"Anything go wrong?\", issueSelect, \"Skip if it all went smoothly.\")\nissueSelect = Select(\"issue\", issueOptions, \"Pick the closest thing\", null, $issue, \"medium\")\nissueOptions = [issueNone, issueConfusing, issueSlow, issueBroken, issueMissing, issueOther]\nissueNone = SelectItem(\"none\", \"Nothing — it was fine\")\nissueConfusing = SelectItem(\"confusing\", \"I couldn't figure something out\")\nissueSlow = SelectItem(\"slow\", \"It was too slow\")\nissueBroken = SelectItem(\"broken\", \"Something was broken\")\nissueMissing = SelectItem(\"missing\", \"What I needed wasn't there\")\nissueOther = SelectItem(\"other\", \"Something else\")\n\ndetailsField = $issue != \"\" ? detailsControl : null\ndetailsControl = FormControl(\"Want to say more?\", detailsArea, \"Optional — even one line helps.\")\ndetailsArea = TextArea(\"details\", \"What happened?\", 3, {maxLength: 400})\n\ncontactField = FormControl(\"Email, if you'd like a reply\", contactInput, \"Leave blank to stay anonymous.\")\ncontactInput = Input(\"email\", \"you@example.com\", \"email\", {email: true})\n\nformButtons = Buttons([submitBtn, skipBtn], \"column\")\nsubmitBtn = Button(\"Send feedback\", Action([@ToAssistant(\"Submit my feedback\")]), \"primary\", \"normal\", \"large\")\nskipBtn = Button(\"Maybe later\", Action([@ToAssistant(\"Dismiss the feedback form\")]), \"tertiary\", \"normal\", \"large\")\n\nfooterNote = TextContent(\"We read every response. No follow-up unless you ask for one.\", \"small\")","promptTokens":7009,"completionTokens":936,"latencyMs":15779,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/minimal","repeat":3,"output":"root = Stack([header, areaTable, areaChart, note])\n\nheader = CardHeader(\"Three Largest Countries by Area\", \"Total area in square kilometers, including land and water\")\n\nareaTable = Card([tableHeader, countryTable])\n\ntableHeader = CardHeader(\"Ranked by Total Area\")\n\ncountryTable = Table([rankCol, countryCol, continentCol, areaCol, shareCol])\n\nrankCol = Col(\"Rank\", [\"1\", \"2\", \"3\"])\ncountryCol = Col(\"Country\", [\"Russia\", \"Canada\", \"China\"])\ncontinentCol = Col(\"Region\", [\"Europe / Asia\", \"North America\", \"Asia\"])\nareaCol = Col(\"Area (km²)\", [17098246, 9984670, 9596961], \"number\")\nshareCol = Col(\"Share of Earth's Land\", [\"11.5%\", \"6.7%\", \"6.4%\"])\n\nareaChart = Card([chartHeader, barChart])\n\nchartHeader = CardHeader(\"Area Comparison\", \"Millions of square kilometers\")\n\nbarChart = HorizontalBarChart([\"Russia\", \"Canada\", \"China\"], [areaSeries], \"grouped\", \"Area (million km²)\", \"Country\")\n\nareaSeries = Series(\"Total area\", [17.1, 9.98, 9.6])\n\nnote = TextCallout(\"info\", \"Note on rankings\", \"China and the United States are very close in size; their order can swap depending on whether disputed territories and coastal waters are counted. Russia is by far the largest, spanning 11 time zones.\")","promptTokens":6943,"completionTokens":515,"latencyMs":8547,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/minimal","repeat":4,"output":"root = Stack([header, areaTable, areaChart, note])\n\nheader = CardHeader(\"Three Largest Countries by Area\", \"Total area including land and inland water, in square kilometers\")\n\nareaTable = Card([tableInner])\ntableInner = Table([rankCol, countryCol, continentCol, areaCol, shareCol])\n\nrankCol = Col(\"Rank\", [\"1\", \"2\", \"3\"])\ncountryCol = Col(\"Country\", [\"Russia\", \"Canada\", \"China\"])\ncontinentCol = Col(\"Region\", [asiaEuropeTag, northAmericaTag, asiaTag])\nareaCol = Col(\"Area (km²)\", [17098246, 9984670, 9596961], \"number\")\nshareCol = Col(\"% of World Land\", [11.0, 6.1, 6.3], \"number\")\n\nasiaEuropeTag = Tag(\"Europe / Asia\", null, \"sm\", \"info\")\nnorthAmericaTag = Tag(\"North America\", null, \"sm\", \"success\")\nasiaTag = Tag(\"Asia\", null, \"sm\", \"warning\")\n\nareaChart = Card([chartHeader, barChart])\nchartHeader = CardHeader(\"Area Comparison\", \"In millions of square kilometers\")\nbarChart = HorizontalBarChart([\"Russia\", \"Canada\", \"China\"], [areaSeries], \"grouped\", \"Area (million km²)\", \"Country\")\nareaSeries = Series(\"Area\", [17.1, 9.98, 9.6])\n\nnote = TextCallout(\"info\", \"Note on rankings\", \"China and the United States are sometimes swapped in the 3rd/4th spots depending on whether disputed territories and coastal waters are included in the measurement.\")","promptTokens":6943,"completionTokens":563,"latencyMs":8662,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/realistic","repeat":0,"output":"root = Stack([header, kpiRow, mainCard, chartTabs, footnote], \"column\", \"l\")\n\nheader = CardHeader(\"Q3 FY24 Sales Performance — EMEA & APAC\", \"8 quota-carrying reps · quarter closed 30 Sep · figures in USD, new ARR\")\n\nkpiRow = Stack([kpi1, kpi2, kpi3, kpi4], \"row\", \"m\", \"stretch\")\n\nkpi1 = Card([kpi1Label, kpi1Value, kpi1Sub])\nkpi1Label = TextContent(\"Total Closed Revenue\", \"small\")\nkpi1Value = TextContent(\"$5.69M\", \"large-heavy\")\nkpi1Sub = TextContent(\"vs $5.60M team quota\", \"small\")\n\nkpi2 = Card([kpi2Label, kpi2Value, kpi2Sub])\nkpi2Label = TextContent(\"Team Quota Attainment\", \"small\")\nkpi2Value = TextContent(\"102%\", \"large-heavy\")\nkpi2Sub = TextContent(\"+2 pts vs Q2\", \"small\")\n\nkpi3 = Card([kpi3Label, kpi3Value, kpi3Sub])\nkpi3Label = TextContent(\"Deals Closed\", \"small\")\nkpi3Value = TextContent(\"103\", \"large-heavy\")\nkpi3Sub = TextContent(\"Avg deal size $55.3K\", \"small\")\n\nkpi4 = Card([kpi4Label, kpi4Value, kpi4Sub])\nkpi4Label = TextContent(\"Reps Over Quota\", \"small\")\nkpi4Value = TextContent(\"4 / 8\", \"large-heavy\")\nkpi4Sub = TextContent(\"2 within 5 pts of target\", \"small\")\n\nmainCard = Card([tableHeader, repTable])\ntableHeader = CardHeader(\"Rep Scorecard\", \"Ranked by quota attainment\")\nrepTable = Table([colRep, colRegion, colDeals, colRevenue, colQuota, colAttain, colStatus])\n\ncolRep = Col(\"Rep\", repNames)\ncolRegion = Col(\"Region\", regions)\ncolDeals = Col(\"Deals Closed\", deals, \"number\")\ncolRevenue = Col(\"Revenue\", revenueLabels)\ncolQuota = Col(\"Quota\", quotaLabels)\ncolAttain = Col(\"Attainment\", attainLabels)\ncolStatus = Col(\"Status\", statusTags)\n\nrepNames = [\"Nadia Haddad\", \"Rajesh Iyer\", \"Mei Tanaka\", \"Sofia Almeida\", \"Liam O'Connor\", \"Grace Lim\", \"Daniel Park\", \"Jonas Weber\"]\nregions = [\"EMEA\", \"APAC\", \"APAC\", \"EMEA\", \"EMEA\", \"APAC\", \"APAC\", \"EMEA\"]\ndeals = [17, 19, 13, 14, 11, 9, 12, 8]\nrevenueLabels = [\"$1,024,000\", \"$918,000\", \"$735,000\", \"$842,000\", \"$681,000\", \"$488,000\", \"$604,000\", \"$402,000\"]\nquotaLabels = [\"$850,000\", \"$800,000\", \"$650,000\", \"$750,000\", \"$700,000\", \"$550,000\", \"$700,000\", \"$600,000\"]\nattainLabels = [\"120%\", \"115%\", \"113%\", \"112%\", \"97%\", \"89%\", \"86%\", \"67%\"]\n\nstatusTags = [tagOver1, tagOver2, tagOver3, tagOver4, tagNear1, tagNear2, tagNear3, tagMiss]\ntagOver1 = Tag(\"Over quota\", null, \"sm\", \"success\")\ntagOver2 = Tag(\"Over quota\", null, \"sm\", \"success\")\ntagOver3 = Tag(\"Over quota\", null, \"sm\", \"success\")\ntagOver4 = Tag(\"Over quota\", null, \"sm\", \"success\")\ntagNear1 = Tag(\"Near quota\", null, \"sm\", \"warning\")\ntagNear2 = Tag(\"Near quota\", null, \"sm\", \"warning\")\ntagNear3 = Tag(\"Near quota\", null, \"sm\", \"warning\")\ntagMiss = Tag(\"Under quota\", null, \"sm\", \"danger\")\n\nchartTabs = Tabs([tabRevenue, tabAttain, tabRegion])\n\ntabRevenue = TabItem(\"revenue\", \"Revenue vs Quota\", [revenueChart])\nrevenueChart = HorizontalBarChart(repNames, [seriesRevenue, seriesQuota], \"grouped\", \"USD\", \"Rep\")\nseriesRevenue = Series(\"Closed Revenue\", [1024000, 918000, 735000, 842000, 681000, 488000, 604000, 402000])\nseriesQuota = Series(\"Quota\", [850000, 800000, 650000, 750000, 700000, 550000, 700000, 600000])\n\ntabAttain = TabItem(\"attain\", \"Attainment %\", [attainChart])\nattainChart = BarChart(repNames, [seriesAttain], \"grouped\", \"Rep\", \"% of quota\")\nseriesAttain = Series(\"Attainment\", [120, 115, 113, 112, 97, 89, 86, 67])\n\ntabRegion = TabItem(\"region\", \"Region Split\", [regionChart, regionNote])\nregionChart = PieChart([\"EMEA\", \"APAC\"], [2949000, 2745000], \"donut\")\nregionNote = TextContent(\"EMEA closed **$2.95M** (52%) across 50 deals; APAC closed **$2.75M** (48%) across 53 deals — APAC runs a higher-volume, smaller-deal motion.\", \"small\")\n\nfootnote = Callout(\"info\", \"Coaching focus for Q4\", \"Jonas Weber (67%) had two enterprise deals slip to October worth ~$210K combined. Daniel Park and Grace Lim are both trending to quota if current pipeline coverage (3.1x) holds.\")","promptTokens":6999,"completionTokens":2571,"latencyMs":27177,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/realistic","repeat":1,"output":"root = Stack([header, kpis, tabs, notes], \"column\", \"l\")\n\nheader = CardHeader(\"Q3 FY25 — Sales Team Results\", \"8 quota-carrying reps across EMEA & APAC · closed-won bookings, new ARR\")\n\nkpis = Stack([kpiRevenue, kpiDeals, kpiAttain, kpiAvg], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiRevenue = Card([kpiRevenueLabel, kpiRevenueValue, kpiRevenueSub])\nkpiRevenueLabel = TextContent(\"Total New ARR\", \"small\")\nkpiRevenueValue = TextContent(\"$5.34M\", \"large-heavy\")\nkpiRevenueSub = TextContent(\"vs $5.00M team quota\", \"small\")\n\nkpiDeals = Card([kpiDealsLabel, kpiDealsValue, kpiDealsSub])\nkpiDealsLabel = TextContent(\"Deals Closed\", \"small\")\nkpiDealsValue = TextContent(\"93\", \"large-heavy\")\nkpiDealsSub = TextContent(\"+12 vs Q2\", \"small\")\n\nkpiAttain = Card([kpiAttainLabel, kpiAttainValue, kpiAttainSub])\nkpiAttainLabel = TextContent(\"Team Attainment\", \"small\")\nkpiAttainValue = TextContent(\"107%\", \"large-heavy\")\nkpiAttainSub = TextContent(\"5 of 8 reps at or above quota\", \"small\")\n\nkpiAvg = Card([kpiAvgLabel, kpiAvgValue, kpiAvgSub])\nkpiAvgLabel = TextContent(\"Avg Deal Size\", \"small\")\nkpiAvgValue = TextContent(\"$57.4K\", \"large-heavy\")\nkpiAvgSub = TextContent(\"Down 4% QoQ\", \"small\")\n\ntabs = Tabs([tabTable, tabRevenue, tabRegion])\n\ntabTable = TabItem(\"table\", \"Rep Detail\", [repTable])\ntabRevenue = TabItem(\"revenue\", \"Revenue by Rep\", [revenueChart])\ntabRegion = TabItem(\"region\", \"Region Split\", [regionChart, regionBar])\n\nrepTable = Table([colRep, colRegion, colDeals, colRevenue, colQuota, colAttain, colStatus])\n\ncolRep = Col(\"Rep\", repNames)\ncolRegion = Col(\"Region\", regions)\ncolDeals = Col(\"Deals Closed\", deals, \"number\")\ncolRevenue = Col(\"New ARR\", revenueLabels)\ncolQuota = Col(\"Quota\", quotaLabels)\ncolAttain = Col(\"% of Quota\", attainLabels)\ncolStatus = Col(\"Status\", statusTags)\n\nrepNames = [\"Sofia Marchetti\", \"Priya Raghavan\", \"Amara Okafor\", \"Wei Chen\", \"Lukas Brandt\", \"Kenji Sato\", \"Daniel Nyberg\", \"Mei Ling Tan\"]\nregions = [\"EMEA\", \"APAC\", \"EMEA\", \"APAC\", \"EMEA\", \"APAC\", \"EMEA\", \"APAC\"]\ndeals = [17, 15, 14, 13, 11, 9, 8, 6]\nrevenueLabels = [\"$1,024,000\", \"$868,500\", \"$842,000\", \"$735,000\", \"$610,500\", \"$521,000\", \"$398,000\", \"$342,000\"]\nquotaLabels = [\"$780,000\", \"$700,000\", \"$715,000\", \"$675,000\", \"$650,000\", \"$600,000\", \"$560,000\", \"$540,000\"]\nattainLabels = [\"131%\", \"124%\", \"118%\", \"109%\", \"94%\", \"87%\", \"71%\", \"63%\"]\n\nstatusTags = [tagSofia, tagPriya, tagAmara, tagWei, tagLukas, tagKenji, tagDaniel, tagMei]\ntagSofia = Tag(\"Exceeded\", null, \"sm\", \"success\")\ntagPriya = Tag(\"Exceeded\", null, \"sm\", \"success\")\ntagAmara = Tag(\"Exceeded\", null, \"sm\", \"success\")\ntagWei = Tag(\"Exceeded\", null, \"sm\", \"success\")\ntagLukas = Tag(\"Near quota\", null, \"sm\", \"warning\")\ntagKenji = Tag(\"Near quota\", null, \"sm\", \"warning\")\ntagDaniel = Tag(\"Missed\", null, \"sm\", \"danger\")\ntagMei = Tag(\"Missed\", null, \"sm\", \"danger\")\n\nrevenueChart = HorizontalBarChart(repNames, [revenueSeries, quotaSeries], \"grouped\", \"USD (thousands)\", \"Rep\")\nrevenueSeries = Series(\"New ARR\", [1024, 868.5, 842, 735, 610.5, 521, 398, 342])\nquotaSeries = Series(\"Quota\", [780, 700, 715, 675, 650, 600, 560, 540])\n\nregionChart = PieChart([\"EMEA\", \"APAC\"], [2874.5, 2466.5], \"donut\")\nregionBar = BarChart([\"EMEA\", \"APAC\"], [regionActual, regionQuota], \"grouped\", \"Region\", \"USD (thousands)\")\nregionActual = Series(\"New ARR\", [2874.5, 2466.5])\nregionQuota = Series(\"Quota\", [2705, 2515])\n\nnotes = Card([notesHeader, notesBody], \"sunk\")\nnotesHeader = CardHeader(\"Q3 Takeaways\")\nnotesBody = MarkDownRenderer(\"- **EMEA** finished at **106%** of quota ($2.87M / $2.71M), carried by Sofia Marchetti's two seven-figure-pipeline enterprise renewals.\\n- **APAC** finished at **98%** of quota ($2.47M / $2.52M) — Mei Ling Tan (63%) is in her second quarter post-onboarding and Kenji Sato slipped two deals into Q4 worth ~$130K.\\n- Average deal size fell to **$57.4K** from $59.8K in Q2 as more mid-market logos closed relative to enterprise.\\n- **Daniel Nyberg (71%)** and **Mei Ling Tan (63%)** should move onto a 60-day coaching plan; combined gap to quota is **$260K**.\", \"clear\")","promptTokens":6999,"completionTokens":2716,"latencyMs":29009,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/realistic","repeat":2,"output":"root = Stack([header, kpiRow, resultsCard, chartTabs, footnote], \"column\", \"l\")\n\nheader = CardHeader(\"Q3 Sales Team Results\", \"8 quota-carrying reps across EMEA & APAC — quarter closed 30 Sep\")\n\nkpiRow = Stack([kpiRevenue, kpiDeals, kpiAttain, kpiOverQuota], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiRevenue = Card([kpiRevenueLabel, kpiRevenueValue, kpiRevenueSub])\nkpiRevenueLabel = TextContent(\"Closed Revenue\", \"small\")\nkpiRevenueValue = TextContent(\"$4.36M\", \"large-heavy\")\nkpiRevenueSub = TextContent(\"vs $4.30M target\", \"small\")\n\nkpiDeals = Card([kpiDealsLabel, kpiDealsValue, kpiDealsSub])\nkpiDealsLabel = TextContent(\"Deals Closed\", \"small\")\nkpiDealsValue = TextContent(\"99\", \"large-heavy\")\nkpiDealsSub = TextContent(\"Avg deal size $44.1K\", \"small\")\n\nkpiAttain = Card([kpiAttainLabel, kpiAttainValue, kpiAttainSub])\nkpiAttainLabel = TextContent(\"Team Attainment\", \"small\")\nkpiAttainValue = TextContent(\"101%\", \"large-heavy\")\nkpiAttainSub = TextContent(\"+3 pts vs Q2\", \"small\")\n\nkpiOverQuota = Card([kpiOverLabel, kpiOverValue, kpiOverSub])\nkpiOverLabel = TextContent(\"Reps At/Over Quota\", \"small\")\nkpiOverValue = TextContent(\"5 of 8\", \"large-heavy\")\nkpiOverSub = TextContent(\"2 reps below 85%\", \"small\")\n\nresultsCard = Card([resultsHeader, repTable])\nresultsHeader = CardHeader(\"Rep Scorecard\", \"Sorted by quota attainment\")\nrepTable = Table([colRep, colRegion, colDeals, colRevenue, colQuota, colAttain])\n\ncolRep = Col(\"Rep\", repNames)\ncolRegion = Col(\"Region\", regionTags)\ncolDeals = Col(\"Deals\", dealCounts, \"number\")\ncolRevenue = Col(\"Revenue\", revenueLabels)\ncolQuota = Col(\"Quota\", quotaLabels)\ncolAttain = Col(\"vs Quota\", attainTags)\n\nrepNames = [\"Amara Okafor\", \"Wei Chen\", \"Sofia Almeida\", \"Priya Nair\", \"Mei-Lin Tan\", \"Lukas Meyer\", \"Kenji Sato\", \"Daniel Roche\"]\n\nregionTags = [tagEmea1, tagApac1, tagEmea2, tagApac2, tagApac3, tagEmea3, tagApac4, tagEmea4]\ntagEmea1 = Tag(\"EMEA\", null, \"sm\", \"info\")\ntagApac1 = Tag(\"APAC\", null, \"sm\", \"neutral\")\ntagEmea2 = Tag(\"EMEA\", null, \"sm\", \"info\")\ntagApac2 = Tag(\"APAC\", null, \"sm\", \"neutral\")\ntagApac3 = Tag(\"APAC\", null, \"sm\", \"neutral\")\ntagEmea3 = Tag(\"EMEA\", null, \"sm\", \"info\")\ntagApac4 = Tag(\"APAC\", null, \"sm\", \"neutral\")\ntagEmea4 = Tag(\"EMEA\", null, \"sm\", \"info\")\n\ndealCounts = [17, 15, 14, 12, 13, 11, 9, 8]\nrevenueLabels = [\"$735,000\", \"$658,000\", \"$612,000\", \"$521,000\", \"$549,000\", \"$487,500\", \"$438,000\", \"$362,000\"]\nquotaLabels = [\"$600,000\", \"$575,000\", \"$550,000\", \"$500,000\", \"$550,000\", \"$500,000\", \"$525,000\", \"$500,000\"]\n\nattainTags = [att1, att2, att3, att4, att5, att6, att7, att8]\natt1 = Tag(\"123%\", null, \"sm\", \"success\")\natt2 = Tag(\"114%\", null, \"sm\", \"success\")\natt3 = Tag(\"111%\", null, \"sm\", \"success\")\natt4 = Tag(\"104%\", null, \"sm\", \"success\")\natt5 = Tag(\"100%\", null, \"sm\", \"success\")\natt6 = Tag(\"98%\", null, \"sm\", \"warning\")\natt7 = Tag(\"83%\", null, \"sm\", \"danger\")\natt8 = Tag(\"72%\", null, \"sm\", \"danger\")\n\nchartTabs = Tabs([tabRevenue, tabAttain, tabRegion])\n\ntabRevenue = TabItem(\"revenue\", \"Revenue by Rep\", [revenueChart])\nrevenueChart = HorizontalBarChart(repNames, [revenueSeries], \"grouped\", \"Revenue (USD)\", \"Rep\")\nrevenueSeries = Series(\"Q3 Closed Revenue\", [735000, 658000, 612000, 521000, 549000, 487500, 438000, 362000])\n\ntabAttain = TabItem(\"attain\", \"Revenue vs Quota\", [attainChart])\nattainChart = BarChart(repNames, [seriesClosed, seriesQuota], \"grouped\", \"Rep\", \"USD\")\nseriesClosed = Series(\"Closed\", [735000, 658000, 612000, 521000, 549000, 487500, 438000, 362000])\nseriesQuota = Series(\"Quota\", [600000, 575000, 550000, 500000, 550000, 500000, 525000, 500000])\n\ntabRegion = TabItem(\"region\", \"Regional Split\", [regionPie, regionNote])\nregionPie = PieChart([\"EMEA\", \"APAC\"], [2196500, 2166000], \"donut\")\nregionNote = TextContent(\"EMEA closed **$2.20M** across 50 deals (101% of a $2.15M target); APAC closed **$2.17M** across 49 deals (100% of a $2.15M target).\")\n\nfootnote = Callout(\"info\", \"Coverage note\", \"Kenji Sato and Daniel Roche both carried territories with mid-quarter account reassignments — 3 deals worth ~$180K slipped into Q4 pipeline and are forecast to close by 15 Nov.\")","promptTokens":6999,"completionTokens":2755,"latencyMs":28446,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/realistic","repeat":4,"output":"root = Stack([header, kpiRow, tabs, notesCard], \"column\", \"l\")\n\nheader = CardHeader(\"Q3 FY25 Sales Team Results\", \"8 quota-carrying reps across EMEA & APAC — quarter closed Sept 30\")\n\nkpiRow = Stack([kpiRevenue, kpiDeals, kpiQuota, kpiAsp], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiRevenue = Card([kpiRevenueLabel, kpiRevenueValue, kpiRevenueSub])\nkpiRevenueLabel = TextContent(\"Closed Revenue\", \"small\")\nkpiRevenueValue = TextContent(\"$4.14M\", \"large-heavy\")\nkpiRevenueSub = TextContent(\"vs $4.17M team quota\", \"small\")\n\nkpiDeals = Card([kpiDealsLabel, kpiDealsValue, kpiDealsSub])\nkpiDealsLabel = TextContent(\"Deals Closed\", \"small\")\nkpiDealsValue = TextContent(\"99\", \"large-heavy\")\nkpiDealsSub = TextContent(\"+12 vs Q2\", \"small\")\n\nkpiQuota = Card([kpiQuotaLabel, kpiQuotaValue, kpiQuotaSub])\nkpiQuotaLabel = TextContent(\"Avg Quota Attainment\", \"small\")\nkpiQuotaValue = TextContent(\"99.4%\", \"large-heavy\")\nkpiQuotaSub = TextContent(\"5 of 8 reps at or above plan\", \"small\")\n\nkpiAsp = Card([kpiAspLabel, kpiAspValue, kpiAspSub])\nkpiAspLabel = TextContent(\"Average Deal Size\", \"small\")\nkpiAspValue = TextContent(\"$41.9K\", \"large-heavy\")\nkpiAspSub = TextContent(\"EMEA $45.6K / APAC $38.6K\", \"small\")\n\ntabs = Tabs([tabTable, tabRevenue, tabMix])\n\ntabTable = TabItem(\"table\", \"Rep Detail\", [repTable])\ntabRevenue = TabItem(\"revenue\", \"Revenue by Rep\", [revenueChart])\ntabMix = TabItem(\"mix\", \"Region Mix\", [regionPie, regionBar])\n\nrepTable = Table([colRep, colRegion, colDeals, colRevenue, colQuotaTarget, colAttainment])\n\ncolRep = Col(\"Rep\", repNames)\ncolRegion = Col(\"Region\", regionTags)\ncolDeals = Col(\"Deals Closed\", dealCounts, \"number\")\ncolRevenue = Col(\"Revenue\", revenueLabels)\ncolQuotaTarget = Col(\"Quota\", quotaLabels)\ncolAttainment = Col(\"vs Quota\", attainmentTags)\n\nrepNames = [\"Sofia Marchetti\", \"Priya Raghavan\", \"Anna Kowalski\", \"Wei Chen\", \"Grace Tan\", \"Liam O'Sullivan\", \"Tobias Lindqvist\", \"Kenji Nakamura\"]\n\nregionTags = [tagEmea1, tagApac1, tagEmea2, tagApac2, tagApac3, tagEmea3, tagEmea4, tagApac4]\ntagEmea1 = Tag(\"EMEA\", null, \"sm\", \"info\")\ntagApac1 = Tag(\"APAC\", null, \"sm\", \"neutral\")\ntagEmea2 = Tag(\"EMEA\", null, \"sm\", \"info\")\ntagApac2 = Tag(\"APAC\", null, \"sm\", \"neutral\")\ntagApac3 = Tag(\"APAC\", null, \"sm\", \"neutral\")\ntagEmea3 = Tag(\"EMEA\", null, \"sm\", \"info\")\ntagEmea4 = Tag(\"EMEA\", null, \"sm\", \"info\")\ntagApac4 = Tag(\"APAC\", null, \"sm\", \"neutral\")\n\ndealCounts = [17, 15, 14, 13, 12, 11, 9, 8]\n\nrevenueLabels = [\"$743,000\", \"$665,000\", \"$612,000\", \"$528,000\", \"$441,000\", \"$487,500\", \"$356,000\", \"$312,000\"]\n\nquotaLabels = [\"$600,000\", \"$565,000\", \"$565,000\", \"$500,000\", \"$450,000\", \"$530,000\", \"$450,000\", \"$440,000\"]\n\nattainmentTags = [att1, att2, att3, att4, att5, att6, att7, att8]\natt1 = Tag(\"124%\", null, \"sm\", \"success\")\natt2 = Tag(\"118%\", null, \"sm\", \"success\")\natt3 = Tag(\"108%\", null, \"sm\", \"success\")\natt4 = Tag(\"106%\", null, \"sm\", \"success\")\natt5 = Tag(\"98%\", null, \"sm\", \"warning\")\natt6 = Tag(\"92%\", null, \"sm\", \"warning\")\natt7 = Tag(\"79%\", null, \"sm\", \"danger\")\natt8 = Tag(\"71%\", null, \"sm\", \"danger\")\n\nrevenueChart = HorizontalBarChart(repNames, [revenueSeries, quotaSeries], \"grouped\", \"USD\", \"Rep\")\nrevenueSeries = Series(\"Closed Revenue\", [743000, 665000, 612000, 528000, 441000, 487500, 356000, 312000])\nquotaSeries = Series(\"Quota\", [600000, 565000, 565000, 500000, 450000, 530000, 450000, 440000])\n\nregionPie = PieChart([\"EMEA\", \"APAC\"], [2198500, 1946000], \"donut\")\n\nregionBar = BarChart([\"Deals\", \"Reps at Plan\"], [emeaSeries, apacSeries], \"grouped\")\nemeaSeries = Series(\"EMEA\", [51, 2])\napacSeries = Series(\"APAC\", [48, 3])\n\nnotesCard = Card([notesHeader, notesText, notesTags])\nnotesHeader = CardHeader(\"Quarter Notes\", \"Manager commentary\")\nnotesText = TextContent(\"**Sofia Marchetti** closed the quarter's largest deal ($198K, 3-year platform renewal with a German logistics group) and carried EMEA past plan. **Priya Raghavan** posted the strongest new-logo count in APAC with six first-time customers.\\n\\nTwo reps finished below 80%: **Tobias Lindqvist** had two enterprise deals slip to Q4 (~$140K combined, both verbal commit), and **Kenji Nakamura** is in his second quarter post-ramp — pipeline coverage is now 3.4x heading into Q4.\")\nnotesTags = TagBlock([\"Q3 FY25\", \"EMEA\", \"APAC\", \"99 deals\", \"$4.14M closed\", \"3 deals slipped\"])","promptTokens":6999,"completionTokens":2929,"latencyMs":30958,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/realistic","repeat":3,"output":"root = Stack([header, kpiRow, mainCard, chartTabs, notesCard])\n\nheader = CardHeader(\"Q3 FY24 Sales Team Performance\", \"8 quota-carrying reps · EMEA & APAC · closed-won bookings, new ARR\")\n\nkpiRow = Stack([kpi1, kpi2, kpi3, kpi4], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpi1 = Card([kpi1Label, kpi1Value, kpi1Sub])\nkpi1Label = TextContent(\"Total Revenue\", \"small\")\nkpi1Value = TextContent(\"$4.63M\", \"large-heavy\")\nkpi1Sub = TextContent(\"vs $4.30M quota\", \"small\")\n\nkpi2 = Card([kpi2Label, kpi2Value, kpi2Sub])\nkpi2Label = TextContent(\"Team Attainment\", \"small\")\nkpi2Value = TextContent(\"108%\", \"large-heavy\")\nkpi2Sub = TextContent(\"+8 pts over plan\", \"small\")\n\nkpi3 = Card([kpi3Label, kpi3Value, kpi3Sub])\nkpi3Label = TextContent(\"Deals Closed\", \"small\")\nkpi3Value = TextContent(\"106\", \"large-heavy\")\nkpi3Sub = TextContent(\"avg $43.6K per deal\", \"small\")\n\nkpi4 = Card([kpi4Label, kpi4Value, kpi4Sub])\nkpi4Label = TextContent(\"Reps at Quota\", \"small\")\nkpi4Value = TextContent(\"5 of 8\", \"large-heavy\")\nkpi4Sub = TextContent(\"3 below plan\", \"small\")\n\nmainCard = Card([tableHeader, resultsTable])\ntableHeader = CardHeader(\"Rep Scorecard\", \"Sorted by attainment, highest first\")\n\nresultsTable = Table([colRep, colRegion, colDeals, colRevenue, colQuota, colAttain, colStatus])\n\ncolRep = Col(\"Rep\", repNames)\ncolRegion = Col(\"Region\", regions)\ncolDeals = Col(\"Deals Closed\", deals, \"number\")\ncolRevenue = Col(\"Revenue\", revenueLabels)\ncolQuota = Col(\"Quota\", quotaLabels)\ncolAttain = Col(\"% of Quota\", attainLabels)\ncolStatus = Col(\"Status\", statusTags)\n\nrepNames = [\"Sofia Marchetti\", \"Wei Chen\", \"Grace Tan\", \"Anna Kowalski\", \"Priya Raghavan\", \"Liam O'Sullivan\", \"Kenji Nakamura\", \"Tobias Berger\"]\nregions = [\"EMEA\", \"APAC\", \"APAC\", \"EMEA\", \"APAC\", \"EMEA\", \"APAC\", \"EMEA\"]\ndeals = [18, 16, 15, 14, 13, 11, 10, 9]\nrevenueLabels = [\"$824K\", \"$702K\", \"$667K\", \"$612K\", \"$548K\", \"$485K\", \"$431K\", \"$356K\"]\nquotaLabels = [\"$640K\", \"$600K\", \"$600K\", \"$600K\", \"$570K\", \"$550K\", \"$545K\", \"$500K\"]\nattainLabels = [\"129%\", \"117%\", \"111%\", \"102%\", \"96%\", \"88%\", \"79%\", \"71%\"]\n\nstatusTags = [tagOver, tagOver, tagOver, tagMet, tagClose, tagUnder, tagUnder, tagRisk]\ntagOver = Tag(\"Over quota\", null, \"sm\", \"success\")\ntagMet = Tag(\"Met quota\", null, \"sm\", \"success\")\ntagClose = Tag(\"Just short\", null, \"sm\", \"warning\")\ntagUnder = Tag(\"Below quota\", null, \"sm\", \"warning\")\ntagRisk = Tag(\"Needs plan\", null, \"sm\", \"danger\")\n\nchartTabs = Tabs([tabRevenue, tabAttain, tabRegion, tabDeals])\n\ntabRevenue = TabItem(\"revenue\", \"Revenue by Rep\", [revenueChart])\nrevenueChart = HorizontalBarChart(repNames, [revenueSeries, quotaSeries], \"grouped\", \"USD (thousands)\", \"Rep\")\nrevenueSeries = Series(\"Closed Revenue\", [824, 702, 667, 612, 548, 485, 431, 356])\nquotaSeries = Series(\"Quota\", [640, 600, 600, 600, 570, 550, 545, 500])\n\ntabAttain = TabItem(\"attain\", \"Attainment\", [attainChart])\nattainChart = BarChart(repNames, [attainSeries], \"grouped\", \"Rep\", \"% of quota\")\nattainSeries = Series(\"% of Quota\", [129, 117, 111, 102, 96, 88, 79, 71])\n\ntabRegion = TabItem(\"region\", \"Region Split\", [regionChart, regionNote])\nregionChart = PieChart([\"EMEA\", \"APAC\"], [2277, 2348], \"donut\")\nregionNote = TextContent(\"EMEA closed **$2.28M** across 52 deals (avg $43.8K). APAC closed **$2.35M** across 54 deals (avg $43.5K). APAC edged ahead for the first time this fiscal year.\", \"small\")\n\ntabDeals = TabItem(\"deals\", \"Deal Volume\", [dealsChart])\ndealsChart = HorizontalBarChart(repNames, [dealsSeries], \"grouped\", \"Deals closed\", \"Rep\")\ndealsSeries = Series(\"Deals Closed\", [18, 16, 15, 14, 13, 11, 10, 9])\n\nnotesCard = Card([notesHeader, notesBody, coachingSteps, actionButtons], \"sunk\")\nnotesHeader = CardHeader(\"Takeaways & Next Steps\", \"Manager review notes for Q4 planning\")\nnotesBody = TextContent(\"Top of the board: **Sofia Marchetti** (129%) landed two enterprise expansions in DACH worth $310K combined. **Wei Chen** carried APAC with a $195K multi-year renewal in Singapore.\\n\\nSoft spots: **Kenji Nakamura** and **Tobias Berger** both slipped late-stage deals into Q4 — roughly $180K of committed pipeline pushed rather than lost.\", \"default\")\n\ncoachingSteps = Steps([step1, step2, step3])\nstep1 = StepsItem(\"Recover pushed pipeline\", \"Run joint deal reviews with Kenji and Tobias on the four late-stage opportunities that slipped from Q3. Target close by week 3 of Q4.\")\nstep2 = StepsItem(\"Rebalance EMEA territories\", \"Sofia is carrying 36% of EMEA revenue on 640K quota. Reassign two mid-market accounts to Liam to spread coverage and lift his attainment.\")\nstep3 = StepsItem(\"Set Q4 quotas\", \"Team quota rises to $4.75M. Raise Sofia and Wei by 8%, hold Kenji and Tobias flat pending recovery on Q3 slippage.\")\n\nactionButtons = Buttons([btnRegion, btnPipeline, btnQuota])\nbtnRegion = Button(\"Break down EMEA vs APAC pipeline\", Action([@ToAssistant(\"Break down the Q3 EMEA vs APAC pipeline in more detail, including win rates and average sales cycle by region.\")]), \"primary\")\nbtnPipeline = Button(\"Show the deals that slipped to Q4\", Action([@ToAssistant(\"Show me the late-stage deals that slipped from Q3 to Q4, with rep, account, value and stage.\")]), \"secondary\")\nbtnQuota = Button(\"Draft Q4 quota proposal\", Action([@ToAssistant(\"Draft a Q4 quota proposal for all eight reps based on their Q3 attainment and territory coverage.\")]), \"secondary\")","promptTokens":6999,"completionTokens":3184,"latencyMs":35661,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/minimal","repeat":0,"output":"root = Stack([header, chartCard, summaryCard])\n\nheader = CardHeader(\"Monthly Revenue\", \"Last 6 months — Jan to Jun 2024\")\n\nchartCard = Card([revenueChart])\nrevenueChart = BarChart(months, [revenueSeries], \"grouped\", \"Month\", \"Revenue (USD)\")\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"]\nrevenueSeries = Series(\"Revenue\", [128000, 142500, 137800, 165300, 178900, 194200])\n\nsummaryCard = Card([summaryHeader, kpiRow, revenueTable], \"sunk\")\nsummaryHeader = CardHeader(\"Summary\", \"Totals and growth\")\n\nkpiRow = Stack([kpiTotal, kpiAvg, kpiBest, kpiGrowth], \"row\", \"l\", \"stretch\", \"start\", true)\n\nkpiTotal = Card([kpiTotalLabel, kpiTotalValue])\nkpiTotalLabel = TextContent(\"Total Revenue\", \"small\")\nkpiTotalValue = TextContent(\"$946,700\", \"large-heavy\")\n\nkpiAvg = Card([kpiAvgLabel, kpiAvgValue])\nkpiAvgLabel = TextContent(\"Monthly Average\", \"small\")\nkpiAvgValue = TextContent(\"$157,783\", \"large-heavy\")\n\nkpiBest = Card([kpiBestLabel, kpiBestValue])\nkpiBestLabel = TextContent(\"Best Month\", \"small\")\nkpiBestValue = TextContent(\"Jun · $194,200\", \"large-heavy\")\n\nkpiGrowth = Card([kpiGrowthLabel, kpiGrowthValue])\nkpiGrowthLabel = TextContent(\"Growth (Jan→Jun)\", \"small\")\nkpiGrowthValue = TextContent(\"+51.7%\", \"large-heavy\")\n\nrevenueTable = Table([colMonth, colRevenue, colChange])\ncolMonth = Col(\"Month\", months)\ncolRevenue = Col(\"Revenue\", [128000, 142500, 137800, 165300, 178900, 194200], \"number\")\ncolChange = Col(\"MoM Change\", changeTags)\nchangeTags = [tagJan, tagFeb, tagMar, tagApr, tagMay, tagJun]\ntagJan = Tag(\"—\", null, \"sm\", \"neutral\")\ntagFeb = Tag(\"+11.3%\", null, \"sm\", \"success\")\ntagMar = Tag(\"-3.3%\", null, \"sm\", \"danger\")\ntagApr = Tag(\"+19.9%\", null, \"sm\", \"success\")\ntagMay = Tag(\"+8.2%\", null, \"sm\", \"success\")\ntagJun = Tag(\"+8.6%\", null, \"sm\", \"success\")","promptTokens":6952,"completionTokens":931,"latencyMs":9857,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/minimal","repeat":1,"output":"root = Stack([header, chartCard, statsRow, note])\n\nheader = CardHeader(\"Monthly Revenue\", \"Last 6 months · figures in USD\")\n\nchartCard = Card([chartHeader, revenueChart])\nchartHeader = CardHeader(\"Revenue by Month\", \"Apr 2024 – Sep 2024\")\nrevenueChart = BarChart(months, [revenueSeries], \"grouped\", \"Month\", \"Revenue (USD)\")\nmonths = [\"Apr\", \"May\", \"Jun\", \"Jul\", \"Aug\", \"Sep\"]\nrevenueSeries = Series(\"Revenue\", revenueValues)\nrevenueValues = [128400, 141200, 137800, 158600, 172300, 189500]\n\nstatsRow = Stack([totalCard, avgCard, bestCard, growthCard], \"row\", \"m\", \"stretch\", \"start\", true)\n\ntotalCard = Card([totalLabel, totalValue])\ntotalLabel = TextContent(\"Total Revenue\", \"small\")\ntotalValue = TextContent(\"$927,800\", \"large-heavy\")\n\navgCard = Card([avgLabel, avgValue])\navgLabel = TextContent(\"Monthly Average\", \"small\")\navgValue = TextContent(\"$154,633\", \"large-heavy\")\n\nbestCard = Card([bestLabel, bestValue])\nbestLabel = TextContent(\"Best Month\", \"small\")\nbestValue = TextContent(\"Sep · $189,500\", \"large-heavy\")\n\ngrowthCard = Card([growthLabel, growthValue])\ngrowthLabel = TextContent(\"Growth vs Apr\", \"small\")\ngrowthValue = TextContent(\"+47.6%\", \"large-heavy\")\n\nnote = TextCallout(\"success\", \"Steady upward trend\", \"Revenue grew in 5 of the last 6 months, with a small dip in June. Q3 (Jul–Sep) accounted for 56% of the period total.\")","promptTokens":6952,"completionTokens":652,"latencyMs":9426,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/minimal","repeat":2,"output":"root = Stack([header, chartCard, kpiRow, tableCard, note])\n\nheader = CardHeader(\"Monthly Revenue\", \"Last 6 months — total and growth trend\")\n\nchartCard = Card([chartHeader, revenueChart])\nchartHeader = CardHeader(\"Revenue by Month\", \"USD, net of refunds\")\nrevenueChart = BarChart(months, [revenueSeries], \"grouped\", \"Month\", \"Revenue (USD)\")\n\nmonths = [\"Apr\", \"May\", \"Jun\", \"Jul\", \"Aug\", \"Sep\"]\nrevenueSeries = Series(\"Revenue\", revenueValues)\nrevenueValues = [128400, 141200, 136900, 158300, 172500, 189700]\n\nkpiRow = Stack([kpiTotal, kpiBest, kpiGrowth], \"row\", \"m\", \"stretch\")\n\nkpiTotal = Card([kpiTotalLabel, kpiTotalValue])\nkpiTotalLabel = TextContent(\"6-Month Total\", \"small\")\nkpiTotalValue = TextContent(\"$927,000\", \"large-heavy\")\n\nkpiBest = Card([kpiBestLabel, kpiBestValue])\nkpiBestLabel = TextContent(\"Best Month\", \"small\")\nkpiBestValue = TextContent(\"Sep — $189.7K\", \"large-heavy\")\n\nkpiGrowth = Card([kpiGrowthLabel, kpiGrowthValue])\nkpiGrowthLabel = TextContent(\"Apr → Sep Growth\", \"small\")\nkpiGrowthValue = TextContent(\"+47.7%\", \"large-heavy\")\n\ntableCard = Card([tableHeader, revenueTable])\ntableHeader = CardHeader(\"Monthly Detail\", \"Revenue and month-over-month change\")\nrevenueTable = Table([colMonth, colRevenue, colChange])\ncolMonth = Col(\"Month\", months)\ncolRevenue = Col(\"Revenue (USD)\", revenueValues, \"number\")\ncolChange = Col(\"MoM Change\", changeTags)\nchangeTags = [tagApr, tagMay, tagJun, tagJul, tagAug, tagSep]\ntagApr = Tag(\"—\", null, \"sm\", \"neutral\")\ntagMay = Tag(\"+10.0%\", null, \"sm\", \"success\")\ntagJun = Tag(\"-3.0%\", null, \"sm\", \"danger\")\ntagJul = Tag(\"+15.6%\", null, \"sm\", \"success\")\ntagAug = Tag(\"+9.0%\", null, \"sm\", \"success\")\ntagSep = Tag(\"+10.0%\", null, \"sm\", \"success\")\n\nnote = TextCallout(\"info\", \"Trend\", \"Revenue dipped slightly in June, then posted three consecutive months of growth, closing September at an all-time high.\")","promptTokens":6952,"completionTokens":937,"latencyMs":10651,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/minimal","repeat":3,"output":"root = Stack([header, chartCard, statsRow, insight])\n\nheader = CardHeader(\"Monthly Revenue\", \"Last 6 months · figures in USD\")\n\nchartCard = Card([revenueChart])\n\nrevenueChart = BarChart(months, [revenueSeries], \"grouped\", \"Month\", \"Revenue (USD)\")\n\nmonths = [\"Apr\", \"May\", \"Jun\", \"Jul\", \"Aug\", \"Sep\"]\n\nrevenueSeries = Series(\"Revenue\", [128400, 141200, 136800, 159500, 172300, 188900])\n\nstatsRow = Stack([totalCard, avgCard, bestCard, growthCard], \"row\", \"m\", \"stretch\", \"start\", true)\n\ntotalCard = Card([totalLabel, totalValue])\ntotalLabel = TextContent(\"Total (6 mo)\", \"small\")\ntotalValue = TextContent(\"$927,100\", \"large-heavy\")\n\navgCard = Card([avgLabel, avgValue])\navgLabel = TextContent(\"Monthly Average\", \"small\")\navgValue = TextContent(\"$154,517\", \"large-heavy\")\n\nbestCard = Card([bestLabel, bestValue])\nbestLabel = TextContent(\"Best Month\", \"small\")\nbestValue = TextContent(\"Sep · $188.9K\", \"large-heavy\")\n\ngrowthCard = Card([growthLabel, growthValue])\ngrowthLabel = TextContent(\"Apr → Sep Growth\", \"small\")\ngrowthValue = TextContent(\"+47.1%\", \"large-heavy\")\n\ninsight = TextCallout(\"success\", \"Steady upward trend\", \"Revenue grew in 5 of the last 6 months, with a single dip in June (-3.1%). Momentum accelerated over Q3, adding roughly $16K per month on average since July.\")","promptTokens":6952,"completionTokens":599,"latencyMs":7964,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/minimal","repeat":4,"output":"root = Stack([header, chartCard, kpiRow, tableCard, note])\n\nheader = CardHeader(\"Monthly Revenue\", \"Last 6 months · figures in USD\")\n\nchartCard = Card([chartHeader, revenueChart])\nchartHeader = CardHeader(\"Revenue by Month\", \"Apr 2024 – Sep 2024\")\nrevenueChart = BarChart(months, [revenueSeries], \"grouped\", \"Month\", \"Revenue (USD)\")\nrevenueSeries = Series(\"Revenue\", revenueValues)\n\nmonths = [\"Apr\", \"May\", \"Jun\", \"Jul\", \"Aug\", \"Sep\"]\nrevenueValues = [182400, 196800, 188200, 214500, 236900, 251300]\n\nkpiRow = Stack([kpiTotal, kpiAvg, kpiBest, kpiGrowth], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiTotal = Card([kpiTotalLabel, kpiTotalValue])\nkpiTotalLabel = TextContent(\"Total Revenue\", \"small\")\nkpiTotalValue = TextContent(\"$1,270,100\", \"large-heavy\")\n\nkpiAvg = Card([kpiAvgLabel, kpiAvgValue])\nkpiAvgLabel = TextContent(\"Monthly Average\", \"small\")\nkpiAvgValue = TextContent(\"$211,683\", \"large-heavy\")\n\nkpiBest = Card([kpiBestLabel, kpiBestValue])\nkpiBestLabel = TextContent(\"Best Month\", \"small\")\nkpiBestValue = TextContent(\"Sep · $251,300\", \"large-heavy\")\n\nkpiGrowth = Card([kpiGrowthLabel, kpiGrowthValue])\nkpiGrowthLabel = TextContent(\"Growth (Apr → Sep)\", \"small\")\nkpiGrowthValue = TextContent(\"+37.8%\", \"large-heavy\")\n\ntableCard = Card([tableHeader, revenueTable])\ntableHeader = CardHeader(\"Monthly Detail\", \"Revenue and month-over-month change\")\nrevenueTable = Table([colMonth, colRevenue, colChange])\ncolMonth = Col(\"Month\", months)\ncolRevenue = Col(\"Revenue\", revenueValues, \"number\")\ncolChange = Col(\"MoM Change\", changeTags)\nchangeTags = [tagApr, tagMay, tagJun, tagJul, tagAug, tagSep]\ntagApr = Tag(\"—\", null, \"sm\", \"neutral\")\ntagMay = Tag(\"+7.9%\", null, \"sm\", \"success\")\ntagJun = Tag(\"-4.4%\", null, \"sm\", \"danger\")\ntagJul = Tag(\"+14.0%\", null, \"sm\", \"success\")\ntagAug = Tag(\"+10.4%\", null, \"sm\", \"success\")\ntagSep = Tag(\"+6.1%\", null, \"sm\", \"success\")\n\nnote = TextCallout(\"info\", \"Trend\", \"Revenue dipped slightly in June, then posted three consecutive months of growth, closing September at a six-month high.\")","promptTokens":6952,"completionTokens":1039,"latencyMs":11587,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/realistic","repeat":0,"output":"root = Stack([header, chartCard, tableCard, insightsCard, followUp], \"column\", \"l\")\n\nheader = CardHeader(\"Traffic Sources — Last Month\", \"October 1–31 · 248,500 total sessions\")\n\nchartCard = Card([chartHeader, trafficPie, shareBar, legendTags])\n\nchartHeader = CardHeader(\"Session Share by Channel\", \"Percentages sum to 100%\")\n\ntrafficPie = PieChart(channelLabels, channelSessions, \"donut\")\n\nshareBar = SingleStackedBarChart(channelLabels, channelPercents)\n\nlegendTags = TagBlock([\"Organic Search 41.2%\", \"Direct 22.4%\", \"Paid Ads 18.1%\", \"Social 11.7%\", \"Referrals 6.6%\"])\n\nchannelLabels = [\"Organic Search\", \"Direct\", \"Paid Ads\", \"Social\", \"Referrals\"]\nchannelSessions = [102380, 55664, 44980, 29082, 16394]\nchannelPercents = [41.2, 22.4, 18.1, 11.7, 6.6]\n\ntableCard = Card([tableHeader, trafficTable])\n\ntableHeader = CardHeader(\"Channel Detail\", \"Sessions, share of total, and month-over-month change\")\n\ntrafficTable = Table([colChannel, colSessions, colShare, colChange, colBounce])\n\ncolChannel = Col(\"Channel\", channelLabels)\ncolSessions = Col(\"Sessions\", channelSessions, \"number\")\ncolShare = Col(\"Share %\", channelPercents, \"number\")\ncolChange = Col(\"MoM Change\", changeTags)\ncolBounce = Col(\"Bounce Rate %\", [38.4, 41.9, 52.7, 61.3, 44.8], \"number\")\n\nchangeTags = [tagOrganic, tagDirect, tagPaid, tagSocial, tagReferral]\ntagOrganic = Tag(\"+8.3%\", null, \"sm\", \"success\")\ntagDirect = Tag(\"+1.1%\", null, \"sm\", \"neutral\")\ntagPaid = Tag(\"-4.6%\", null, \"sm\", \"danger\")\ntagSocial = Tag(\"+12.9%\", null, \"sm\", \"success\")\ntagReferral = Tag(\"-0.8%\", null, \"sm\", \"warning\")\n\ninsightsCard = Card([insightsHeader, insightsSteps, paidCallout], \"sunk\")\n\ninsightsHeader = CardHeader(\"What Stands Out\", null)\n\ninsightsSteps = Steps([step1, step2, step3])\nstep1 = StepsItem(\"Organic is carrying growth\", \"Organic search added roughly 7,800 sessions month over month and now accounts for 41.2% of all traffic — the lowest bounce rate of any channel at 38.4%.\")\nstep2 = StepsItem(\"Social is the fastest riser\", \"Social grew 12.9% off a smaller base (29,082 sessions). High bounce rate of 61.3% suggests the traffic is browsing, not converting yet.\")\nstep3 = StepsItem(\"Paid is shrinking\", \"Paid ads fell 4.6% to 44,980 sessions. Check whether budget pacing or rising CPCs cut delivery in the back half of the month.\")\n\npaidCallout = Callout(\"warning\", \"Paid efficiency needs a look\", \"Paid ads are 18.1% of sessions but the only channel losing volume while bounce rate sits above 52%. Worth auditing campaign-level spend before next month's budget lock.\")\n\nfollowUp = Buttons([btnCompare, btnConversions, btnPaid])\nbtnCompare = Button(\"Compare to previous 3 months\", Action([@ToAssistant(\"Show me website traffic by channel for the last three months so I can compare trends.\")]), \"primary\")\nbtnConversions = Button(\"Add conversion rates\", Action([@ToAssistant(\"Add conversion rate and revenue per channel to this traffic breakdown.\")]), \"secondary\")\nbtnPaid = Button(\"Drill into paid ads\", Action([@ToAssistant(\"Break down last month's paid ad traffic by campaign and platform.\")]), \"secondary\")","promptTokens":6979,"completionTokens":1391,"latencyMs":19321,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/adversarial","repeat":0,"output":"root = Stack([intro, verdict, kpiRow, marketCard, tabs, radarCard, scatterCard, deepDive, pickCard, howToChoose, caveat, followUps], \"column\", \"l\")\n\nintro = CardHeader(\"Top 5 Cloud Providers — Full Spec Teardown\", \"AWS · Microsoft Azure · Google Cloud · Oracle Cloud (OCI) · Alibaba Cloud — sorted by overall suitability for a typical production workload (my scoring, descending). Figures are US-East-equivalent list prices, early-2025 public data. Verify before signing anything.\")\n\nverdict = Callout(\"success\", \"My pick: AWS — unless one of three things is true\", \"AWS wins on breadth, region depth, compliance and hiring pool, so it is the lowest-regret default. Pick Azure instead if you are already an Enterprise Agreement / Active Directory shop (Hybrid Benefit alone can cut your bill 40%). Pick Google Cloud if the workload is data/ML-centric (BigQuery + TPUs are genuinely best-in-class). Pick OCI if egress or raw GPU/vCPU cost dominates your bill — its egress is ~10x cheaper than everyone else's and its always-free tier is embarrassingly generous.\")\n\nkpiRow = Stack([kpi1, kpi2, kpi3, kpi4], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpi1 = Card([kpi1a, kpi1b, kpi1c])\nkpi1a = TextContent(\"Cheapest egress\", \"small\")\nkpi1b = TextContent(\"OCI\", \"large-heavy\")\nkpi1c = TextContent(\"$0.0085/GB after 10 TB free — vs $0.09 AWS, $0.12 GCP premium\", \"small\")\n\nkpi2 = Card([kpi2a, kpi2b, kpi2c])\nkpi2a = TextContent(\"Most regions\", \"small\")\nkpi2b = TextContent(\"Azure — 64\", \"large-heavy\")\nkpi2c = TextContent(\"But AWS has more mature AZ-level isolation (108 AZs)\", \"small\")\n\nkpi3 = Card([kpi3a, kpi3b, kpi3c])\nkpi3a = TextContent(\"Fastest cold start\", \"small\")\nkpi3b = TextContent(\"AWS Lambda\", \"large-heavy\")\nkpi3c = TextContent(\"~150–250 ms p50 Node; ~0 ms with Provisioned Concurrency\", \"small\")\n\nkpi4 = Card([kpi4a, kpi4b, kpi4c])\nkpi4a = TextContent(\"Best free tier\", \"small\")\nkpi4b = TextContent(\"OCI Always Free\", \"large-heavy\")\nkpi4c = TextContent(\"4 Ampere OCPU + 24 GB RAM forever, no clock\", \"small\")\n\nmarketCard = Card([marketHeader, marketRow])\nmarketHeader = CardHeader(\"Market position\", \"Q4-2024 IaaS/PaaS revenue share — why the hiring pool and ecosystem follow AWS/Azure\")\nmarketRow = Stack([sharePie, shareBar], \"row\", \"l\", \"stretch\", \"start\", true)\nsharePie = Card([sharePieChart], \"sunk\")\nsharePieChart = PieChart([\"AWS\", \"Azure\", \"Google Cloud\", \"Alibaba\", \"Oracle\", \"Others\"], [31, 25, 11, 4, 3, 26], \"donut\")\nshareBar = Card([regionsChart], \"sunk\")\nregionsChart = BarChart([\"AWS\", \"Azure\", \"GCP\", \"OCI\", \"Alibaba\"], [regionsSeries, azSeries], \"grouped\", \"Provider\", \"Count\")\nregionsSeries = Series(\"Cloud regions\", [34, 64, 40, 50, 30])\nazSeries = Series(\"Availability zones\", [108, 158, 121, 50, 89])\n\ntabs = Tabs([tabPricing, tabReliability, tabCompliance, tabServerless, tabEgress, tabFree, tabAi])\n\ntabPricing = TabItem(\"pricing\", \"Pricing\", [pricingTable, pricingChart, pricingNote])\npricingTable = Table([pricingSpecCol, pricingAws, pricingAzure, pricingGcp, pricingOci, pricingAli])\npricingSpecCol = Col(\"Spec\", [\"General purpose VM (4 vCPU / 16 GB, Linux, on-demand)\", \"Same VM, 3-yr committed\", \"Discount ceiling\", \"Billing granularity\", \"Automatic discounts\", \"Block storage (SSD, per GB-mo)\", \"Object storage (standard, per GB-mo)\", \"Managed Kubernetes control plane\", \"Serverless per 1M invocations (128 MB, 100 ms)\", \"Managed Postgres (2 vCPU/8 GB, HA)\", \"Spot / preemptible discount\", \"Licence portability (Windows/SQL)\"])\npricingAws = Col(\"AWS\", [\"$0.202/hr ≈ $147/mo (m7i.xlarge)\", \"≈ $52/mo (Savings Plan)\", \"Up to 72%\", \"Per second (60s min)\", \"None — you must commit\", \"$0.08 (gp3)\", \"$0.023\", \"$0.10/hr ≈ $73/mo (EKS)\", \"≈ $0.20 + $0.0000167/GB-s\", \"≈ $260/mo (RDS Multi-AZ)\", \"Up to 90%\", \"BYOL via Dedicated Hosts only\"])\npricingAzure = Col(\"Azure\", [\"$0.192/hr ≈ $140/mo (D4as v5)\", \"≈ $46/mo (RI + AHB)\", \"Up to 72% (+ up to 85% w/ AHB)\", \"Per second\", \"None\", \"$0.088 (Premium SSD v2)\", \"$0.021 (hot LRS)\", \"Free (AKS standard tier $0.10/hr)\", \"≈ $0.20 + GB-s\", \"≈ $250/mo (Flexible Server ZR)\", \"Up to 90%\", \"Best — Azure Hybrid Benefit\"])\npricingGcp = Col(\"Google Cloud\", [\"$0.194/hr ≈ $142/mo (n2-standard-4)\", \"≈ $50/mo (3-yr CUD)\", \"Up to 70%\", \"Per second (60s min)\", \"Yes — sustained-use up to 30%\", \"$0.085 (pd-ssd) / $0.17 (hyperdisk)\", \"$0.020\", \"$0.10/hr ≈ $73/mo (GKE Std; Autopilot pods only)\", \"≈ $0.40 (Cloud Run req-based)\", \"≈ $270/mo (Cloud SQL HA)\", \"Up to 91%\", \"Sole-tenant BYOL\"])\npricingOci = Col(\"Oracle OCI\", [\"$0.098/hr ≈ $71/mo (E5.Flex, 4 OCPU)\", \"≈ $57/mo (annual flex)\", \"Up to 30% (list already lowest)\", \"Per second\", \"Yes — flat global pricing, no region tax\", \"$0.0425 + $0.017/VPU\", \"$0.0255\", \"Free (OKE basic)\", \"≈ $0.20 after 2M free\", \"≈ $210/mo (Autonomous DB / Postgres)\", \"Up to 50%\", \"Best for Oracle DB/apps licensing\"])\npricingAli = Col(\"Alibaba\", [\"$0.18/hr ≈ $131/mo (ecs.g7.xlarge)\", \"≈ $48/mo (3-yr subscription)\", \"Up to 65%\", \"Per second (intl.) / hourly (CN)\", \"None\", \"$0.10 (ESSD PL1)\", \"$0.018\", \"Free (ACK basic)\", \"≈ $0.20\", \"≈ $230/mo (ApsaraDB HA)\", \"Up to 90%\", \"Limited\"])\npricingChart = Card([pricingChartHeader, pricingBar], \"sunk\")\npricingChartHeader = CardHeader(\"Monthly cost, 4 vCPU / 16 GB Linux VM\", \"On-demand vs 3-year commitment (USD)\")\npricingBar = BarChart([\"AWS\", \"Azure\", \"GCP\", \"OCI\", \"Alibaba\"], [onDemandSeries, committedSeries], \"grouped\", \"Provider\", \"USD / month\")\nonDemandSeries = Series(\"On-demand\", [147, 140, 142, 71, 131])\ncommittedSeries = Series(\"3-yr committed\", [52, 46, 50, 57, 48])\npricingNote = TextCallout(\"info\", \"Read the fine print\", \"OCI wins on-demand by ~2x but has the *weakest* commit discount, so at 3-year scale the hyperscalers close the gap on compute. Where OCI stays 5–10x ahead is egress and support. Also note GCP's sustained-use discount applies automatically — nice for spiky, uncommitted workloads.\")\n\ntabReliability = TabItem(\"reliability\", \"Uptime & SLAs\", [slaTable, slaNote])\nslaTable = Table([slaSpecCol, slaAws, slaAzure, slaGcp, slaOci, slaAli])\nslaSpecCol = Col(\"Spec\", [\"Single VM SLA\", \"Multi-AZ / zonal-redundant VM SLA\", \"Managed K8s control plane SLA\", \"Object storage SLA\", \"Managed SQL (HA) SLA\", \"Global LB / CDN SLA\", \"SLA credit at <99%\", \"Separate performance SLA?\", \"Notable multi-hour outages (last 3 yrs)\", \"Status page candour\"])\nslaAws = Col(\"AWS\", [\"99.5%\", \"99.99% (region)\", \"99.95%\", \"99.9% (S3 Standard availability)\", \"99.95%\", \"99.99% (CloudFront)\", \"100% of monthly bill\", \"No\", \"us-east-1 Dec-2021, Jun-2023, Jul-2024 (regional)\", \"Slow to go red, detailed post-mortems\"])\nslaAzure = Col(\"Azure\", [\"99.9% (Premium SSD)\", \"99.99% (Availability Zones)\", \"99.95% (paid tier)\", \"99.9%\", \"99.99% (ZR Business Critical)\", \"99.99% (Front Door)\", \"100% at <95%\", \"No\", \"Entra ID/M365 auth incidents 2023–24, Central-US Jul-2024\", \"Improving; some incidents under-reported\"])\nslaGcp = Col(\"Google Cloud\", [\"99.5%\", \"99.99% (regional MIG)\", \"99.95%\", \"99.95% (multi-region)\", \"99.99% (Cloud SQL Enterprise+)\", \"99.99%\", \"50% at <95%\", \"No\", \"Paris eu-west-9 fire Apr-2023, Jun-2025 IAM/global control-plane\", \"Fast, detailed incident reports\"])\nslaOci = Col(\"Oracle OCI\", [\"99.95%\", \"99.99%\", \"99.95%\", \"99.9%\", \"99.995% (Autonomous DB, Exadata)\", \"99.9%\", \"Tiered service credits\", \"Yes — unique availability + manageability + performance SLAs\", \"Fewer public large-scale events (smaller blast radius)\", \"Sparse but honest\"])\nslaAli = Col(\"Alibaba\", [\"99.975%\", \"99.995% (multi-zone)\", \"99.95%\", \"99.995% (durability 99.9999999999%)\", \"99.99%\", \"99.9%\", \"Tiered credits\", \"No\", \"Nov-2023 global control-plane outage (~3 hrs)\", \"Limited English detail\"])\nslaNote = TextCallout(\"warning\", \"SLA numbers are marketing, blast radius is engineering\", \"Alibaba and OCI quote the highest single-instance numbers, but what actually keeps you up is control-plane and IAM dependency. AWS us-east-1 and Azure Entra ID are the two single points of failure that have taken down the most of the internet. Design for regional failure regardless of who you pick — the SLA credit will never cover your revenue loss.\")\n\ntabCompliance = TabItem(\"compliance\", \"Compliance & Support\", [complianceTable, supportTable, supportChart])\ncomplianceTable = Table([compSpecCol, compAws, compAzure, compGcp, compOci, compAli])\ncompSpecCol = Col(\"Spec\", [\"Total attestations/certifications\", \"SOC 1/2/3\", \"ISO 27001/27017/27018/27701\", \"PCI DSS Level 1\", \"HIPAA BAA\", \"FedRAMP\", \"DoD Impact Level\", \"EU sovereignty offering\", \"UK/AU/CA data-residency\", \"China mainland presence\", \"Confidential computing\", \"Customer-managed / HYOK keys\"])\ncompAws = Col(\"AWS\", [\"143+\", \"Yes\", \"Yes\", \"Yes\", \"Yes\", \"High (GovCloud)\", \"IL2–IL6\", \"European Sovereign Cloud (Brandenburg, 2025)\", \"Yes\", \"Via Sinnet/NWCD partners\", \"Nitro Enclaves, SEV-SNP, TDX\", \"KMS + CloudHSM, XKS external keys\"])\ncompAzure = Col(\"Azure\", [\"100+ (broadest gov catalogue)\", \"Yes\", \"Yes\", \"Yes\", \"Yes\", \"High + DoD SRG\", \"IL2–IL6\", \"EU Data Boundary (live)\", \"Yes\", \"Operated by 21Vianet\", \"Best-in-class: SEV-SNP, TDX, confidential VMs+containers\", \"Managed HSM, HYOK for M365\"])\ncompGcp = Col(\"Google Cloud\", [\"100+\", \"Yes\", \"Yes\", \"Yes\", \"Yes\", \"High\", \"IL2, IL4, IL5\", \"Sovereign Controls w/ T-Systems, S3NS\", \"Yes\", \"No (partner edge only)\",","promptTokens":7016,"completionTokens":8192,"latencyMs":105956,"finishReason":"length"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/adversarial","repeat":1,"output":"root = Stack([pageHeader, kpiRow, scoreCard, specTabs, radarCard, egressCard, deepDive, pickCard, frameworkCard, caveatCallout, ctaButtons], \"column\", \"l\")\n\npageHeader = Card([pageHeaderInner, headerTags])\npageHeaderInner = CardHeader(\"Top 5 Cloud Providers — Full Spec Comparison\", \"AWS · Microsoft Azure · Google Cloud · Oracle Cloud (OCI) · Alibaba Cloud — ranked by overall value score. Figures are US-East-equivalent list prices, Q1 2025, USD.\")\nheaderTags = TagBlock([\"Pricing\", \"Regions\", \"Uptime SLA\", \"Compliance\", \"Support\", \"Cold starts\", \"Free tier\", \"Egress\"])\n\nkpiRow = Stack([kpiAws, kpiAzure, kpiGcp, kpiOci, kpiAli], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiAws = Card([kpiAwsName, kpiAwsScore, kpiAwsNote])\nkpiAwsName = TextContent(\"AWS\", \"small-heavy\")\nkpiAwsScore = TextContent(\"92 / 100\", \"large-heavy\")\nkpiAwsNote = TextContent(\"~30% market share · widest service catalog\", \"small\")\n\nkpiAzure = Card([kpiAzureName, kpiAzureScore, kpiAzureNote])\nkpiAzureName = TextContent(\"Microsoft Azure\", \"small-heavy\")\nkpiAzureScore = TextContent(\"87 / 100\", \"large-heavy\")\nkpiAzureNote = TextContent(\"~21% share · best enterprise/AD integration\", \"small\")\n\nkpiGcp = Card([kpiGcpName, kpiGcpScore, kpiGcpNote])\nkpiGcpName = TextContent(\"Google Cloud\", \"small-heavy\")\nkpiGcpScore = TextContent(\"86 / 100\", \"large-heavy\")\nkpiGcpNote = TextContent(\"~12% share · best data/ML + networking\", \"small\")\n\nkpiOci = Card([kpiOciName, kpiOciScore, kpiOciNote])\nkpiOciName = TextContent(\"Oracle Cloud (OCI)\", \"small-heavy\")\nkpiOciScore = TextContent(\"78 / 100\", \"large-heavy\")\nkpiOciNote = TextContent(\"~3% share · cheapest compute + egress\", \"small\")\n\nkpiAli = Card([kpiAliName, kpiAliScore, kpiAliNote])\nkpiAliName = TextContent(\"Alibaba Cloud\", \"small-heavy\")\nkpiAliScore = TextContent(\"72 / 100\", \"large-heavy\")\nkpiAliNote = TextContent(\"~4% share · #1 in APAC/China\", \"small\")\n\nscoreCard = Card([scoreHeader, scoreChart, scoreNote])\nscoreHeader = CardHeader(\"Overall value score (weighted)\", \"Breadth 25% · price 20% · reliability 20% · compliance 15% · developer experience 10% · global reach 10%\")\nscoreChart = HorizontalBarChart([\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"], [scoreSeries], \"grouped\", \"Score (0-100)\", \"Provider\")\nscoreSeries = Series(\"Overall score\", [92, 87, 86, 78, 72])\nscoreNote = TextContent(\"Scores are my own weighting for a general-purpose SaaS/product team. Change the weights and the order changes — see the decision framework at the bottom.\", \"small\")\n\nspecTabs = Tabs([tabScale, tabPricing, tabServerless, tabSupport, tabCompliance, tabFree, tabEcosystem])\n\ntabScale = TabItem(\"scale\", \"Scale & Uptime\", [tableScale, scaleNote])\ntableScale = Table([colProviderA, colRegions, colAZs, colEdge, colSlaSingle, colSlaMulti, colObserved])\ncolProviderA = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncolRegions = Col(\"Regions\", [34, 62, 41, 50, 30], \"number\")\ncolAZs = Col(\"AZs / Zones\", [108, 165, 124, 61, 89], \"number\")\ncolEdge = Col(\"Edge PoPs\", [600, 192, 202, 80, 3200], \"number\")\ncolSlaSingle = Col(\"Single-VM SLA\", [\"99.5%\", \"99.9%\", \"99.5%\", \"99.9%\", \"99.975%\"])\ncolSlaMulti = Col(\"Multi-AZ compute SLA\", [\"99.99%\", \"99.99%\", \"99.99%\", \"99.95%\", \"99.995%\"])\ncolObserved = Col(\"Observed 2024 uptime\", [\"99.99%\", \"99.96%\", \"99.98%\", \"99.98%\", \"99.97%\"])\nscaleNote = TextContent(\"Azure counts the most regions but many are single-zone paired regions; AWS regions are more uniformly 3+ AZ. Alibaba's PoP count includes its CDN network. Azure had the largest 2024 incident footprint (multiple Entra ID / Front Door global events).\", \"small\")\n\ntabPricing = TabItem(\"pricing\", \"Pricing\", [tablePricing, tabPricingCommit, pricingNote])\ntablePricing = Table([colProviderB, colVm, colVmHr, colBlock, colObject, colSpot, colGpu])\ncolProviderB = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncolVm = Col(\"2 vCPU / 8 GB on-demand ($/mo)\", [73.0, 70.1, 71.4, 43.8, 66.2], \"number\")\ncolVmHr = Col(\"Same, per hour ($)\", [0.1008, 0.096, 0.0971, 0.06, 0.0907], \"number\")\ncolBlock = Col(\"Block storage SSD ($/GB-mo)\", [0.08, 0.075, 0.17, 0.0425, 0.09], \"number\")\ncolObject = Col(\"Object storage hot ($/GB-mo)\", [0.023, 0.018, 0.02, 0.0255, 0.02], \"number\")\ncolSpot = Col(\"Spot/preemptible discount\", [\"up to 90%\", \"up to 90%\", \"up to 91%\", \"n/a (preemptible ~50%)\", \"up to 90%\"])\ncolGpu = Col(\"H100 8-GPU node ($/hr, on-demand)\", [98.32, 98.0, 88.49, 80.0, 92.5], \"number\")\ntabPricingCommit = Table([colProviderB2, colCommit1, colCommit3, colDiscountModel, colBilling])\ncolProviderB2 = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncolCommit1 = Col(\"1-yr commit discount\", [\"~28%\", \"~26%\", \"~28% (CUD) + 30% SUD\", \"~33%\", \"~25%\"])\ncolCommit3 = Col(\"3-yr commit discount\", [\"~46%\", \"~53% (RI) / 65% w/ AHB\", \"~46-55%\", \"~48%\", \"~45%\"])\ncolDiscountModel = Col(\"Discount model\", [\"Savings Plans / RI\", \"Reserved Instances + Hybrid Benefit\", \"CUDs + automatic Sustained-Use\", \"Universal Credits\", \"Subscription + Savings Plan\"])\ncolBilling = Col(\"Billing granularity\", [\"per second (60s min)\", \"per second\", \"per second (60s min)\", \"per second\", \"per second\"])\npricingNote = TextContent(\"OCI is consistently 30-50% cheaper on raw compute and block storage, and it prices identically in every region — AWS/Azure/GCP charge 10-40% more in São Paulo, Zurich, Sydney, etc. GCP block storage looks expensive because the quoted SKU is Hyperdisk Balanced with provisioned IOPS included.\", \"small\")\n\ntabServerless = TabItem(\"serverless\", \"Serverless & Cold Starts\", [tableServerless, coldChart, coldNote])\ntableServerless = Table([colProviderC, colFaas, colColdNode, colColdJava, colMaxTimeout, colGbs, colConcurrency])\ncolProviderC = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncolFaas = Col(\"Function product\", [\"Lambda\", \"Functions (Flex Consumption)\", \"Cloud Run / Functions 2nd gen\", \"OCI Functions (Fn)\", \"Function Compute 3.0\"])\ncolColdNode = Col(\"Cold start, Node (p50 ms)\", [180, 320, 240, 1400, 210], \"number\")\ncolColdJava = Col(\"Cold start, Java (p50 ms)\", [220, 1100, 900, 3200, 650], \"number\")\ncolMaxTimeout = Col(\"Max execution\", [\"15 min (Lambda) / 60 min (Fargate task)\", \"unbounded (Flex), 10 min (Consumption)\", \"60 min (Cloud Run request)\", \"5 min\", \"24 hr (async)\"])\ncolGbs = Col(\"Price per GB-second ($)\", [1.667e-5, 1.6e-5, 2.4e-5, 1.417e-5, 1.6e-5], \"number\")\ncolConcurrency = Col(\"Per-instance concurrency\", [\"1 (Lambda)\", \"up to 100\", \"up to 1000 (Cloud Run)\", \"1\", \"up to 200\"])\ncoldChart = BarChart([\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"], [coldSeriesNode, coldSeriesJava], \"grouped\", \"Provider\", \"Cold start p50 (ms)\")\ncoldSeriesNode = Series(\"Node 20 (ms)\", [180, 320, 240, 1400, 210])\ncoldSeriesJava = Series(\"Java 21 (ms)\", [220, 1100, 900, 3200, 650])\ncoldNote = TextContent(\"AWS Java number assumes Lambda SnapStart is enabled — without it, expect 1.5-3s. Cloud Run wins on sustained throughput (1000 concurrent requests per instance means far fewer cold starts in practice). OCI Functions is the clear laggard.\", \"small\")\n\ntabSupport = TabItem(\"support\", \"Support Tiers\", [tableSupport, supportNote])\ntableSupport = Table([colProviderD, colTierFree, colTierDev, colTierBiz, colTierEnt, colP1, colTam])\ncolProviderD = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncolTierFree = Col(\"Free tier\", [\"Basic (docs + billing only)\", \"Basic (billing only)\", \"Basic (docs + billing)\", \"Included 24/7 sev-1 support\", \"Basic (billing)\"])\ncolTierDev = Col(\"Entry paid\", [\"Developer — $29/mo or 3%\", \"Developer — $29/mo\", \"Standard — $29/mo + 3%\", \"n/a (all included)\", \"Developer — $63/mo\"])\ncolTierBiz = Col(\"Mid tier\", [\"Business — max($100, 10/7/5/3% tiered)\", \"Standard — $100/mo\", \"Enhanced — $500/mo + 3%\", \"n/a\", \"Business — $400/mo\"])\ncolTierEnt = Col(\"Top tier\", [\"Enterprise — max($15k, 10-3%)\", \"Premier — custom (~$50k+/yr)\", \"Premium — $12.5k/mo + 4%\", \"Included / My Oracle Support\", \"Enterprise — custom\"])\ncolP1 = Col(\"Sev-1 response (top tier)\", [\"< 15 min\", \"< 15 min\", \"< 15 min\", \"< 15 min\", \"< 15 min\"])\ncolTam = Col(\"Named TAM included?\", [\"Enterprise only\", \"Premier only\", \"Premium only\", \"Yes, on spend commitment\", \"Enterprise only\"])\nsupportNote = TextContent(\"Big one: OCI includes 24/7 sev-1 support with every account at no charge. AWS Enterprise support is the most expensive line item most large accounts don't budget for — 10% of the first $150k/mo of spend, floor $15,000/mo. GCP Premium adds a 4% consumption charge on top of the flat fee.\", \"small\")\n\ntabCompliance = TabItem(\"compliance\", \"Compliance & Security\", [tableCompliance, complianceTags, complianceNote])\ntableCompliance = Table([colProviderE, colCerts, colFedramp, colIl, colHipaa, colPci, colSovereign, colHsm])\ncolProviderE = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncolCerts = Col(\"Audited programs/certs\", [143, 100, 125, 80, 140], \"number\")\ncolFedramp = Col(\"FedRAMP High\", [tagFedAws, tagFedAzure, tagFedGcp, tagFedOci, tagFedAli])\ntagFedAws = Tag(\"Yes (+ GovCloud, IL6)\", null, \"sm\", \"success\")\ntagFedAzure = Tag(\"Yes (+ Gov, IL6)\", null, \"sm\", \"success\")\ntagFedGcp = Tag(\"Yes (+ Assured Wkld, IL5)\", null, \"sm\", \"success\")\ntagFedOci = Tag(\"Yes (+ Gov, IL5/IL6)\", null, \"sm\", \"success\")\ntagFedAli = Tag(\"No\", null, \"sm\", \"danger\")\ncolIl = Col(\"DoD Impact Level\", [\"IL2-IL6\", \"IL2-IL6\", \"IL2-IL5\", \"IL2-IL6\", \"None\"])\ncolHipaa = Col(\"HIPAA BAA\", [\"Yes, 150+ services\", \"Yes, broad\", \"Yes, broad\", \"Yes\", \"Limited\"])\ncolPci = Col(\"PCI DSS 4.0\", [\"Level 1\", \"Level 1\", \"Level 1\", \"Level 1\", \"Level 1\"])\ncolSovereign = Col(\"EU sovereign option\", [\"European Sovereign Cloud (Brandenburg, 2025)\", \"EU Data Boundary (GA)\", \"Sovereign Controls / T-Systems, S3NS\", \"EU Sovereign Cloud (GA, 2 regions)\", \"Frankfurt only\"])\ncolHsm = Col(\"Confidential computing\", [\"Nitro Enclaves, SEV-SNP, TDX\", \"SEV-SNP, TDX, Always Encrypted\", \"Confidential VMs/GKE nodes\", \"Shielded instances\", \"Confidential VMs\"])\ncomplianceTags = TagBlock([\"SOC 1/2/3 — all five\", \"ISO 27001/27017/27018 — all five\", \"GDPR DPA — all five\", \"C5 (DE) — all five\", \"IRAP (AU) — AWS/Azure/GCP/OCI\", \"K-ISMS\", \"TISAX\"])\ncomplianceNote = TextContent(\"All five clear the standard SOC 2 / ISO 27001 / GDPR bar, so compliance rarely decides between AWS, Azure, GCP and OCI. It decides *against* Alibaba Cloud for any US-federal or EU-data-residency-sensitive workload.\", \"small\")\n\ntabFree = TabItem(\"free\", \"Free Tier & Egress\", [tableFree, tableEgress, freeNote])\ntableFree = Table([colProviderF, colTrial, colAlwaysFree, colFreeCompute, colFreeDb])\ncolProviderF = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncolTrial = Col(\"Trial credit\", [\"$200 credits (6 mo, 2025 model)\", \"$200 (30 days)\", \"$300 (90 days)\", \"$300 (30 days)\", \"$450 voucher pack\"])\ncolAlwaysFree = Col(\"Always-free tier\", [\"Yes, ~30 services\", \"Yes, ~25 services (12-mo + always)\", \"Yes, ~20 products\", \"Yes — the most generous\", \"Limited, mostly 12-month\"])\ncolFreeCompute = Col(\"Free always-on compute\", [\"1M Lambda req + 400k GB-s\", \"1M Functions exec\", \"2M Cloud Run req + 180k vCPU-s\", \"4 Arm cores / 24 GB RAM VMs, forever\", \"1M FC invocations\"])\ncolFreeDb = Col(\"Free managed DB\", [\"None (750h RDS, 12 mo only)\", \"250 GB SQL DB (12 mo)\", \"Firestore 1 GB + 50k reads/day\", \"2 × Autonomous DB, 20 GB, forever\", \"None\"])\ntableEgress = Table([colProviderG, colFreeEgress, colEgressFirst, colEgressHigh, colInterAz, colCdnEgress])\ncolProviderG = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncolFreeEgress = Col(\"Free egress / month\", [\"100 GB\", \"100 GB\", \"200 GB\", \"10,000 GB\", \"20 GB\"])\ncolEgressFirst = Col(\"Egress $/GB (first 10 TB)\", [0.09, 0.087, 0.12, 0.0085, 0.12], \"number\")\ncolEgressHigh = Col(\"Egress $/GB (at 500 TB)\", [0.05, 0.05, 0.08, 0.0085, 0.074], \"number\")\ncolInterAz = Col(\"Cross-AZ traffic\", [\"$0.01/GB each way\", \"Free (same region)\", \"$0.01/GB\", \"Free\", \"Free\"])\ncolCdnEgress = Col(\"CDN egress $/GB (NA)\", [0.085, 0.081, 0.08, 0.0085, 0.081], \"number\")\nfreeNote = TextContent(\"Egress is where the bill surprises you. Moving 50 TB/month out: AWS ","promptTokens":7016,"completionTokens":8192,"latencyMs":97381,"finishReason":"length"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/realistic","repeat":1,"output":"root = Stack([header, kpiRow, chartCard, breakdownCard, insightCallout, followUps], \"column\", \"l\")\n\nheader = CardHeader(\"Traffic Sources — Last Month\", \"November 1–30 · 128,450 total sessions across all channels\")\n\nkpiRow = Stack([kpiTotal, kpiTop, kpiPaidShare, kpiChannels], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiTotal = Card([kpiTotalLabel, kpiTotalValue, kpiTotalSub])\nkpiTotalLabel = TextContent(\"Total Sessions\", \"small\")\nkpiTotalValue = TextContent(\"128,450\", \"large-heavy\")\nkpiTotalSub = TextContent(\"+8.2% vs October\", \"small\")\n\nkpiTop = Card([kpiTopLabel, kpiTopValue, kpiTopSub])\nkpiTopLabel = TextContent(\"Largest Channel\", \"small\")\nkpiTopValue = TextContent(\"42.3%\", \"large-heavy\")\nkpiTopSub = TextContent(\"Organic search · 54,334\", \"small\")\n\nkpiPaidShare = Card([kpiPaidLabel, kpiPaidValue, kpiPaidSub])\nkpiPaidLabel = TextContent(\"Paid Share\", \"small\")\nkpiPaidValue = TextContent(\"18.4%\", \"large-heavy\")\nkpiPaidSub = TextContent(\"23,635 sessions from ads\", \"small\")\n\nkpiChannels = Card([kpiChanLabel, kpiChanValue, kpiChanSub])\nkpiChanLabel = TextContent(\"Unpaid Traffic\", \"small\")\nkpiChanValue = TextContent(\"81.6%\", \"large-heavy\")\nkpiChanSub = TextContent(\"Organic + direct + social + referral\", \"small\")\n\nchartCard = Card([chartHeader, chartTabs])\nchartHeader = CardHeader(\"Share of Sessions\", \"Percentages sum to 100.0%\")\nchartTabs = Tabs([tabDonut, tabPie, tabBar, tabRanked])\n\ntabDonut = TabItem(\"donut\", \"Donut\", [donutChart, shareBar])\ntabPie = TabItem(\"pie\", \"Pie\", [pieChart])\ntabBar = TabItem(\"radial\", \"Radial\", [radialChart])\ntabRanked = TabItem(\"ranked\", \"Ranked\", [rankedBars])\n\nchannelLabels = [\"Organic Search\", \"Direct\", \"Paid Ads\", \"Social\", \"Referrals\"]\nchannelSessions = [54334, 27745, 23635, 14386, 8350]\nchannelPercents = [42.3, 21.6, 18.4, 11.2, 6.5]\n\ndonutChart = PieChart(channelLabels, channelPercents, \"donut\")\npieChart = PieChart(channelLabels, channelSessions, \"pie\")\nradialChart = RadialChart(channelLabels, channelPercents)\nshareBar = SingleStackedBarChart(channelLabels, channelPercents)\nrankedBars = HorizontalBarChart(channelLabels, [sessionSeries], \"grouped\", \"Sessions\", \"Channel\")\nsessionSeries = Series(\"Sessions\", channelSessions)\n\nbreakdownCard = Card([breakdownHeader, breakdownTable, breakdownNote])\nbreakdownHeader = CardHeader(\"Channel Detail\", \"Sessions, share of total, and month-over-month change\")\nbreakdownTable = Table([colChannel, colSessions, colShare, colChange, colTrend])\n\ncolChannel = Col(\"Channel\", channelLabels)\ncolSessions = Col(\"Sessions\", channelSessions, \"number\")\ncolShare = Col(\"Share %\", channelPercents, \"number\")\ncolChange = Col(\"MoM Change\", [\"+12.4%\", \"+3.1%\", \"+19.8%\", \"-4.6%\", \"+1.9%\"])\ncolTrend = Col(\"Trend\", trendTags)\ntrendTags = [tagUp1, tagUp2, tagUp3, tagDown, tagFlat]\ntagUp1 = Tag(\"Growing\", null, \"sm\", \"success\")\ntagUp2 = Tag(\"Steady\", null, \"sm\", \"info\")\ntagUp3 = Tag(\"Spiking\", null, \"sm\", \"success\")\ntagDown = Tag(\"Declining\", null, \"sm\", \"danger\")\ntagFlat = Tag(\"Flat\", null, \"sm\", \"neutral\")\n\nbreakdownNote = TextContent(\"42.3 + 21.6 + 18.4 + 11.2 + 6.5 = **100.0%** — no unattributed bucket, all sessions are accounted for.\", \"small\")\n\ninsightCallout = Callout(\"info\", \"Paid ads drove the fastest growth\", \"Paid sessions rose 19.8% MoM after the November promo push, lifting their share from 16.6% to 18.4%. Social slipped 4.6%, the only channel to decline.\")\n\nfollowUps = Buttons([btnConversion, btnPaid, btnSocial, btnCompare])\nbtnConversion = Button(\"Show conversion rate by channel\", null, \"secondary\")\nbtnPaid = Button(\"Break down paid ads by campaign\", null, \"secondary\")\nbtnSocial = Button(\"Why did social traffic drop?\", null, \"secondary\")\nbtnCompare = Button(\"Compare to the previous 6 months\", null, \"tertiary\")","promptTokens":6979,"completionTokens":2044,"latencyMs":21114,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/adversarial","repeat":4,"output":"root = Stack([intro, verdictCallout, sortNote, kpiRow, specTabs, moneyCard, coldStartCard, complianceAcc, supportCard, scoreCard, pickSteps, caveats, followUps], \"column\", \"l\")\n\nintro = Card([introHeader, introText])\nintroHeader = CardHeader(\"Top 5 Cloud Providers — Full Spec Comparison\", \"AWS vs Azure vs Google Cloud vs Oracle Cloud vs Alibaba Cloud · figures reflect public list pricing, us-east-equivalent regions, 2025\")\nintroText = TextContent(\"Everything below is **on-demand list price** in USD unless noted. Real bills differ once you apply committed-use discounts, enterprise agreements, and reserved capacity — typically **30–65% off** these numbers. Providers ranked by *market share* in the main table, and by *my own weighted value score* in the scoring section.\")\n\nverdictCallout = Callout(\"success\", \"My pick: AWS for most teams — Oracle Cloud if the bill is the deciding factor\", \"AWS wins on breadth, hiring pool, ecosystem, and third-party integration depth. Google Cloud is the better pick for data/ML-heavy work, and OCI is 40-60% cheaper with near-free egress if you can live with a thinner service catalog.\")\n\nsortNote = TextContent(\"Sorted by **market share descending** — it is the most useful default sort because ecosystem size drives hiring, tooling, Stack Overflow answers, Terraform provider quality, and third-party integrations more than any single spec.\", \"small\")\n\nkpiRow = Stack([kpiAws, kpiAzure, kpiGcp, kpiOci, kpiAli], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiAws = Card([kpiAwsName, kpiAwsShare, kpiAwsLabel, kpiAwsTag])\nkpiAwsName = TextContent(\"Amazon Web Services\", \"small-heavy\")\nkpiAwsShare = TextContent(\"30%\", \"large-heavy\")\nkpiAwsLabel = TextContent(\"market share · 240+ services · 36 regions\", \"small\")\nkpiAwsTag = Tag(\"Broadest catalog\", null, \"sm\", \"success\")\n\nkpiAzure = Card([kpiAzureName, kpiAzureShare, kpiAzureLabel, kpiAzureTag])\nkpiAzureName = TextContent(\"Microsoft Azure\", \"small-heavy\")\nkpiAzureShare = TextContent(\"21%\", \"large-heavy\")\nkpiAzureLabel = TextContent(\"market share · 200+ services · 64 regions\", \"small\")\nkpiAzureTag = Tag(\"Best enterprise/hybrid\", null, \"sm\", \"info\")\n\nkpiGcp = Card([kpiGcpName, kpiGcpShare, kpiGcpLabel, kpiGcpTag])\nkpiGcpName = TextContent(\"Google Cloud\", \"small-heavy\")\nkpiGcpShare = TextContent(\"12%\", \"large-heavy\")\nkpiGcpLabel = TextContent(\"market share · 150+ services · 42 regions\", \"small\")\nkpiGcpTag = Tag(\"Best data + ML\", null, \"sm\", \"info\")\n\nkpiOci = Card([kpiOciName, kpiOciShare, kpiOciLabel, kpiOciTag])\nkpiOciName = TextContent(\"Oracle Cloud (OCI)\", \"small-heavy\")\nkpiOciShare = TextContent(\"3%\", \"large-heavy\")\nkpiOciLabel = TextContent(\"market share · 100+ services · 50 regions\", \"small\")\nkpiOciTag = Tag(\"Cheapest egress\", null, \"sm\", \"success\")\n\nkpiAli = Card([kpiAliName, kpiAliShare, kpiAliLabel, kpiAliTag])\nkpiAliName = TextContent(\"Alibaba Cloud\", \"small-heavy\")\nkpiAliShare = TextContent(\"4%\", \"large-heavy\")\nkpiAliLabel = TextContent(\"market share · 100+ services · 31 regions\", \"small\")\nkpiAliTag = Tag(\"APAC / China reach\", null, \"sm\", \"warning\")\n\nspecTabs = Tabs([tabOverview, tabPricing, tabScale, tabReliability, tabServerless, tabFreeTier, tabCharts])\n\ntabOverview = TabItem(\"overview\", \"Overview\", [overviewTable, overviewNote])\noverviewTable = Table([colProvider, colParent, colLaunch, colShare, colRevenue, colServices, colStrength, colWeakness])\ncolProvider = Col(\"Provider\", providers)\ncolParent = Col(\"Parent\", [\"Amazon\", \"Microsoft\", \"Alphabet\", \"Oracle\", \"Alibaba Group\"])\ncolLaunch = Col(\"Launched\", [2006, 2010, 2008, 2016, 2009], \"number\")\ncolShare = Col(\"Market share %\", [30, 21, 12, 4, 3], \"number\")\ncolRevenue = Col(\"Annualised run-rate\", [\"~$115B\", \"~$80B (Intelligent Cloud)\", \"~$47B\", \"~$26B (total cloud)\", \"~$16B\"])\ncolServices = Col(\"Services\", [\"240+\", \"200+\", \"150+\", \"100+\", \"100+\"])\ncolStrength = Col(\"Biggest strength\", [\"Catalog depth, ecosystem, maturity\", \"Enterprise licensing, AD/hybrid, Copilot\", \"Data, BigQuery, Kubernetes, networking\", \"Raw price-performance, ARM, egress\", \"China + SEA presence, price\"])\ncolWeakness = Col(\"Biggest weakness\", [\"Cost complexity, console sprawl\", \"Reliability/incident record, docs churn\", \"Product deprecation reputation, smaller ecosystem\", \"Thin service catalog, smaller community\", \"Compliance/geopolitics outside APAC\"])\noverviewNote = TextContent(\"Cloudflare, IBM Cloud, DigitalOcean and Tencent Cloud are the notable honourable mentions — Cloudflare in particular beats all five on cold start (~0 ms) and egress (zero) for edge workloads, but it is not a general-purpose IaaS.\", \"small\")\n\ntabPricing = TabItem(\"pricing\", \"Pricing\", [pricingTable, pricingNote, discountTable])\npricingTable = Table([colProvider2, colVmName, colVmHr, colVmMo, colArm, colStorage, colBlock, colK8s, colLb])\ncolProvider2 = Col(\"Provider\", providers)\ncolVmName = Col(\"Reference 4 vCPU / 16 GB VM\", [\"m7i.xlarge\", \"D4as_v5\", \"n2-standard-4\", \"VM.Standard.E5.Flex\", \"ecs.g7.xlarge\"])\ncolVmHr = Col(\"On-demand $/hr\", [0.2016, 0.192, 0.1942, 0.098, 0.19], \"number\")\ncolVmMo = Col(\"On-demand $/mo (730h)\", [147.17, 140.16, 141.77, 71.54, 138.7], \"number\")\ncolArm = Col(\"ARM $/vCPU/hr\", [0.0408, 0.0392, 0.0385, 0.01, 0.0374], \"number\")\ncolStorage = Col(\"Object storage $/GB/mo\", [0.023, 0.018, 0.02, 0.0255, 0.0165], \"number\")\ncolBlock = Col(\"Block SSD $/GB/mo\", [0.08, 0.075, 0.085, 0.0425, 0.07], \"number\")\ncolK8s = Col(\"Managed K8s control plane\", [\"$0.10/hr (EKS)\", \"Free (Std tier $0.10/hr)\", \"$0.10/hr (GKE, 1 free zonal)\", \"Free (OKE Basic)\", \"Free (ACK Basic)\"])\ncolLb = Col(\"Load balancer $/mo base\", [16.43, 18.25, 18.26, 7.3, 14.6], \"number\")\npricingNote = TextContent(\"**Important caveat on OCI:** 1 OCPU = 2 vCPU, so its per-vCPU price is roughly half of the hyperscalers. That gap is the single largest structural pricing difference in this table. Ampere A1 ARM at ~$0.01/core/hr is the cheapest general-purpose compute of any major provider.\", \"small\")\n\ndiscountTable = Table([colProvider3, colCommit, colSpot, colSustained, colLicense])\ncolProvider3 = Col(\"Provider\", providers)\ncolCommit = Col(\"Max committed discount\", [\"Up to 72% (3-yr Savings Plan / RI)\", \"Up to 72% (3-yr RI) + Hybrid Benefit\", \"Up to 57% (3-yr CUD)\", \"~20-33% (annual universal credits)\", \"Up to 60% (3-yr subscription)\"])\ncolSpot = Col(\"Spot / preemptible savings\", [\"Up to 90%\", \"Up to 90%\", \"Up to 91% (Spot VMs)\", \"Up to 50% (preemptible)\", \"Up to 90%\"])\ncolSustained = Col(\"Automatic discounts\", [\"No (must commit)\", \"No\", \"Yes — sustained use discount up to 30%\", \"No\", \"No\"])\ncolLicense = Col(\"Windows / DB licensing edge\", [\"BYOL, License Manager\", \"Best — AHUB, free Extended Security Updates\", \"BYOL, sole-tenant nodes\", \"Best for Oracle DB — huge BYOL advantage\", \"BYOL\"])\n\ntabScale = TabItem(\"scale\", \"Regions & Network\", [scaleTable, scaleNote])\nscaleTable = Table([colProvider4, colRegions, colAZs, colEdge, colBackbone, colPrivate, colChina])\ncolProvider4 = Col(\"Provider\", providers)\ncolRegions = Col(\"Regions\", [36, 64, 42, 50, 31], \"number\")\ncolAZs = Col(\"AZs / zones\", [\"114 AZs\", \"160+ zones\", \"127 zones\", \"~50 (single-AD in many)\", \"89 zones\"])\ncolEdge = Col(\"Edge / PoP locations\", [\"600+ CloudFront PoPs\", \"192 CDN PoPs\", \"200+ PoPs\", \"80+ PoPs\", \"3,200 CDN nodes\"])\ncolBackbone = Col(\"Private backbone\", [\"Yes, extensive\", \"Yes, 250k+ km fibre\", \"Yes — best-in-class (Jupiter/B4)\", \"Yes, FastConnect\", \"Yes, CEN\"])\ncolPrivate = Col(\"Dedicated interconnect\", [\"Direct Connect\", \"ExpressRoute\", \"Cloud Interconnect\", \"FastConnect (cheapest)\", \"Express Connect\"])\ncolChina = Col(\"Mainland China\", [\"Yes (Sinnet/NWCD partner)\", \"Yes (21Vianet)\", \"No\", \"Limited\", \"Native — market leader\"])\n\ntabReliability = TabItem(\"reliability\", \"Uptime & SLA\", [slaTable, slaNote])\nslaTable = Table([colProvider5, colVmSingle, colVmMulti, colStorageSla, colDbSla, colCredit, colIncidents])\ncolProvider5 = Col(\"Provider\", providers)\ncolVmSingle = Col(\"Single-instance compute SLA\", [\"99.5% (99.9% w/ EBS-only)\", \"99.9% (premium SSD)\", \"99.5%\", \"99.9%\", \"99.975%\"])\ncolVmMulti = Col(\"Multi-AZ compute SLA\", [\"99.99%\", \"99.99%\", \"99.99%\", \"99.95% (fewer multi-AD regions)\", \"99.995%\"])\ncolStorageSla = Col(\"Object storage SLA\", [\"99.9% (11 nines durability)\", \"99.9% RA-GRS read 99.99%\", \"99.95% multi-region\", \"99.9%\", \"99.95%\"])\ncolDbSla = Col(\"Managed DB SLA\", [\"99.99% (Aurora multi-AZ)\", \"99.99% (SQL DB zone-redundant)\", \"99.99% (Cloud SQL HA)\", \"99.995% (Autonomous DB)\", \"99.99%\"])\ncolCredit = Col(\"Service credit at <99%\", [\"30%\", \"25-100%\", \"30%\", \"25%\", \"30%\"])\ncolIncidents = Col(\"Notable recent outage pattern\", [\"Rare but huge blast radius (us-east-1)\", \"Most frequent major incidents of the five\", \"Infrequent, occasional global control-plane\", \"Infrequent, smaller user base\", \"Infrequent outside APAC\"])\nslaNote = TextContent(\"Read SLAs carefully: **99.9% = 43.8 min/month of allowed downtime; 99.99% = 4.4 min/month.** Nobody pays out more than service credits — none of these five will compensate you for lost revenue. Architect for failure, do not architect for the SLA document.\", \"small\")\n\ntabServerless = TabItem(\"serverless\", \"Serverless & Cold Start\", [coldTable, coldNote])\ncoldTable = Table([colProvider6, colFn, colColdNode, colColdJvm, colMaxDur, colMaxMem, colFnPrice, colAlways])\ncolProvider6 = Col(\"Provider\", providers)\ncolFn = Col(\"Function product\", [\"Lambda\", \"Azure Functions\", \"Cloud Run / Cloud Functions\", \"OCI Functions\", \"Function Compute 3.0\"])\ncolColdNode = Col(\"Cold start, Node/Python (ms)\", [180, 320, 220, 550, 260], \"number\")\ncolColdJvm = Col(\"Cold start, JVM/.NET (ms)\", [\"~1,800 (200 w/ SnapStart)\", \"~1,100 (.NET isolated)\", \"~1,400 (~300 w/ startup CPU boost)\", \"~2,500\", \"~1,600\"])\ncolMaxDur = Col(\"Max execution\", [\"15 min (Lambda) / 60 min (Fargate task)\", \"Unlimited on Premium/Dedicated\", \"60 min (Cloud Run request) / 24h jobs\", \"5 min\", \"24 h\"])\ncolMaxMem = Col(\"Max memory\", [\"10 GB\", \"14 GB (Premium)\", \"32 GB (Cloud Run)\", \"2 GB\", \"32 GB\"])\ncolFnPrice = Col(\"$ per 1M invocations\", [0.2, 0.2, 0.4, 0.2, 0.2], \"number\")\ncolAlways = Col(\"Cold-start mitigation\", [\"Provisioned concurrency, SnapStart\", \"Premium plan pre-warmed instances\", \"Min instances, CPU boost\", \"Provisioned concurrency (limited)\", \"Reserved instances, snapshot start\"])\ncoldNote = TextContent(\"Numbers are **median p50 for a small dependency-light handler in a warm region**. p99 can be 3-5x higher on all five. If sub-50 ms cold start is a hard requirement, none of these five are the right answer — use Cloudflare Workers or Deno Deploy (V8 isolates, ~0-5 ms).\", \"small\")\n\ntabFreeTier = TabItem(\"free\", \"Free Tier & Egress\", [freeTable, egressTable])\nfreeTable = Table([colProvider7, colCredit, colCreditWindow, colTwelve, colAlwaysFree])\ncolProvider7 = Col(\"Provider\", providers)\ncolCredit = Col(\"Signup credit\", [\"None (free tier only)\", \"$200\", \"$300\", \"$300\", \"$300 (varies by region)\"])\ncolCreditWindow = Col(\"Credit validity\", [\"—\", \"30 days\", \"90 days\", \"30 days\", \"60 days\"])\ncolTwelve = Col(\"12-month trial tier\", [\"750 h t3.micro, 100 GB CloudFront\", \"750 h B1s, 64 GB SQL DB, 5 GB blob\", \"Included in credit period\", \"Included in credit period\", \"Limited\"])\ncolAlwaysFree = Col(\"Always-free allowance\", [\"1M Lambda calls, 25 GB DynamoDB, 1M SQS\", \"1M Functions, 5 GB blob, 250 GB SQL serverless\", \"e2-micro VM, 5 GB GCS, 2M Cloud Run calls, 1 GB BigQuery/mo\", \"**Best** — 4 OCPU + 24 GB Ampere ARM, 200 GB block, 2 Autonomous DBs, 10 TB egress/mo\", \"Modest, mostly trial-only\"])\n\negressTable = Table([colProvider8, colEgressFree, colEgressRate, colEgressTb, colInterAz, colFreeExit])\ncolProvider8 = Col(\"Provider\", providers)\ncolEgressFree = Col(\"Free egress/month\", [\"100 GB\", \"100 GB\", \"200","promptTokens":7016,"completionTokens":8192,"latencyMs":101060,"finishReason":"length"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/adversarial","repeat":2,"output":"","promptTokens":7016,"completionTokens":8192,"latencyMs":108399,"finishReason":"length"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/realistic","repeat":3,"output":"root = Stack([header, chartCard, tableCard, notesCard], \"column\", \"l\")\n\nheader = CardHeader(\"Traffic Sources — Last Month\", \"Sessions by channel, March 1–31 · 100,000 total sessions\")\n\nchartCard = Card([chartHeader, pie, shareBar, legend])\nchartHeader = CardHeader(\"Channel Mix\", \"Percentages sum to exactly 100%\")\npie = PieChart(channelLabels, channelSessions, \"donut\")\nshareBar = SingleStackedBarChart(channelLabels, channelSessions)\nlegend = TagBlock([\"Organic 42.5%\", \"Paid 21.0%\", \"Direct 18.5%\", \"Social 11.0%\", \"Referral 7.0%\"])\n\nchannelLabels = [\"Organic Search\", \"Paid Ads\", \"Direct\", \"Social\", \"Referrals\"]\nchannelSessions = [42500, 21000, 18500, 11000, 7000]\n\ntableCard = Card([tableHeader, breakdownTable, totalLine])\ntableHeader = CardHeader(\"Breakdown\", \"Sessions, share of total, and month-over-month change\")\nbreakdownTable = Table([colChannel, colSessions, colShare, colMoM])\ncolChannel = Col(\"Channel\", channelLabels)\ncolSessions = Col(\"Sessions\", channelSessions, \"number\")\ncolShare = Col(\"Share\", [\"42.5%\", \"21.0%\", \"18.5%\", \"11.0%\", \"7.0%\"])\ncolMoM = Col(\"MoM\", momTags)\nmomTags = [tagOrganic, tagPaid, tagDirect, tagSocial, tagReferral]\ntagOrganic = Tag(\"+6.2%\", null, \"sm\", \"success\")\ntagPaid = Tag(\"-3.4%\", null, \"sm\", \"danger\")\ntagDirect = Tag(\"+1.1%\", null, \"sm\", \"success\")\ntagSocial = Tag(\"+9.8%\", null, \"sm\", \"success\")\ntagReferral = Tag(\"0.0%\", null, \"sm\", \"neutral\")\ntotalLine = TextContent(\"**Total: 100,000 sessions (100.0%)**\", \"small-heavy\")\n\nnotesCard = Card([notesHeader, note1, note2, followUps], \"sunk\")\nnotesHeader = CardHeader(\"Takeaways\")\nnote1 = TextCallout(\"success\", \"Organic is doing the heavy lifting\", \"Organic search alone accounts for 42.5% of sessions and grew 6.2% month over month — the largest absolute gain of any channel.\")\nnote2 = TextCallout(\"warning\", \"Paid efficiency slipped\", \"Paid ads fell to 21.0% of traffic (-3.4%) despite flat spend, which pushed effective cost per session up.\")\nfollowUps = Buttons([btnLanding, btnPaid, btnTrend])\nbtnLanding = Button(\"Which landing pages drove organic growth?\", null, \"secondary\")\nbtnPaid = Button(\"Diagnose the paid ads decline\", null, \"secondary\")\nbtnTrend = Button(\"Show a 6-month trend by channel\", null, \"tertiary\")","promptTokens":6979,"completionTokens":1227,"latencyMs":15841,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/realistic","repeat":2,"output":"root = Stack([header, kpiRow, mainCard, tabsSection, insightsCard, footerButtons], \"column\", \"l\")\n\nheader = CardHeader(\"Traffic Sources — Last Month\", \"128,450 total sessions · Oct 1–31 · all channels attributed (last non-direct click)\")\n\nkpiRow = Stack([kpi1, kpi2, kpi3, kpi4], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpi1 = Card([kpi1Label, kpi1Value, kpi1Tag])\nkpi1Label = TextContent(\"Total Sessions\", \"small\")\nkpi1Value = TextContent(\"128,450\", \"large-heavy\")\nkpi1Tag = Tag(\"+8.4% MoM\", null, \"sm\", \"success\")\n\nkpi2 = Card([kpi2Label, kpi2Value, kpi2Tag])\nkpi2Label = TextContent(\"Top Channel\", \"small\")\nkpi2Value = TextContent(\"Organic Search\", \"large-heavy\")\nkpi2Tag = Tag(\"40.7% share\", null, \"sm\", \"info\")\n\nkpi3 = Card([kpi3Label, kpi3Value, kpi3Tag])\nkpi3Label = TextContent(\"Unpaid Share\", \"small\")\nkpi3Value = TextContent(\"82.8%\", \"large-heavy\")\nkpi3Tag = Tag(\"Organic + Direct + Social + Referral\", null, \"sm\", \"neutral\")\n\nkpi4 = Card([kpi4Label, kpi4Value, kpi4Tag])\nkpi4Label = TextContent(\"Fastest Growing\", \"small\")\nkpi4Value = TextContent(\"Referrals\", \"large-heavy\")\nkpi4Tag = Tag(\"+31.2% MoM\", null, \"sm\", \"success\")\n\nmainCard = Card([mainHeader, donut, shareBar, mainNote])\nmainHeader = CardHeader(\"Session Share by Channel\", \"Percentages sum to 100.0%\")\ndonut = PieChart(channelLabels, channelValues, \"donut\", \"circular\")\nshareBar = SingleStackedBarChart(channelLabels, channelValues)\nmainNote = TextContent(\"Slices are computed from raw session counts, so the five channels account for **100.0%** of measured traffic — no \\\"other\\\" or unattributed bucket.\", \"small\")\n\nchannelLabels = [\"Organic Search\", \"Direct\", \"Paid Ads\", \"Social\", \"Referrals\"]\nchannelValues = [52300, 28900, 22150, 15400, 9700]\n\ntabsSection = Tabs([tabTable, tabRanked, tabTrend])\n\ntabTable = TabItem(\"table\", \"Breakdown\", [breakdownTable, sumNote])\nbreakdownTable = Table([colChannel, colSessions, colShare, colMoM, colBadge])\ncolChannel = Col(\"Channel\", channelLabels)\ncolSessions = Col(\"Sessions\", channelValues, \"number\")\ncolShare = Col(\"Share %\", [40.7, 22.5, 17.2, 12.0, 7.6], \"number\")\ncolMoM = Col(\"MoM %\", [6.1, 3.4, 14.8, -2.7, 31.2], \"number\")\ncolBadge = Col(\"Trend\", [tagUp, tagUp2, tagUpStrong, tagDown, tagUpStrong2])\ntagUp = Tag(\"Up\", null, \"sm\", \"success\")\ntagUp2 = Tag(\"Up\", null, \"sm\", \"success\")\ntagUpStrong = Tag(\"Strong up\", null, \"sm\", \"success\")\ntagDown = Tag(\"Down\", null, \"sm\", \"warning\")\ntagUpStrong2 = Tag(\"Strong up\", null, \"sm\", \"success\")\nsumNote = TextContent(\"Total: **128,450 sessions** · Share column sums to **100.0%**\", \"small-heavy\")\n\ntabRanked = TabItem(\"ranked\", \"Ranked\", [rankedBars])\nrankedBars = HorizontalBarChart(channelLabels, [rankedSeries], \"grouped\", \"Sessions\", \"Channel\")\nrankedSeries = Series(\"Sessions\", channelValues)\n\ntabTrend = TabItem(\"trend\", \"3-Month Trend\", [trendChart])\ntrendChart = LineChart(trendMonths, [sOrganic, sDirect, sPaid, sSocial, sReferral], \"natural\", \"Month\", \"Sessions\")\ntrendMonths = [\"August\", \"September\", \"October\"]\nsOrganic = Series(\"Organic Search\", [46800, 49300, 52300])\nsDirect = Series(\"Direct\", [26100, 27950, 28900])\nsPaid = Series(\"Paid Ads\", [17400, 19300, 22150])\nsSocial = Series(\"Social\", [16900, 15830, 15400])\nsReferral = Series(\"Referrals\", [6800, 7390, 9700])\n\ninsightsCard = Card([insightsHeader, insightSteps, calloutPaid], \"sunk\")\ninsightsHeader = CardHeader(\"What Stands Out\", \"Three things worth acting on\")\ninsightSteps = Steps([step1, step2, step3])\nstep1 = StepsItem(\"Organic is the engine — and still compounding\", \"52,300 sessions (40.7%) and up 6.1% MoM for the third month running. The blog cluster on pricing comparisons drove most of the lift.\")\nstep2 = StepsItem(\"Referrals are the breakout channel\", \"9,700 sessions, up 31.2% MoM. Two partner newsletters and a Product Hunt mention account for roughly 4,100 of those sessions.\")\nstep3 = StepsItem(\"Social is the only channel shrinking\", \"15,400 sessions, down 2.7% MoM. Reach declined on the two largest accounts while posting cadence stayed flat — likely algorithmic, not content quality.\")\ncalloutPaid = TextCallout(\"info\", \"Paid ads grew fastest in absolute spend terms\", \"Paid sessions rose 14.8% to 22,150, but cost per session climbed 9%. Worth checking whether the incremental traffic is converting before scaling the budget further.\")\n\nfooterButtons = Buttons([btnConversion, btnSocial, btnSource], \"row\")\nbtnConversion = Button(\"Show conversion rate by channel\", Action([@ToAssistant(\"Break down conversion rate and revenue per session for each of these five traffic channels last month.\")]), \"primary\")\nbtnSocial = Button(\"Diagnose the social decline\", Action([@ToAssistant(\"Dig into why social traffic dropped 2.7% last month — break it down by platform and post type.\")]), \"secondary\")\nbtnSource = Button(\"Compare to last year\", Action([@ToAssistant(\"Compare last month's traffic source mix to the same month last year, and highlight the biggest shifts in channel share.\")]), \"secondary\")","promptTokens":6979,"completionTokens":2530,"latencyMs":28526,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"data-table/adversarial","repeat":3,"output":"","promptTokens":7016,"completionTokens":8192,"latencyMs":111809,"finishReason":"length"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/adversarial","repeat":1,"output":"root = Stack([header, chartCard, tableCard, insights])\n\nheader = CardHeader(\"Monthly Totals\", \"Jan – Jul\")\n\nchartCard = Card([chartHeader, views])\nchartHeader = CardHeader(\"Trend\", \"Peak in June at 7,100\")\nviews = Tabs([lineTab, barTab, areaTab])\n\nlineTab = TabItem(\"line\", \"Line\", [lineChart])\nbarTab = TabItem(\"bar\", \"Bars\", [barChart])\nareaTab = TabItem(\"area\", \"Area\", [areaChart])\n\nlineChart = LineChart(months, series, \"natural\", \"Month\", \"Value\")\nbarChart = BarChart(months, series, \"grouped\", \"Month\", \"Value\")\nareaChart = AreaChart(months, series, \"natural\", \"Month\", \"Value\")\n\nseries = [mainSeries]\nmainSeries = Series(\"Total\", values)\n\ntableCard = Card([tableHeader, table])\ntableHeader = CardHeader(\"Raw Numbers\", \"Values with month-over-month change\")\ntable = Table([monthCol, valueCol, changeCol, trendCol])\n\nmonthCol = Col(\"Month\", months)\nvalueCol = Col(\"Value\", values, \"number\")\nchangeCol = Col(\"MoM Change\", changes)\ntrendCol = Col(\"Direction\", trendTags)\n\ntrendTags = [tagUp0, tagDown, tagUp1, tagDown2, tagUp2, tagUp3, tagDown3]\ntagUp0 = Tag(\"baseline\", null, \"sm\", \"neutral\")\ntagDown = Tag(\"down\", null, \"sm\", \"danger\")\ntagUp1 = Tag(\"up\", null, \"sm\", \"success\")\ntagDown2 = Tag(\"down\", null, \"sm\", \"danger\")\ntagUp2 = Tag(\"up\", null, \"sm\", \"success\")\ntagUp3 = Tag(\"up\", null, \"sm\", \"success\")\ntagDown3 = Tag(\"down\", null, \"sm\", \"danger\")\n\ninsights = Card([kpis, note], \"sunk\")\nkpis = Stack([kpiTotal, kpiAvg, kpiPeak, kpiGrowth], \"row\", \"l\", \"stretch\", \"start\", true)\n\nkpiTotal = Card([kpiTotalLabel, kpiTotalValue])\nkpiTotalLabel = TextContent(\"Total\", \"small\")\nkpiTotalValue = TextContent(\"38,200\", \"large-heavy\")\n\nkpiAvg = Card([kpiAvgLabel, kpiAvgValue])\nkpiAvgLabel = TextContent(\"Average / month\", \"small\")\nkpiAvgValue = TextContent(\"5,457\", \"large-heavy\")\n\nkpiPeak = Card([kpiPeakLabel, kpiPeakValue])\nkpiPeakLabel = TextContent(\"Peak\", \"small\")\nkpiPeakValue = TextContent(\"Jun · 7,100\", \"large-heavy\")\n\nkpiGrowth = Card([kpiGrowthLabel, kpiGrowthValue])\nkpiGrowthLabel = TextContent(\"Jan → Jul\", \"small\")\nkpiGrowthValue = TextContent(\"+61.9%\", \"large-heavy\")\n\nnote = TextContent(\"Overall upward trend with two dips: **Feb** (-9.5%) and **Apr** (-3.9%), plus a pullback in **Jul** (-4.2%) after the June peak.\")\n\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nvalues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]\nchanges = [\"—\", \"-400 (-9.5%)\", \"+1,300 (+34.2%)\", \"-200 (-3.9%)\", \"+1,400 (+28.6%)\", \"+800 (+12.7%)\", \"-300 (-4.2%)\"]","promptTokens":7008,"completionTokens":1296,"latencyMs":13639,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/adversarial","repeat":0,"output":"root = Stack([header, kpiRow, chartCard, tableCard, notes])\n\nheader = CardHeader(\"Monthly Totals — Jan to Jul\", \"7 months of data, plotted and tabulated\")\n\nkpiRow = Stack([kpiTotal, kpiAvg, kpiPeak, kpiGrowth], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiTotal = Card([kpiTotalLabel, kpiTotalValue])\nkpiTotalLabel = TextContent(\"Total\", \"small\")\nkpiTotalValue = TextContent(\"38,200\", \"large-heavy\")\n\nkpiAvg = Card([kpiAvgLabel, kpiAvgValue])\nkpiAvgLabel = TextContent(\"Monthly average\", \"small\")\nkpiAvgValue = TextContent(\"5,457\", \"large-heavy\")\n\nkpiPeak = Card([kpiPeakLabel, kpiPeakValue])\nkpiPeakLabel = TextContent(\"Peak month\", \"small\")\nkpiPeakValue = TextContent(\"Jun · 7,100\", \"large-heavy\")\n\nkpiGrowth = Card([kpiGrowthLabel, kpiGrowthValue])\nkpiGrowthLabel = TextContent(\"Jan → Jul\", \"small\")\nkpiGrowthValue = TextContent(\"+61.9%\", \"large-heavy\")\n\nchartCard = Card([chartHeader, chartTabs])\nchartHeader = CardHeader(\"Trend\", \"Same series, three views\")\nchartTabs = Tabs([tabLine, tabBar, tabArea])\n\ntabLine = TabItem(\"line\", \"Line\", [lineChart])\ntabBar = TabItem(\"bar\", \"Bars\", [barChart])\ntabArea = TabItem(\"area\", \"Area\", [areaChart])\n\nlineChart = LineChart(months, series, \"natural\", \"Month\", \"Value\")\nbarChart = BarChart(months, series, \"grouped\", \"Month\", \"Value\")\nareaChart = AreaChart(months, series, \"natural\", \"Month\", \"Value\")\n\nseries = [mainSeries]\nmainSeries = Series(\"Value\", values)\n\ntableCard = Card([tableHeader, table])\ntableHeader = CardHeader(\"Raw numbers\", \"Exact values plus month-over-month change\")\ntable = Table([colMonth, colValue, colChange, colPct])\n\ncolMonth = Col(\"Month\", months)\ncolValue = Col(\"Value\", values, \"number\")\ncolChange = Col(\"Change\", changeTags)\ncolPct = Col(\"% change\", pctValues)\n\nchangeTags = [tagJan, tagFeb, tagMar, tagApr, tagMay, tagJun, tagJul]\ntagJan = Tag(\"—\", null, \"sm\", \"neutral\")\ntagFeb = Tag(\"-400\", null, \"sm\", \"danger\")\ntagMar = Tag(\"+1,300\", null, \"sm\", \"success\")\ntagApr = Tag(\"-200\", null, \"sm\", \"danger\")\ntagMay = Tag(\"+1,400\", null, \"sm\", \"success\")\ntagJun = Tag(\"+800\", null, \"sm\", \"success\")\ntagJul = Tag(\"-300\", null, \"sm\", \"danger\")\n\npctValues = [\"—\", \"-9.5%\", \"+34.2%\", \"-3.9%\", \"+28.6%\", \"+12.7%\", \"-4.2%\"]\n\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nvalues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]\n\nnotes = Callout(\"info\", \"Shape of the data\", \"Clear upward trend overall (+61.9% Jan→Jul), but it steps rather than climbs smoothly: dips in Feb and Apr, a strong run in May–Jun, then a slight pullback in Jul. Feb 3,800 is the floor, Jun 7,100 the ceiling.\")","promptTokens":7008,"completionTokens":1570,"latencyMs":16572,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/realistic","repeat":4,"output":"root = Stack([header, pieCard, kpiRow, tableCard, insightsCard, footerNote], \"column\", \"l\")\n\nheader = CardHeader(\"Traffic Sources — Last Month\", \"248,500 total sessions across 5 channels · Oct 1–31\")\n\npieCard = Card([pieHeader, trafficPie, shareBar])\npieHeader = CardHeader(\"Share of Sessions\", \"Percentages total 100%\")\ntrafficPie = PieChart(channelLabels, channelSessions, \"donut\")\nshareBar = SingleStackedBarChart(channelLabels, channelPercents)\n\nchannelLabels = [\"Organic Search\", \"Direct\", \"Paid Ads\", \"Social\", \"Referrals\"]\nchannelSessions = [104370, 54670, 44730, 27335, 17395]\nchannelPercents = [42, 22, 18, 11, 7]\n\nkpiRow = Stack([kpiOrganic, kpiDirect, kpiPaid, kpiSocial, kpiReferral], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiOrganic = Card([kpiOrganicLabel, kpiOrganicValue, kpiOrganicSub])\nkpiOrganicLabel = TextContent(\"Organic Search\", \"small\")\nkpiOrganicValue = TextContent(\"42.0%\", \"large-heavy\")\nkpiOrganicSub = TextContent(\"104,370 sessions\", \"small\")\n\nkpiDirect = Card([kpiDirectLabel, kpiDirectValue, kpiDirectSub])\nkpiDirectLabel = TextContent(\"Direct\", \"small\")\nkpiDirectValue = TextContent(\"22.0%\", \"large-heavy\")\nkpiDirectSub = TextContent(\"54,670 sessions\", \"small\")\n\nkpiPaid = Card([kpiPaidLabel, kpiPaidValue, kpiPaidSub])\nkpiPaidLabel = TextContent(\"Paid Ads\", \"small\")\nkpiPaidValue = TextContent(\"18.0%\", \"large-heavy\")\nkpiPaidSub = TextContent(\"44,730 sessions\", \"small\")\n\nkpiSocial = Card([kpiSocialLabel, kpiSocialValue, kpiSocialSub])\nkpiSocialLabel = TextContent(\"Social\", \"small\")\nkpiSocialValue = TextContent(\"11.0%\", \"large-heavy\")\nkpiSocialSub = TextContent(\"27,335 sessions\", \"small\")\n\nkpiReferral = Card([kpiReferralLabel, kpiReferralValue, kpiReferralSub])\nkpiReferralLabel = TextContent(\"Referrals\", \"small\")\nkpiReferralValue = TextContent(\"7.0%\", \"large-heavy\")\nkpiReferralSub = TextContent(\"17,395 sessions\", \"small\")\n\ntableCard = Card([tableHeader, trafficTable, totalNote])\ntableHeader = CardHeader(\"Channel Detail\", \"Sessions, share, and engagement quality\")\ntrafficTable = Table([colChannel, colSessions, colShare, colBounce, colConv, colTrend])\ncolChannel = Col(\"Channel\", channelLabels)\ncolSessions = Col(\"Sessions\", channelSessions, \"number\")\ncolShare = Col(\"Share %\", channelPercents, \"number\")\ncolBounce = Col(\"Bounce %\", [38.4, 41.2, 52.9, 61.5, 44.8], \"number\")\ncolConv = Col(\"Conv %\", [3.1, 3.8, 2.4, 1.2, 2.9], \"number\")\ncolTrend = Col(\"MoM\", trendTags)\ntrendTags = [tagOrganic, tagDirect, tagPaid, tagSocial, tagReferral]\ntagOrganic = Tag(\"+8.3%\", null, \"sm\", \"success\")\ntagDirect = Tag(\"+1.9%\", null, \"sm\", \"success\")\ntagPaid = Tag(\"-4.6%\", null, \"sm\", \"danger\")\ntagSocial = Tag(\"+12.7%\", null, \"sm\", \"success\")\ntagReferral = Tag(\"-0.8%\", null, \"sm\", \"warning\")\n\ntotalNote = TextContent(\"**Total:** 248,500 sessions — 42% + 22% + 18% + 11% + 7% = 100%\", \"small\")\n\ninsightsCard = Card([insightsHeader, insightsSteps, insightCallout])\ninsightsHeader = CardHeader(\"What Stands Out\", \"Three takeaways from the mix\")\ninsightsSteps = Steps([stepOrganic, stepPaid, stepSocial])\nstepOrganic = StepsItem(\"Organic is doing the heavy lifting\", \"At 42% of sessions with the lowest bounce rate (38.4%), organic search is both the largest and the highest-quality channel. Content investment from Q3 is compounding.\")\nstepPaid = StepsItem(\"Paid spend is losing efficiency\", \"Paid ads fell 4.6% month over month while holding 18% share, and its 52.9% bounce rate is the second-worst. Worth auditing creative and match types before renewing budget.\")\nstepSocial = StepsItem(\"Social is growing but shallow\", \"Social grew fastest (+12.7%) yet converts at only 1.2%. High reach, low intent — treat it as top-of-funnel rather than a revenue channel.\")\n\ninsightCallout = Callout(\"info\", \"Concentration risk\", \"Nearly two-thirds of traffic (64%) comes from organic search and direct combined. A single algorithm update could move a large share of your volume.\")\n\nfooterNote = TextContent(\"_Figures are sessions from last full calendar month; percentages rounded to one decimal and normalized to 100%._\", \"small\")","promptTokens":6979,"completionTokens":2121,"latencyMs":24045,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/adversarial","repeat":2,"output":"root = Stack([header, kpiRow, viewTabs, rawCard, note])\n\nheader = CardHeader(\"Monthly Totals — Jan to Jul\", \"7 months of data, with a raw number table for verification\")\n\nkpiRow = Stack([kpiTotal, kpiAvg, kpiPeak, kpiGrowth], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiTotal = Card([kpiTotalLabel, kpiTotalValue])\nkpiTotalLabel = TextContent(\"Total\", \"small\")\nkpiTotalValue = TextContent(\"38,200\", \"large-heavy\")\n\nkpiAvg = Card([kpiAvgLabel, kpiAvgValue])\nkpiAvgLabel = TextContent(\"Monthly average\", \"small\")\nkpiAvgValue = TextContent(\"5,457\", \"large-heavy\")\n\nkpiPeak = Card([kpiPeakLabel, kpiPeakValue])\nkpiPeakLabel = TextContent(\"Peak month\", \"small\")\nkpiPeakValue = TextContent(\"Jun — 7,100\", \"large-heavy\")\n\nkpiGrowth = Card([kpiGrowthLabel, kpiGrowthValue])\nkpiGrowthLabel = TextContent(\"Jan → Jul change\", \"small\")\nkpiGrowthValue = TextContent(\"+61.9%\", \"large-heavy\")\n\nviewTabs = Tabs([tabLine, tabBar, tabArea, tabShare])\n\ntabLine = TabItem(\"line\", \"Trend\", [lineChart])\ntabBar = TabItem(\"bar\", \"Bars\", [barChart])\ntabArea = TabItem(\"area\", \"Area\", [areaChart])\ntabShare = TabItem(\"share\", \"Share of total\", [pieChart])\n\nlineChart = LineChart(months, mainSeries, \"natural\", \"Month\", \"Value\")\nbarChart = BarChart(months, mainSeries, \"grouped\", \"Month\", \"Value\")\nareaChart = AreaChart(months, mainSeries, \"natural\", \"Month\", \"Value\")\npieChart = PieChart(months, values, \"donut\")\n\nmainSeries = [Series(\"Monthly total\", values)]\n\nrawCard = Card([rawHeader, rawTable])\nrawHeader = CardHeader(\"Raw numbers\", \"Exactly as provided, plus month-over-month deltas\")\nrawTable = Table([colMonth, colValue, colChange, colPct])\n\ncolMonth = Col(\"Month\", months, \"string\")\ncolValue = Col(\"Value\", values, \"number\")\ncolChange = Col(\"Change\", changeTags)\ncolPct = Col(\"Change %\", changePct, \"string\")\n\nchangeTags = [tagJan, tagFeb, tagMar, tagApr, tagMay, tagJun, tagJul]\ntagJan = Tag(\"—\", null, \"sm\", \"neutral\")\ntagFeb = Tag(\"-400\", null, \"sm\", \"danger\")\ntagMar = Tag(\"+1,300\", null, \"sm\", \"success\")\ntagApr = Tag(\"-200\", null, \"sm\", \"danger\")\ntagMay = Tag(\"+1,400\", null, \"sm\", \"success\")\ntagJun = Tag(\"+800\", null, \"sm\", \"success\")\ntagJul = Tag(\"-300\", null, \"sm\", \"danger\")\n\nchangePct = [\"—\", \"-9.5%\", \"+34.2%\", \"-3.9%\", \"+28.6%\", \"+12.7%\", \"-4.2%\"]\n\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nvalues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]\n\nnote = TextCallout(\"info\", \"Pattern\", \"Broadly upward: a soft February dip, then strong gains in March and May, peaking in June at 7,100 before a modest July pullback to 6,800.\")","promptTokens":7008,"completionTokens":1507,"latencyMs":15865,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/adversarial","repeat":4,"output":"root = Stack([header, kpiRow, chartCard, tableCard], \"column\", \"l\")\n\nheader = CardHeader(\"Monthly Totals — Jan to Jul\", \"Upward trend with a mid-year peak in June\")\n\nkpiRow = Stack([kpiTotal, kpiAvg, kpiPeak, kpiGrowth], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiTotal = Card([kpiTotalLabel, kpiTotalValue])\nkpiTotalLabel = TextContent(\"Total (7 months)\", \"small\")\nkpiTotalValue = TextContent(\"38,200\", \"large-heavy\")\n\nkpiAvg = Card([kpiAvgLabel, kpiAvgValue])\nkpiAvgLabel = TextContent(\"Monthly average\", \"small\")\nkpiAvgValue = TextContent(\"5,457\", \"large-heavy\")\n\nkpiPeak = Card([kpiPeakLabel, kpiPeakValue])\nkpiPeakLabel = TextContent(\"Peak month\", \"small\")\nkpiPeakValue = TextContent(\"Jun — 7,100\", \"large-heavy\")\n\nkpiGrowth = Card([kpiGrowthLabel, kpiGrowthValue])\nkpiGrowthLabel = TextContent(\"Jan → Jul change\", \"small\")\nkpiGrowthValue = TextContent(\"+61.9%\", \"large-heavy\")\n\nchartCard = Card([chartHeader, chartTabs])\nchartHeader = CardHeader(\"Trend\", \"Same data, three views\")\nchartTabs = Tabs([tabLine, tabBar, tabArea])\n\ntabLine = TabItem(\"line\", \"Line\", [lineChart])\ntabBar = TabItem(\"bar\", \"Bar\", [barChart])\ntabArea = TabItem(\"area\", \"Area\", [areaChart])\n\nlineChart = LineChart(months, series, \"natural\", \"Month\", \"Value\")\nbarChart = BarChart(months, series, \"grouped\", \"Month\", \"Value\")\nareaChart = AreaChart(months, series, \"natural\", \"Month\", \"Value\")\n\nseries = [mainSeries]\nmainSeries = Series(\"Value\", values)\n\ntableCard = Card([tableHeader, table, note])\ntableHeader = CardHeader(\"Raw numbers\", \"Exact values with month-over-month change\")\ntable = Table([colMonth, colValue, colChange, colPct])\ncolMonth = Col(\"Month\", months)\ncolValue = Col(\"Value\", values, \"number\")\ncolChange = Col(\"MoM change\", changes, \"number\")\ncolPct = Col(\"MoM %\", pcts)\nnote = TextContent(\"Sum = 38,200 · Min = 3,800 (Feb) · Max = 7,100 (Jun)\", \"small\")\n\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nvalues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]\nchanges = [0, -400, 1300, -200, 1400, 800, -300]\npcts = [\"—\", \"-9.5%\", \"+34.2%\", \"-3.9%\", \"+28.6%\", \"+12.7%\", \"-4.2%\"]","promptTokens":7008,"completionTokens":1274,"latencyMs":16111,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"chart/adversarial","repeat":3,"output":"root = Stack([header, kpis, chartCard, tableCard, notes])\n\nheader = CardHeader(\"Monthly Series — Jan to Jul\", \"Trend view with raw figures for verification\")\n\nkpis = Stack([kpiTotal, kpiAvg, kpiPeak, kpiGrowth], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiTotal = Card([kpiTotalLabel, kpiTotalValue])\nkpiTotalLabel = TextContent(\"Total\", \"small\")\nkpiTotalValue = TextContent(\"38,200\", \"large-heavy\")\n\nkpiAvg = Card([kpiAvgLabel, kpiAvgValue])\nkpiAvgLabel = TextContent(\"Monthly average\", \"small\")\nkpiAvgValue = TextContent(\"5,457\", \"large-heavy\")\n\nkpiPeak = Card([kpiPeakLabel, kpiPeakValue])\nkpiPeakLabel = TextContent(\"Peak month\", \"small\")\nkpiPeakValue = TextContent(\"Jun — 7,100\", \"large-heavy\")\n\nkpiGrowth = Card([kpiGrowthLabel, kpiGrowthValue])\nkpiGrowthLabel = TextContent(\"Jan → Jul change\", \"small\")\nkpiGrowthValue = TextContent(\"+61.9%\", \"large-heavy\")\n\nchartCard = Card([chartHeader, chartTabs])\nchartHeader = CardHeader(\"Trend\", \"Same data, three views\")\nchartTabs = Tabs([tabLine, tabBar, tabArea])\n\ntabLine = TabItem(\"line\", \"Line\", [lineChart])\ntabBar = TabItem(\"bar\", \"Bars\", [barChart])\ntabArea = TabItem(\"area\", \"Area\", [areaChart])\n\nlineChart = LineChart(months, series, \"natural\", \"Month\", \"Value\")\nbarChart = BarChart(months, series, \"grouped\", \"Month\", \"Value\")\nareaChart = AreaChart(months, series, \"natural\", \"Month\", \"Value\")\n\nseries = [mainSeries]\nmainSeries = Series(\"Value\", values)\n\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nvalues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]\n\ntableCard = Card([tableHeader, rawTable])\ntableHeader = CardHeader(\"Raw numbers\", \"Exact values plus month-over-month deltas\")\nrawTable = Table([colMonth, colValue, colDelta, colPct])\n\ncolMonth = Col(\"Month\", months)\ncolValue = Col(\"Value\", values, \"number\")\ncolDelta = Col(\"Change\", deltas, \"number\")\ncolPct = Col(\"Change %\", pcts)\n\ndeltas = [0, -400, 1300, -200, 1400, 800, -300]\npcts = [\"—\", \"-9.5%\", \"+34.2%\", \"-3.9%\", \"+28.6%\", \"+12.7%\", \"-4.2%\"]\n\nnotes = Callout(\"info\", \"Reading the shape\", \"Two dips (Feb and Apr, Jul) sit inside an overall upward trend: Feb is the low point at 3,800, momentum builds strongly through May–Jun peaking at 7,100, then Jul eases slightly to 6,800. Net movement from Jan to Jul is +2,600 (+61.9%).\")","promptTokens":7008,"completionTokens":1395,"latencyMs":17647,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/minimal","repeat":1,"output":"root = Stack([warningCallout, detailsCard, actionButtons], \"column\", \"l\")\n\nwarningCallout = Callout(\"warning\", \"Your trial expires in 3 days\", \"Your free trial ends on March 14. Upgrade now to keep access to your projects, data, and integrations without interruption.\")\n\ndetailsCard = Card([detailsHeader, detailsTable, expiryNote], \"card\")\n\ndetailsHeader = CardHeader(\"What happens when your trial ends\", \"Here's what to expect if no plan is selected\")\n\ndetailsTable = Table([featureCol, trialCol, afterCol])\n\nfeatureCol = Col(\"Feature\", [\"Projects\", \"Team members\", \"API access\", \"Data retention\", \"Support\"])\n\ntrialCol = Col(\"During trial\", [\"Unlimited\", \"Up to 10\", \"Enabled\", \"Full history\", \"Priority\"])\n\nafterCol = Col(\"After expiry\", [\"Read-only\", \"1 (owner)\", \"Disabled\", \"30 days, then deleted\", \"Community only\"])\n\nexpiryNote = TextContent(\"**Note:** Your data is retained for 30 days after expiry, so you can upgrade at any time within that window and pick up exactly where you left off.\", \"small\")\n\nactionButtons = Buttons([upgradeBtn, compareBtn, remindBtn], \"row\")\n\nupgradeBtn = Button(\"Upgrade now\", Action([@ToAssistant(\"Show me the available upgrade plans and pricing.\")]), \"primary\", \"normal\", \"medium\")\n\ncompareBtn = Button(\"Compare plans\", Action([@ToAssistant(\"Compare the plan tiers and what each includes.\")]), \"secondary\", \"normal\", \"medium\")\n\nremindBtn = Button(\"Remind me later\", Action([@ToAssistant(\"Remind me about my trial expiring in 2 days.\")]), \"tertiary\", \"normal\", \"medium\")","promptTokens":6945,"completionTokens":618,"latencyMs":8710,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/minimal","repeat":0,"output":"root = Stack([header, warning, details, actions])\n\nheader = CardHeader(\"Trial ending soon\", \"Your free trial expires in 3 days\")\n\nwarning = Callout(\"warning\", \"3 days left in your trial\", \"Your Pro trial ends on Friday. After that, your workspace will switch to the Free plan and premium features will be locked.\")\n\ndetails = Card([detailsHeader, whatHappens, planTable])\n\ndetailsHeader = CardHeader(\"What happens when the trial ends\", \"Here's what changes if you don't upgrade\")\n\nwhatHappens = Steps([step1, step2, step3])\n\nstep1 = StepsItem(\"Day 3 — Trial ends\", \"Your workspace is automatically moved to the Free plan. No charge is made.\")\nstep2 = StepsItem(\"Premium features lock\", \"Advanced analytics, unlimited exports, and team seats become read-only.\")\nstep3 = StepsItem(\"Data is kept for 30 days\", \"Nothing is deleted. Upgrade any time within 30 days to restore full access instantly.\")\n\nplanTable = Table([featureCol, trialCol, freeCol])\n\nfeatureCol = Col(\"Feature\", [\"Team seats\", \"Exports per month\", \"Analytics history\", \"Priority support\"])\ntrialCol = Col(\"Pro trial (now)\", [\"Unlimited\", \"Unlimited\", \"12 months\", \"Included\"])\nfreeCol = Col(\"Free plan (in 3 days)\", [\"1 seat\", \"5\", \"7 days\", \"Not included\"])\n\nactions = Buttons([upgradeBtn, compareBtn, remindBtn])\n\nupgradeBtn = Button(\"Upgrade to Pro\", Action([@ToAssistant(\"I'd like to upgrade to the Pro plan before my trial ends.\")]), \"primary\")\ncompareBtn = Button(\"Compare plans\", Action([@ToAssistant(\"Show me a detailed comparison of the Free, Pro, and Team plans.\")]), \"secondary\")\nremindBtn = Button(\"Remind me later\", Action([@ToAssistant(\"Remind me about my trial expiration tomorrow.\")]), \"tertiary\")","promptTokens":6945,"completionTokens":714,"latencyMs":9814,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/minimal","repeat":2,"output":"root = Stack([warningCallout, detailsCard, actionButtons])\n\nwarningCallout = Callout(\"warning\", \"Your trial expires in 3 days\", \"After that, access to premium features will be paused until you choose a plan.\")\n\ndetailsCard = Card([detailsHeader, trialSteps, separator, keepNote])\n\ndetailsHeader = CardHeader(\"What happens next\", \"Trial ends on your current billing cycle date\")\n\ntrialSteps = Steps([step1, step2, step3])\n\nstep1 = StepsItem(\"Day 3 — Reminder\", \"You'll get one final email notice 24 hours before your trial ends.\")\n\nstep2 = StepsItem(\"Day 0 — Trial ends\", \"Premium features are paused. Your account stays active and your data is safe.\")\n\nstep3 = StepsItem(\"Day 30 — Data retention limit\", \"Unused trial workspaces are archived after 30 days of inactivity.\")\n\nseparator = Separator(\"horizontal\")\n\nkeepNote = TextContent(\"**Nothing is deleted.** Upgrade any time to pick up exactly where you left off.\", \"small\")\n\nactionButtons = Buttons([upgradeBtn, plansBtn, remindBtn], \"row\")\n\nupgradeBtn = Button(\"Upgrade now\", Action([@ToAssistant(\"I'd like to upgrade before my trial expires — what are my options?\")]), \"primary\")\n\nplansBtn = Button(\"Compare plans\", Action([@ToAssistant(\"Show me a comparison of the available plans and pricing.\")]), \"secondary\")\n\nremindBtn = Button(\"Remind me later\", Action([@ToAssistant(\"Remind me about my trial expiration tomorrow.\")]), \"tertiary\")","promptTokens":6945,"completionTokens":557,"latencyMs":8711,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/minimal","repeat":3,"output":"root = Stack([trialCallout, detailsCard, actions])\n\ntrialCallout = Callout(\"warning\", \"Your trial expires in 3 days\", \"Your free trial ends on Friday. Upgrade now to keep access to your workspace, data, and integrations.\")\n\ndetailsCard = Card([detailsHeader, detailsTable, whatHappensNote])\n\ndetailsHeader = CardHeader(\"Trial summary\", \"Pro plan trial — 14 days\")\n\ndetailsTable = Table([colItem, colValue])\n\ncolItem = Col(\"Detail\", [\"Plan\", \"Days remaining\", \"Expires on\", \"Seats in use\", \"Payment method\"])\n\ncolValue = Col(\"Value\", [\"Pro (trial)\", \"3\", \"Friday, 14 March\", \"4 of 5\", \"None on file\"])\n\nwhatHappensNote = TextContent(\"**What happens if the trial ends?** Your workspace switches to read-only, scheduled reports pause, and integrations disconnect. Data is retained for 30 days.\", \"small\")\n\nactions = Buttons([upgradeBtn, compareBtn, remindBtn])\n\nupgradeBtn = Button(\"Upgrade now\", Action([@ToAssistant(\"I want to upgrade my trial to a paid plan.\")]), \"primary\")\n\ncompareBtn = Button(\"Compare plans\", Action([@ToAssistant(\"Show me a comparison of the available plans and pricing.\")]), \"secondary\")\n\nremindBtn = Button(\"Remind me tomorrow\", Action([@ToAssistant(\"Remind me about my trial expiration tomorrow.\")]), \"tertiary\")","promptTokens":6945,"completionTokens":548,"latencyMs":9858,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/minimal","repeat":4,"output":"root = Stack([warningCallout, detailCard, actionButtons], \"column\", \"l\")\n\nwarningCallout = Callout(\"warning\", \"Your trial expires in 3 days\", \"Your free trial ends on your billing anniversary. Upgrade now to keep access to all features without interruption.\")\n\ndetailCard = Card([detailHeader, trialTable, detailNote], \"card\")\n\ndetailHeader = CardHeader(\"Trial summary\", \"What happens when your trial ends\")\n\ntrialTable = Table([planCol, statusCol, endsCol])\n\nplanCol = Col(\"Plan\", [\"Pro Trial\", \"Team Seats\", \"API Access\"])\nstatusCol = Col(\"Status\", [activeTag, activeTag2, activeTag3])\nendsCol = Col(\"Days left\", [3, 3, 3], \"number\")\n\nactiveTag = Tag(\"Active\", null, \"sm\", \"success\")\nactiveTag2 = Tag(\"Active\", null, \"sm\", \"success\")\nactiveTag3 = Tag(\"Expiring\", null, \"sm\", \"warning\")\n\ndetailNote = TextContent(\"After the trial ends, your workspace switches to the **Free** plan. Projects stay intact, but automations, API keys, and team collaboration are paused until you upgrade.\", \"small\")\n\nactionButtons = Buttons([upgradeBtn, compareBtn, remindBtn], \"row\")\n\nupgradeBtn = Button(\"Upgrade now\", Action([@ToAssistant(\"Show me the upgrade options and pricing.\")]), \"primary\")\ncompareBtn = Button(\"Compare plans\", Action([@ToAssistant(\"Compare the Free, Pro, and Team plans for me.\")]), \"secondary\")\nremindBtn = Button(\"Remind me later\", Action([@ToAssistant(\"Remind me about my trial expiration tomorrow.\")]), \"tertiary\")","promptTokens":6945,"completionTokens":622,"latencyMs":9382,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/realistic","repeat":3,"output":"root = Stack([banner, headerCard, impactCard, timelineCard, faqCard, actionsCard], \"column\", \"l\")\n\nbanner = Callout(\"info\", \"Planned maintenance — not an outage\", \"Sunday 02:00–06:00 UTC. Payments and exports will be temporarily unavailable. All other services continue to run normally.\")\n\nheaderCard = Card([headerTitle, headerBody, windowTags])\n\nheaderTitle = CardHeader(\"Scheduled Maintenance Window\", \"Sunday, 02:00–06:00 UTC (4 hours)\")\n\nheaderBody = TextContent(\"This is **planned, scheduled work** — it is not an incident or an unplanned outage. We are upgrading our payment processing and data export infrastructure. The work is scheduled during our lowest-traffic window to minimise disruption.\\n\\nNo action is required from you. If you have a payment or export that must run during this window, please schedule it before 02:00 UTC or after 06:00 UTC on Sunday.\")\n\nwindowTags = TagBlock([\"Planned\", \"Sunday 02:00–06:00 UTC\", \"4 hour window\", \"No action required\"])\n\nimpactCard = Card([impactHeader, impactTable, impactNote])\n\nimpactHeader = CardHeader(\"What is affected\", \"Everything not listed as unavailable keeps working as normal\")\n\nimpactTable = Table([serviceCol, statusCol, notesCol])\n\nserviceCol = Col(\"Service\", [\"Card & bank payments\", \"Payouts\", \"Data exports (CSV/PDF)\", \"Scheduled export jobs\", \"Dashboard & reporting\", \"API (read endpoints)\", \"Login & account access\", \"Support chat\"])\n\nstatusCol = Col(\"During window\", [unavailableTag, unavailableTag2, unavailableTag3, queuedTag, availableTag, availableTag2, availableTag3, availableTag4])\n\nunavailableTag = Tag(\"Unavailable\", null, \"sm\", \"danger\")\nunavailableTag2 = Tag(\"Unavailable\", null, \"sm\", \"danger\")\nunavailableTag3 = Tag(\"Unavailable\", null, \"sm\", \"danger\")\nqueuedTag = Tag(\"Queued\", null, \"sm\", \"warning\")\navailableTag = Tag(\"Available\", null, \"sm\", \"success\")\navailableTag2 = Tag(\"Available\", null, \"sm\", \"success\")\navailableTag3 = Tag(\"Available\", null, \"sm\", \"success\")\navailableTag4 = Tag(\"Available\", null, \"sm\", \"success\")\n\nnotesCol = Col(\"Notes\", [\"New payments cannot be submitted or authorised\", \"Any payouts due are released after the window\", \"Export downloads will be disabled\", \"Jobs run automatically once we are back\", \"Full read access, live data\", \"Unchanged\", \"Unchanged\", \"Team available throughout\"])\n\nimpactNote = TextContent(\"Nothing is lost. Payment attempts made during the window will simply be declined at submission — please retry after 06:00 UTC. Scheduled exports are held in a queue and processed automatically.\", \"small\")\n\ntimelineCard = Card([timelineHeader, timelineSteps])\n\ntimelineHeader = CardHeader(\"Timeline (UTC)\", \"What to expect, hour by hour\")\n\ntimelineSteps = Steps([step1, step2, step3, step4])\n\nstep1 = StepsItem(\"01:45 — Pre-maintenance notice\", \"An in-app banner appears. Finish or pause any payment and export work at this point.\")\nstep2 = StepsItem(\"02:00 — Window opens\", \"Payments and exports are switched off. The rest of the platform stays fully online.\")\nstep3 = StepsItem(\"02:00–05:30 — Upgrade and verification\", \"Infrastructure work followed by internal testing before services are re-enabled.\")\nstep4 = StepsItem(\"06:00 — Window closes\", \"Payments and exports are back. Queued export jobs and pending payouts process automatically.\")\n\nfaqCard = Card([faqHeader, faq])\n\nfaqHeader = CardHeader(\"Questions you might have\", null)\n\nfaq = Accordion([faq1, faq2, faq3, faq4, faq5])\n\nfaq1 = AccordionItem(\"planned\", \"Is this an outage?\", [faq1Text])\nfaq1Text = TextContent(\"No. This is **planned maintenance** that we scheduled in advance and are notifying you about ahead of time. Our status page will show a scheduled maintenance entry, not an incident.\")\n\nfaq2 = AccordionItem(\"payments\", \"What happens to a payment attempted during the window?\", [faq2Text])\nfaq2Text = TextContent(\"It will not be accepted or charged. Customers will see a clear message asking them to try again shortly. No duplicate or partial charges are created.\")\n\nfaq3 = AccordionItem(\"exports\", \"Will I lose a scheduled export?\", [faq3Text])\nfaq3Text = TextContent(\"No. Scheduled exports that fall inside the window are queued and run automatically after 06:00 UTC. You will receive them as normal, just a few hours later.\")\n\nfaq4 = AccordionItem(\"overrun\", \"What if the work finishes early or runs long?\", [faq4Text])\nfaq4Text = TextContent(\"If we finish early we will restore services immediately. In the unlikely event of an overrun, we will post an update on the status page and email affected accounts.\")\n\nfaq5 = AccordionItem(\"action\", \"Do I need to do anything?\", [faq5Text])\nfaq5Text = TextContent(\"Only if you have time-critical payments or exports on Sunday morning. Move them outside 02:00–06:00 UTC and you will not notice the window at all.\")\n\nactionsCard = Card([actionsHeader, actionButtons], \"sunk\")\n\nactionsHeader = CardHeader(\"Need more detail?\", \"We are happy to help you plan around the window\")\n\nactionButtons = Buttons([btnStatus, btnConvert, btnAsk])\n\nbtnStatus = Button(\"View status page\", Action([@OpenUrl(\"https://status.example.com/scheduled/maintenance-sunday\")]), \"primary\")\nbtnConvert = Button(\"Convert to my local time\", Action([@ToAssistant(\"Convert the Sunday 02:00-06:00 UTC maintenance window into major local time zones (US Eastern, US Pacific, UK, CET, IST, SGT, AEST).\")]), \"secondary\")\nbtnAsk = Button(\"Draft a customer email\", Action([@ToAssistant(\"Draft a short customer-facing email announcing the planned Sunday 02:00-06:00 UTC maintenance, making clear payments and exports are unavailable and that this is planned, not an outage.\")]), \"secondary\")","promptTokens":7000,"completionTokens":2256,"latencyMs":26846,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/realistic","repeat":2,"output":"root = Stack([header, banner, summaryCard, impactCard, timelineCard, faqCard, footerNote, actions], \"column\", \"l\")\n\nheader = CardHeader(\"Scheduled Maintenance — Sunday 02:00–06:00 UTC\", \"Planned work. This is not an outage or incident.\")\n\nbanner = TextCallout(\"info\", \"This is planned maintenance\", \"We scheduled this window in advance during our lowest-traffic period. No action is required from you, and no data will be lost.\")\n\nsummaryCard = Card([summaryHeader, summaryText, summaryTags])\nsummaryHeader = CardHeader(\"What's happening\")\nsummaryText = TextContent(\"On **Sunday, 02:00–06:00 UTC** (up to 4 hours) we will upgrade our payments infrastructure and the data export pipeline. During the window, **payments and exports will be unavailable**. Everything else — logins, dashboards, API reads, webhooks and support — keeps working as normal.\")\nsummaryTags = TagBlock([\"Planned\", \"Sunday 02:00–06:00 UTC\", \"Max 4 hours\", \"No data loss\", \"No action required\"])\n\nimpactCard = Card([impactHeader, impactTable, impactNote])\nimpactHeader = CardHeader(\"Service impact at a glance\")\nimpactTable = Table([serviceCol, statusCol, notesCol])\nserviceCol = Col(\"Service\", [\"Card & bank payments\", \"Payouts\", \"Data exports (CSV / PDF)\", \"Scheduled report delivery\", \"Dashboard & reporting\", \"Sign-in & user accounts\", \"API (read endpoints)\", \"Webhooks\", \"Support chat & email\"])\nstatusCol = Col(\"Status\", [tagDown, tagDown, tagDown, tagDelayed, tagUp, tagUp, tagUp, tagUp, tagUp])\ntagDown = Tag(\"Unavailable\", null, \"sm\", \"danger\")\ntagDelayed = Tag(\"Delayed\", null, \"sm\", \"warning\")\ntagUp = Tag(\"Available\", null, \"sm\", \"success\")\nnotesCol = Col(\"What to expect\", [\"New charges will fail and can be retried after 06:00 UTC\", \"Queued and released after the window\", \"Export buttons disabled; requests are not queued\", \"Reports scheduled in the window send after 06:00 UTC\", \"Full read access, including historical data\", \"Normal sign-in\", \"Normal response times\", \"Delivered normally\", \"Normal response times\"])\nimpactNote = TextContent(\"Nothing is lost: any payment attempted during the window simply fails safely and can be retried. Exports can be re-requested once the window closes.\", \"small\")\n\ntimelineCard = Card([timelineHeader, timeline])\ntimelineHeader = CardHeader(\"Timeline (UTC)\")\ntimeline = Steps([step1, step2, step3, step4])\nstep1 = StepsItem(\"Saturday 20:00 — Reminder\", \"We send a final reminder email and post a notice in-product on the payments and exports screens.\")\nstep2 = StepsItem(\"Sunday 02:00 — Window opens\", \"Payments and exports are disabled. A banner in the app explains that maintenance is in progress.\")\nstep3 = StepsItem(\"Sunday ~05:00 — Verification\", \"Upgrades complete and we run verification checks. Services may come back before 06:00.\")\nstep4 = StepsItem(\"Sunday 06:00 — Window closes\", \"All services fully restored. Queued payouts and scheduled reports are processed automatically.\")\n\nfaqCard = Card([faqHeader, faq])\nfaqHeader = CardHeader(\"Questions you might have\")\nfaq = Accordion([faq1, faq2, faq3, faq4, faq5])\nfaq1 = AccordionItem(\"planned\", \"Is this an outage?\", [faq1Text])\nfaq1Text = TextContent(\"No. This is pre-scheduled maintenance chosen for our quietest traffic window. Our status page will show a **Scheduled Maintenance** entry, not an incident.\")\nfaq2 = AccordionItem(\"customers\", \"What will my own customers see?\", [faq2Text])\nfaq2Text = TextContent(\"If they attempt a payment during the window, they will see a clear 'temporarily unavailable, please try again shortly' message. No partial or duplicate charges are created.\")\nfaq3 = AccordionItem(\"prepare\", \"Do I need to do anything?\", [faq3Steps])\nfaq3Steps = Steps([prep1, prep2, prep3])\nprep1 = StepsItem(\"Run exports early\", \"If you need a CSV or PDF for Monday, generate it before Saturday 20:00 UTC.\")\nprep2 = StepsItem(\"Avoid scheduling campaigns\", \"Move any checkout-heavy campaign or invoice run outside 02:00–06:00 UTC.\")\nprep3 = StepsItem(\"Nothing else\", \"No configuration changes, credential rotation or redeployment is required on your side.\")\nfaq4 = AccordionItem(\"api\", \"Will my API integration break?\", [faq4Text])\nfaq4Text = TextContent(\"Read endpoints stay available. Payment and export endpoints will return HTTP **503** with a `Retry-After` header for the duration of the window — safe to retry with backoff.\")\nfaq5 = AccordionItem(\"overrun\", \"What if it takes longer than 4 hours?\", [faq5Text])\nfaq5Text = TextContent(\"We expect to finish early, but if anything runs long we will post updates on the status page every 30 minutes and email account admins.\")\n\nfooterNote = MarkDownRenderer(\"**Convert to your local time:** 02:00–06:00 UTC is 03:00–07:00 CET, 21:00–01:00 US Eastern (Saturday night), and 11:00–15:00 JST.\", \"sunk\")\n\nactions = Buttons([draftBtn, calBtn, statusBtn])\ndraftBtn = Button(\"Draft the customer email\", Action([@ToAssistant(\"Write a short, friendly customer notification email about the planned maintenance this Sunday 02:00-06:00 UTC, making clear it is scheduled and only affects payments and exports.\")]), \"primary\")\ncalBtn = Button(\"Convert to my timezone\", Action([@ToAssistant(\"Convert the Sunday 02:00-06:00 UTC maintenance window into major customer timezones and format it as a table.\")]), \"secondary\")\nstatusBtn = Button(\"View status page\", Action([@OpenUrl(\"https://status.example.com/maintenance\")]), \"tertiary\")","promptTokens":7000,"completionTokens":2255,"latencyMs":28509,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/realistic","repeat":0,"output":"root = Stack([headerCard, noticeCallout, detailsCard, tabs, faqAccordion, ctaCard])\n\nheaderCard = Card([header, tags])\nheader = CardHeader(\"Planned Maintenance — Sunday 02:00–06:00 UTC\", \"Scheduled in advance. This is not an outage.\")\ntags = TagBlock([\"Planned\", \"4-hour window\", \"Payments & Exports affected\", \"All other services online\"])\n\nnoticeCallout = Callout(\"info\", \"This is planned work, not an incident\", \"We are performing scheduled infrastructure maintenance. The window is fixed and announced ahead of time — nothing has gone wrong. If you experience an issue outside this window, please contact support.\")\n\ndetailsCard = Card([detailsHeader, whenText, windowTable])\ndetailsHeader = CardHeader(\"When it happens\", \"Same window shown in several time zones\")\nwhenText = TextContent(\"**Sunday, 02:00–06:00 UTC** (up to 4 hours). Most work is expected to finish earlier; we will close the window as soon as everything is verified.\", \"default\")\nwindowTable = Table([tzCol, startCol, endCol])\ntzCol = Col(\"Time zone\", [\"UTC\", \"US Pacific (PT)\", \"US Eastern (ET)\", \"Central Europe (CET)\", \"India (IST)\", \"Singapore (SGT)\", \"Sydney (AEDT)\"])\nstartCol = Col(\"Window starts\", [\"Sun 02:00\", \"Sat 18:00\", \"Sat 21:00\", \"Sun 03:00\", \"Sun 07:30\", \"Sun 10:00\", \"Sun 13:00\"])\nendCol = Col(\"Window ends\", [\"Sun 06:00\", \"Sat 22:00\", \"Sun 01:00\", \"Sun 07:00\", \"Sun 11:30\", \"Sun 14:00\", \"Sun 17:00\"])\n\ntabs = Tabs([impactTab, availableTab, timelineTab])\n\nimpactTab = TabItem(\"impact\", \"What is affected\", [impactText, impactTable])\nimpactText = TextContent(\"Two areas will be unavailable for the duration of the window. Requests to these endpoints will fail cleanly with a `503` and a `Retry-After` header rather than hanging.\", \"default\")\nimpactTable = Table([areaCol, statusCol, behaviourCol])\nareaCol = Col(\"Area\", [\"Card & bank payments\", \"Refunds and payouts\", \"Subscription renewals\", \"CSV / XLSX exports\", \"Scheduled report delivery\", \"Export API endpoints\"])\nstatusCol = Col(\"Status\", [tagDown, tagDown2, tagQueued, tagDown3, tagQueued2, tagDown4])\ntagDown = Tag(\"Unavailable\", null, \"sm\", \"danger\")\ntagDown2 = Tag(\"Unavailable\", null, \"sm\", \"danger\")\ntagDown3 = Tag(\"Unavailable\", null, \"sm\", \"danger\")\ntagDown4 = Tag(\"Unavailable\", null, \"sm\", \"danger\")\ntagQueued = Tag(\"Queued\", null, \"sm\", \"warning\")\ntagQueued2 = Tag(\"Queued\", null, \"sm\", \"warning\")\nbehaviourCol = Col(\"What happens\", [\"Checkout returns a retryable error\", \"Requests rejected, no partial state\", \"Automatically retried after 06:00 UTC\", \"Export button disabled in the UI\", \"Delivered after the window closes\", \"Return 503 with Retry-After\"])\n\navailableTab = TabItem(\"available\", \"What keeps working\", [availableText, availableTags])\navailableText = TextContent(\"Everything else runs normally throughout the window — no read-only mode, no login interruption.\", \"default\")\navailableTags = TagBlock([\"Sign-in & SSO\", \"Dashboards\", \"Search\", \"Create & edit records\", \"Comments & mentions\", \"Notifications\", \"Webhooks (non-payment)\", \"Mobile apps\", \"Support chat\"])\n\ntimelineTab = TabItem(\"timeline\", \"Timeline\", [timelineSteps])\ntimelineSteps = Steps([step1, step2, step3, step4, step5])\nstep1 = StepsItem(\"01:45 UTC — Banner goes live\", \"An in-app banner appears for all users confirming the window is about to begin.\")\nstep2 = StepsItem(\"02:00 UTC — Window opens\", \"Payments and exports stop accepting new requests. In-flight payments started before 02:00 are allowed to complete.\")\nstep3 = StepsItem(\"02:00–05:30 UTC — Maintenance work\", \"Database upgrade and storage migration for the payments and reporting subsystems.\")\nstep4 = StepsItem(\"05:30–06:00 UTC — Verification\", \"We run end-to-end checks on payments and exports before re-enabling traffic.\")\nstep5 = StepsItem(\"06:00 UTC — Services restored\", \"Queued renewals and scheduled reports begin processing. Status page updated to Operational.\")\n\nfaqAccordion = Accordion([faq1, faq2, faq3, faq4, faq5])\nfaq1 = AccordionItem(\"f1\", \"Is this an outage?\", [faq1Text])\nfaq1Text = TextContent(\"No. This is scheduled maintenance planned weeks in advance and announced ahead of time. The system is behaving exactly as intended during the window.\", \"default\")\nfaq2 = AccordionItem(\"f2\", \"Will I lose any data or payments?\", [faq2Text])\nfaq2Text = TextContent(\"No. Payment requests are rejected rather than partially processed, so there is no risk of duplicate or half-completed charges. Subscription renewals due during the window are queued and processed automatically afterwards.\", \"default\")\nfaq3 = AccordionItem(\"f3\", \"What should I do to prepare?\", [faq3Text])\nfaq3Text = TextContent(\"Run any critical exports before **Sunday 01:00 UTC**, and avoid scheduling payment-dependent launches or campaigns inside the window. No action is required beyond that.\", \"default\")\nfaq4 = AccordionItem(\"f4\", \"How should my integration handle it?\", [faq4Text, faq4Code])\nfaq4Text = TextContent(\"Treat `503` responses as retryable and honour the `Retry-After` header. No code changes are needed if you already retry with backoff.\", \"default\")\nfaq4Code = CodeBlock(\"http\", \"HTTP/1.1 503 Service Unavailable\\nRetry-After: 3600\\nContent-Type: application/json\\n\\n{\\n \\\"error\\\": \\\"scheduled_maintenance\\\",\\n \\\"message\\\": \\\"Payments are unavailable during scheduled maintenance.\\\",\\n \\\"window_ends_at\\\": \\\"2024-00-00T06:00:00Z\\\"\\n}\")\nfaq5 = AccordionItem(\"f5\", \"What if it runs long?\", [faq5Text])\nfaq5Text = TextContent(\"If work extends past 06:00 UTC we will post an update on the status page every 30 minutes until services are fully restored.\", \"default\")\n\nctaCard = Card([ctaHeader, ctaButtons])\nctaHeader = CardHeader(\"Stay informed\", \"Subscribe for live updates or reach out with questions\")\nctaButtons = Buttons([statusBtn, calendarBtn, questionBtn])\nstatusBtn = Button(\"Subscribe to status updates\", Action([@OpenUrl(\"https://status.example.com\")]), \"primary\")\ncalendarBtn = Button(\"Add window to calendar\", Action([@OpenUrl(\"https://status.example.com/maintenance.ics\")]), \"secondary\")\nquestionBtn = Button(\"Ask a question about the maintenance\", Action([@ToAssistant(\"I have a question about the scheduled maintenance this Sunday 02:00-06:00 UTC.\")]), \"tertiary\")","promptTokens":7000,"completionTokens":2610,"latencyMs":31720,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/realistic","repeat":1,"output":"root = Stack([banner, headerCard, detailsCard, tabs, faqAccordion, actionsCard], \"column\", \"l\")\n\nbanner = Callout(\"info\", \"Planned maintenance — not an outage\", \"This is scheduled work we planned in advance. Your data is safe and no action is required from you.\")\n\nheaderCard = Card([header, windowTags, summaryText])\nheader = CardHeader(\"Scheduled Maintenance: Sunday 02:00–06:00 UTC\", \"Approximately 4 hours · Payments & Exports temporarily unavailable\")\nwindowTags = TagBlock([\"Planned\", \"Sunday 02:00–06:00 UTC\", \"~4 hours\", \"Partial impact\"])\nsummaryText = TextContent(\"We are performing planned infrastructure maintenance to improve reliability and speed. During this window, **payment processing** and **data exports** will be paused. Everything else — logins, dashboards, API reads, and reporting views — keeps working normally.\")\n\ndetailsCard = Card([detailsHeader, impactTable])\ndetailsHeader = CardHeader(\"What is affected\", \"Service-by-service breakdown for the maintenance window\")\nimpactTable = Table([serviceCol, statusCol, notesCol])\nserviceCol = Col(\"Service\", [\"Payments & checkout\", \"Subscription renewals\", \"Data exports (CSV/PDF)\", \"Scheduled report delivery\", \"Dashboards & analytics\", \"Login & accounts\", \"API (read)\", \"API (write)\", \"Webhooks\", \"Support chat\"])\nstatusCol = Col(\"Status\", [tagDown, tagDelay, tagDown, tagDelay, tagUp, tagUp, tagUp, tagUp, tagDelay, tagUp])\ntagDown = Tag(\"Unavailable\", null, \"sm\", \"danger\")\ntagDelay = Tag(\"Delayed\", null, \"sm\", \"warning\")\ntagUp = Tag(\"Operational\", null, \"sm\", \"success\")\nnotesCol = Col(\"Notes\", [\"New charges cannot be processed\", \"Retried automatically after 06:00 UTC\", \"Requests can be queued, not generated\", \"Queued and delivered after the window\", \"Full read access, data current to 02:00 UTC\", \"No impact expected\", \"No impact expected\", \"Brief retries possible near 02:00 UTC\", \"Queued and replayed after the window\", \"Team available throughout\"])\n\ntabs = Tabs([timelineTab, localTimeTab, prepTab])\n\ntimelineTab = TabItem(\"timeline\", \"Timeline\", [timelineSteps])\ntimelineSteps = Steps([step1, step2, step3, step4])\nstep1 = StepsItem(\"01:45 UTC — Pre-flight notice\", \"An in-app banner appears reminding anyone mid-checkout to complete or postpone their payment.\")\nstep2 = StepsItem(\"02:00 UTC — Maintenance begins\", \"Payments and exports enter read-only/paused mode. All other services continue as normal.\")\nstep3 = StepsItem(\"02:00–05:30 UTC — Upgrade work\", \"Database migration and export pipeline upgrade, followed by internal verification checks.\")\nstep4 = StepsItem(\"05:30–06:00 UTC — Restore and verify\", \"Payments resume, queued exports and webhooks are processed, and we confirm full service before closing the window.\")\n\nlocalTimeTab = TabItem(\"local\", \"Local times\", [localTable, localNote])\nlocalTable = Table([tzCol, startCol, endCol])\ntzCol = Col(\"Region\", [\"UTC\", \"London (BST)\", \"Berlin (CEST)\", \"New York (EDT)\", \"São Paulo (BRT)\", \"Dubai (GST)\", \"Singapore (SGT)\", \"Sydney (AEST)\"])\nstartCol = Col(\"Start\", [\"Sun 02:00\", \"Sun 03:00\", \"Sun 04:00\", \"Sat 22:00\", \"Sat 23:00\", \"Sun 06:00\", \"Sun 10:00\", \"Sun 12:00\"])\nendCol = Col(\"End\", [\"Sun 06:00\", \"Sun 07:00\", \"Sun 08:00\", \"Sun 02:00\", \"Sun 03:00\", \"Sun 10:00\", \"Sun 14:00\", \"Sun 16:00\"])\nlocalNote = TextContent(\"Times shown for the upcoming Sunday. Daylight saving offsets are already applied.\", \"small\")\n\nprepTab = TabItem(\"prep\", \"How to prepare\", [prepList])\nprepList = MarkDownRenderer(\"**If you take payments:** no configuration change is needed. Any charge attempted between 02:00 and 06:00 UTC will fail cleanly with a retryable error — no double charges, no partial captures.\\n\\n**If you rely on exports:** run any critical CSV or PDF export before 01:30 UTC Sunday, or schedule it for after 06:00 UTC.\\n\\n**If you use webhooks:** events are queued during the window and replayed automatically afterwards. Make sure your endpoint tolerates a delayed burst of events.\\n\\n**If you have a weekend release planned:** avoid write-heavy API jobs between 02:00 and 03:00 UTC, when brief retries are most likely.\")\n\nfaqAccordion = Accordion([faq1, faq2, faq3, faq4])\nfaq1 = AccordionItem(\"f1\", \"Is this an outage?\", [faq1Text])\nfaq1Text = TextContent(\"No. This is planned maintenance, scheduled deliberately during our lowest-traffic window and announced in advance. Nothing has broken, and no incident is in progress.\")\nfaq2 = AccordionItem(\"f2\", \"Will I lose any data?\", [faq2Text])\nfaq2Text = TextContent(\"No data is lost. Records created before 02:00 UTC are fully preserved, and anything queued during the window (webhooks, export requests, renewal retries) is processed once maintenance completes.\")\nfaq3 = AccordionItem(\"f3\", \"What happens to a payment attempted during the window?\", [faq3Text])\nfaq3Text = TextContent(\"It is declined with a clear retryable error rather than being left in a pending state. Your customer can simply try again after 06:00 UTC, and automatic subscription renewals retry on their own.\")\nfaq4 = AccordionItem(\"f4\", \"Could the window run long?\", [faq4Text])\nfaq4Text = TextContent(\"We expect to finish by 06:00 UTC or earlier. If we need more time, we will post updates on the status page every 30 minutes until services are fully restored.\")\n\nactionsCard = Card([actionsHeader, actionButtons])\nactionsHeader = CardHeader(\"Need anything else?\", \"Follow along live or reach out before the window opens\")\nactionButtons = Buttons([statusBtn, subscribeBtn, exportBtn, contactBtn])\nstatusBtn = Button(\"View status page\", Action([@OpenUrl(\"https://status.example.com\")]), \"primary\")\nsubscribeBtn = Button(\"Subscribe to updates\", Action([@OpenUrl(\"https://status.example.com/subscribe\")]), \"secondary\")\nexportBtn = Button(\"Draft a customer email about this\", Action([@ToAssistant(\"Draft a short customer-facing email announcing the planned maintenance this Sunday 02:00-06:00 UTC, noting that payments and exports are unavailable and everything else keeps working.\")]), \"secondary\")\ncontactBtn = Button(\"Contact support\", Action([@ToAssistant(\"I have a question about the scheduled maintenance window this Sunday.\")]), \"tertiary\")","promptTokens":7000,"completionTokens":2530,"latencyMs":30675,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/adversarial","repeat":0,"output":"root = Stack([topAlert, hugeBanner, statusRow, detailsCard, logCard, actionCard], \"column\", \"l\")\n\ntopAlert = Callout(\"error\", \"❌ FAILED — ACTION REQUIRED\", \"The operation did not complete. Nothing was saved. Read the details below before retrying.\")\n\nhugeBanner = Card([bannerStack], \"card\")\nbannerStack = Stack([bannerTags, bannerBig, bannerSub, bannerRule, bannerLine], \"column\", \"s\", \"center\")\nbannerTags = Stack([tagFailed, tagCritical, tagBlocked, tagNoRetry], \"row\", \"xs\", \"center\", \"center\", true)\ntagFailed = Tag(\"FAILED\", null, \"lg\", \"danger\")\ntagCritical = Tag(\"SEVERITY: CRITICAL\", null, \"lg\", \"danger\")\ntagBlocked = Tag(\"BLOCKED\", null, \"lg\", \"danger\")\ntagNoRetry = Tag(\"0 of 3 steps completed\", null, \"lg\", \"warning\")\nbannerBig = TextContent(\"# 🚨 IT FAILED 🚨\", \"large-heavy\")\nbannerSub = TextContent(\"## ⛔ ⛔ ⛔ **NOT SUCCESSFUL** ⛔ ⛔ ⛔\", \"large-heavy\")\nbannerRule = Separator(\"horizontal\")\nbannerLine = TextContent(\"**This did not work. It stopped. It is not running. There is no partial result.**\", \"large-heavy\")\n\nstatusRow = Stack([kpiStatus, kpiCode, kpiTime, kpiRetries], \"row\", \"m\", \"stretch\", \"start\", true)\n\nkpiStatus = Card([kpiStatusLabel, kpiStatusValue, kpiStatusTag], \"card\")\nkpiStatusLabel = TextContent(\"Status\", \"small\")\nkpiStatusValue = TextContent(\"FAILED\", \"large-heavy\")\nkpiStatusTag = Tag(\"terminal state\", null, \"sm\", \"danger\")\n\nkpiCode = Card([kpiCodeLabel, kpiCodeValue, kpiCodeTag], \"card\")\nkpiCodeLabel = TextContent(\"Exit code\", \"small\")\nkpiCodeValue = TextContent(\"1\", \"large-heavy\")\nkpiCodeTag = Tag(\"non-zero\", null, \"sm\", \"danger\")\n\nkpiTime = Card([kpiTimeLabel, kpiTimeValue, kpiTimeTag], \"card\")\nkpiTimeLabel = TextContent(\"Failed after\", \"small\")\nkpiTimeValue = TextContent(\"12.4s\", \"large-heavy\")\nkpiTimeTag = Tag(\"aborted early\", null, \"sm\", \"warning\")\n\nkpiRetries = Card([kpiRetriesLabel, kpiRetriesValue, kpiRetriesTag], \"card\")\nkpiRetriesLabel = TextContent(\"Attempts\", \"small\")\nkpiRetriesValue = TextContent(\"3 / 3\", \"large-heavy\")\nkpiRetriesTag = Tag(\"all failed\", null, \"sm\", \"danger\")\n\ndetailsCard = Card([detailsHeader, detailsCallout, detailsTable, stepsHeader, failSteps], \"card\")\ndetailsHeader = CardHeader(\"What went wrong\", \"Every attempt ended in the same error\")\ndetailsCallout = TextCallout(\"danger\", \"Root cause: connection refused\", \"The target endpoint rejected all three connection attempts. No data was written, so nothing needs to be rolled back.\")\n\ndetailsTable = Table([colField, colValue, colState])\ncolField = Col(\"Field\", [\"Operation\", \"Target\", \"Error type\", \"Error code\", \"Timestamp\", \"Data written\"])\ncolValue = Col(\"Value\", [\"deploy:release\", \"api-prod-eu-west-1\", \"ConnectionRefusedError\", \"ECONNREFUSED\", \"2026-02-11 14:32:07 UTC\", \"None\"])\ncolState = Col(\"State\", [stateFail, stateFail2, stateFail3, stateFail4, stateInfo, stateOk])\nstateFail = Tag(\"failed\", null, \"sm\", \"danger\")\nstateFail2 = Tag(\"unreachable\", null, \"sm\", \"danger\")\nstateFail3 = Tag(\"fatal\", null, \"sm\", \"danger\")\nstateFail4 = Tag(\"ECONNREFUSED\", null, \"sm\", \"danger\")\nstateInfo = Tag(\"logged\", null, \"sm\", \"neutral\")\nstateOk = Tag(\"clean\", null, \"sm\", \"success\")\n\nstepsHeader = CardHeader(\"Where it stopped\", \"It never got past step 1\")\nfailSteps = Steps([step1, step2, step3])\nstep1 = StepsItem(\"1. Connect to target — ❌ FAILED\", \"Three attempts, all refused (ECONNREFUSED). This is where everything stopped.\")\nstep2 = StepsItem(\"2. Upload payload — ⏭ NEVER RAN\", \"Skipped because step 1 failed. Nothing was transferred.\")\nstep3 = StepsItem(\"3. Switch traffic — ⏭ NEVER RAN\", \"Skipped. Live traffic is untouched and still on the previous version.\")\n\nlogCard = Card([logHeader, logBlock], \"sunk\")\nlogHeader = CardHeader(\"Error output\", \"Last 9 lines of the log\")\nlogBlock = CodeBlock(\"bash\", \"[14:32:03] attempt 1/3 -> api-prod-eu-west-1:8443\\n[14:32:04] ERROR connect ECONNREFUSED 10.4.19.22:8443\\n[14:32:05] attempt 2/3 -> api-prod-eu-west-1:8443\\n[14:32:06] ERROR connect ECONNREFUSED 10.4.19.22:8443\\n[14:32:07] attempt 3/3 -> api-prod-eu-west-1:8443\\n[14:32:07] ERROR connect ECONNREFUSED 10.4.19.22:8443\\n[14:32:07] FATAL all attempts exhausted\\n[14:32:07] rollback: nothing to undo (0 bytes written)\\n[14:32:07] exit status 1 ***FAILED***\")\n\nactionCard = Card([actionHeader, actionText, actionButtons], \"card\")\nactionHeader = CardHeader(\"What to do next\", \"Pick one\")\nactionText = TextContent(\"**Nothing is fixed yet.** Retrying without addressing the refused connection will fail again in the same way.\", \"large-heavy\")\nactionButtons = Buttons([btnWhy, btnFix, btnRetry, btnLogs], \"row\")\nbtnWhy = Button(\"Explain why it failed\", Action([@ToAssistant(\"Explain in detail why the deploy failed with ECONNREFUSED and what usually causes it.\")]), \"primary\")\nbtnFix = Button(\"Show me how to fix it\", Action([@ToAssistant(\"Give me step-by-step instructions to fix the ECONNREFUSED error on api-prod-eu-west-1 and verify before retrying.\")]), \"secondary\")\nbtnRetry = Button(\"Retry anyway\", Action([@ToAssistant(\"Retry the failed deploy now, even though the root cause may not be fixed.\")]), \"secondary\", \"destructive\")\nbtnLogs = Button(\"Open full logs\", Action([@OpenUrl(\"https://example.com/logs/deploy-release/latest\")]), \"tertiary\")","promptTokens":6971,"completionTokens":2730,"latencyMs":31873,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/realistic","repeat":4,"output":"root = Stack([banner, headerCard, detailsCard, impactTabs, timelineCard, faqAccordion, actionsCard], \"column\", \"l\")\n\nbanner = Callout(\"info\", \"Planned maintenance — not an outage\", \"Sunday 02:00–06:00 UTC. Payments and exports will be temporarily unavailable. All other services continue as normal.\")\n\nheaderCard = Card([header, headerTags])\nheader = CardHeader(\"Scheduled Maintenance Window\", \"Sunday, 02:00–06:00 UTC (4 hours)\")\nheaderTags = TagBlock([\"Planned\", \"Payments\", \"Exports\", \"4 hour window\", \"No action required\"])\n\ndetailsCard = Card([detailsHeader, detailsText, detailsSeparator, statusTable])\ndetailsHeader = CardHeader(\"What is happening\", \"We are upgrading our database and export pipeline\")\ndetailsText = TextContent(\"This is **planned work**, scheduled in advance during our lowest-traffic window. It is **not an incident or outage**. We are upgrading storage infrastructure and the export pipeline so that large report generation is faster and more reliable.\\n\\nDuring the window, most of the product keeps working normally. Only payment processing and data exports pause.\")\ndetailsSeparator = Separator(\"horizontal\")\nstatusTable = Table([colService, colStatus, colNote])\ncolService = Col(\"Service\", [\"Payments & checkout\", \"Subscription renewals\", \"Data exports / report downloads\", \"App & dashboard\", \"API (read)\", \"API (write)\", \"Login & SSO\", \"Notifications & email\"])\ncolStatus = Col(\"During window\", [tagDown, tagDelayed, tagDown, tagUp, tagUp, tagUp, tagUp, tagUp])\ntagDown = Tag(\"Unavailable\", null, \"sm\", \"danger\")\ntagDelayed = Tag(\"Deferred\", null, \"sm\", \"warning\")\ntagUp = Tag(\"Available\", null, \"sm\", \"success\")\ncolNote = Col(\"Notes\", [\"New charges cannot be completed\", \"Retried automatically after 06:00 UTC\", \"Queued exports run after the window\", \"Full read and write access\", \"No expected impact\", \"No expected impact\", \"No expected impact\", \"Sent as usual\"])\n\nimpactTabs = Tabs([tabCustomers, tabDevs, tabTimezones])\ntabCustomers = TabItem(\"customers\", \"What you should do\", [customersText, customersSteps])\ncustomersText = TextContent(\"For almost everyone, the answer is: nothing. If you or your users need to pay or export during those four hours, plan around the window.\")\ncustomersSteps = Steps([step1, step2, step3])\nstep1 = StepsItem(\"Finish payments before 02:00 UTC\", \"Complete any checkout, invoice payment or plan change before the window opens. Payments attempted during the window will fail cleanly with a retryable error, not a double charge.\")\nstep2 = StepsItem(\"Pull exports early or wait\", \"Trigger any CSV, PDF or scheduled report exports before 02:00 UTC. Exports requested during the window are queued and processed automatically once we are back.\")\nstep3 = StepsItem(\"Nothing else required\", \"No configuration changes, no re-login, no cache clearing. Your data is untouched and the rest of the product stays online.\")\n\ntabDevs = TabItem(\"devs\", \"For developers\", [devsText, devsCode])\ndevsText = TextContent(\"Payment and export endpoints will return HTTP **503** with a `maintenance` code and a `Retry-After` header for the duration of the window. All other endpoints behave normally. We recommend treating 503 as retryable rather than as a hard failure.\")\ndevsCode = CodeBlock(\"json\", \"HTTP/1.1 503 Service Unavailable\\nRetry-After: 3600\\nContent-Type: application/json\\n\\n{\\n \\\"error\\\": {\\n \\\"code\\\": \\\"maintenance\\\",\\n \\\"message\\\": \\\"Payments are temporarily unavailable due to planned maintenance.\\\",\\n \\\"window\\\": {\\n \\\"start\\\": \\\"Sunday 02:00Z\\\",\\n \\\"end\\\": \\\"Sunday 06:00Z\\\"\\n },\\n \\\"retryable\\\": true\\n }\\n}\")\n\ntabTimezones = TabItem(\"tz\", \"Local times\", [tzText, tzTable])\ntzText = TextContent(\"The window is 02:00–06:00 UTC. Converted to common time zones:\")\ntzTable = Table([colZone, colStart, colEnd])\ncolZone = Col(\"Time zone\", [\"UTC\", \"London (BST)\", \"Berlin / Paris (CEST)\", \"New York (EDT)\", \"San Francisco (PDT)\", \"Mumbai (IST)\", \"Singapore (SGT)\", \"Sydney (AEST)\"])\ncolStart = Col(\"Starts\", [\"Sun 02:00\", \"Sun 03:00\", \"Sun 04:00\", \"Sat 22:00\", \"Sat 19:00\", \"Sun 07:30\", \"Sun 10:00\", \"Sun 12:00\"])\ncolEnd = Col(\"Ends\", [\"Sun 06:00\", \"Sun 07:00\", \"Sun 08:00\", \"Sun 02:00\", \"Sat 23:00\", \"Sun 11:30\", \"Sun 14:00\", \"Sun 16:00\"])\n\ntimelineCard = Card([timelineHeader, timelineSteps])\ntimelineHeader = CardHeader(\"Maintenance timeline\", \"All times UTC\")\ntimelineSteps = Steps([t1, t2, t3, t4])\nt1 = StepsItem(\"01:45 — Pre-window notice\", \"In-app banner appears. Payments and exports still fully available.\")\nt2 = StepsItem(\"02:00 — Window opens\", \"Payment processing and export generation pause. Everything else continues serving traffic normally.\")\nt3 = StepsItem(\"05:00 — Verification\", \"Upgrades complete; we run automated checks against payments and exports before re-enabling them.\")\nt4 = StepsItem(\"06:00 — Window closes\", \"Payments resume, queued exports and deferred renewals process automatically. Status page updated to Operational.\")\n\nfaqAccordion = Accordion([faq1, faq2, faq3, faq4, faq5])\nfaq1 = AccordionItem(\"q1\", \"Is this an outage?\", [faq1Text])\nfaq1Text = TextContent(\"No. This is planned, pre-announced maintenance during our lowest-traffic window. We are choosing to pause two subsystems briefly so we can upgrade them safely. An outage is unplanned; this is scheduled and reversible.\")\nfaq2 = AccordionItem(\"q2\", \"Will a payment made just before the window be affected?\", [faq2Text])\nfaq2Text = TextContent(\"No. Any payment that is authorised and captured before 02:00 UTC completes normally. Payments started but not completed at 02:00 UTC will fail cleanly and can simply be retried after 06:00 UTC — there is no risk of a duplicate charge.\")\nfaq3 = AccordionItem(\"q3\", \"What happens to my scheduled exports and reports?\", [faq3Text])\nfaq3Text = TextContent(\"They are queued, not dropped. Any scheduled export whose run time falls inside the window will execute automatically once the window closes, typically within 15–30 minutes of 06:00 UTC.\")\nfaq4 = AccordionItem(\"q4\", \"Could the window run long?\", [faq4Text])\nfaq4Text = TextContent(\"The work is planned to finish before 06:00 UTC, and we often finish early. If we need extra time, we will post an update on the status page every 30 minutes until the window is closed.\")\nfaq5 = AccordionItem(\"q5\", \"Is my data at risk?\", [faq5Text])\nfaq5Text = TextContent(\"No. We take verified backups immediately before the window and the change is fully reversible. No customer data is modified, migrated or deleted as part of this work.\")\n\nactionsCard = Card([actionsHeader, actionButtons], \"sunk\")\nactionsHeader = CardHeader(\"Stay informed\", \"Subscribe to updates or ask us anything\")\nactionButtons = Buttons([btnStatus, btnSubscribe, btnRemind, btnAsk])\nbtnStatus = Button(\"View status page\", Action([@OpenUrl(\"https://status.example.com\")]), \"primary\")\nbtnSubscribe = Button(\"Subscribe to updates\", Action([@OpenUrl(\"https://status.example.com/subscribe\")]), \"secondary\")\nbtnRemind = Button(\"Add to calendar\", Action([@OpenUrl(\"https://status.example.com/maintenance.ics\")]), \"secondary\")\nbtnAsk = Button(\"Ask about impact on my account\", Action([@ToAssistant(\"How will the Sunday 02:00-06:00 UTC maintenance window affect my account specifically?\")]), \"tertiary\")","promptTokens":7000,"completionTokens":2969,"latencyMs":36204,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/adversarial","repeat":1,"output":"root = Stack([topBanner, bigFail, detailCard, whatHappened, nextSteps, actions], \"column\", \"l\")\n\ntopBanner = Callout(\"error\", \"🚨 FAILED 🚨\", \"The operation did not complete. Nothing was saved.\")\n\nbigFail = Card([failTitle, failSub, failTags], \"card\", \"column\", \"s\", \"center\")\nfailTitle = TextContent(\"# ❌ FAILED\", \"large-heavy\")\nfailSub = TextContent(\"**Status: ERROR — the process stopped and rolled back.**\", \"large\")\nfailTags = TagBlock([\"FAILED\", \"ERROR\", \"ABORTED\", \"NOTHING SAVED\", \"ACTION REQUIRED\"])\n\ndetailCard = Card([detailHeader, detailTable])\ndetailHeader = CardHeader(\"Failure summary\", \"Every stage after step 2 was aborted\")\ndetailTable = Table([stageCol, statusCol, noteCol])\nstageCol = Col(\"Stage\", [\"1. Validate input\", \"2. Connect\", \"3. Transfer\", \"4. Verify\", \"5. Commit\"])\nstatusCol = Col(\"Status\", [okTag, failTag, skipTag, skipTag2, skipTag3])\nnoteCol = Col(\"Detail\", [\"Passed\", \"Timed out after 30s\", \"Skipped\", \"Skipped\", \"Rolled back\"])\nokTag = Tag(\"OK\", null, \"sm\", \"success\")\nfailTag = Tag(\"FAILED\", null, \"lg\", \"danger\")\nskipTag = Tag(\"SKIPPED\", null, \"sm\", \"neutral\")\nskipTag2 = Tag(\"SKIPPED\", null, \"sm\", \"neutral\")\nskipTag3 = Tag(\"ROLLED BACK\", null, \"sm\", \"danger\")\n\nwhatHappened = Card([errBlockHeader, errBlock], \"sunk\")\nerrBlockHeader = CardHeader(\"Error output\", \"Exit code 1\")\nerrBlock = CodeBlock(\"bash\", \"ERROR: operation failed\\n ↳ stage: connect\\n ↳ cause: ETIMEDOUT (30000ms)\\n ↳ retries: 3/3 exhausted\\n ↳ committed: false\\n\\nprocess exited with code 1\")\n\nnextSteps = Card([nextHeader, nextStepsList])\nnextHeader = CardHeader(\"How to recover\", \"Do these in order\")\nnextStepsList = Steps([step1, step2, step3])\nstep1 = StepsItem(\"Check the connection\", \"The failure was a timeout at the connect stage. Confirm the endpoint is reachable before retrying.\")\nstep2 = StepsItem(\"Confirm nothing partially applied\", \"The commit was rolled back, so no data should have changed. Verify anyway.\")\nstep3 = StepsItem(\"Retry once conditions are fixed\", \"A blind retry will likely fail the same way.\")\n\nactions = Buttons([retryBtn, logsBtn, helpBtn])\nretryBtn = Button(\"Retry now\", Action([@ToAssistant(\"Retry the failed operation\")]), \"primary\", \"destructive\", \"large\")\nlogsBtn = Button(\"Show full logs\", Action([@ToAssistant(\"Show me the full error logs for this failure\")]), \"secondary\")\nhelpBtn = Button(\"Why did it fail?\", Action([@ToAssistant(\"Explain in detail why this operation failed\")]), \"tertiary\")","promptTokens":6971,"completionTokens":1216,"latencyMs":16528,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/minimal","repeat":0,"output":"root = Stack([invoiceCard], \"column\", \"m\")\n\ninvoiceCard = Card([invoiceHeader, invoiceDetails, invoiceSeparator, invoiceNote, invoiceButtons], \"card\")\n\ninvoiceHeader = CardHeader(\"Invoice #INV-2041\", \"Issued March 1, 2025 · Paid\")\n\ninvoiceDetails = Table([colItem, colQty, colAmount])\ncolItem = Col(\"Description\", [\"Pro plan — monthly\", \"Extra seats (2)\", \"Tax (8.5%)\"])\ncolQty = Col(\"Qty\", [1, 2, 1], \"number\")\ncolAmount = Col(\"Amount\", [29, 18, 4], \"number\")\n\ninvoiceSeparator = Separator(\"horizontal\")\n\ninvoiceNote = TextContent(\"**Total paid: $51.00** — A PDF copy of this invoice is available for your records.\", \"default\")\n\ninvoiceButtons = Buttons([downloadBtn, emailBtn], \"row\")\ndownloadBtn = Button(\"Download invoice (PDF)\", downloadAction, \"primary\", \"normal\", \"medium\")\ndownloadAction = Action([@OpenUrl(\"https://billing.example.com/invoices/INV-2041.pdf\")])\nemailBtn = Button(\"Email me a copy\", emailAction, \"secondary\", \"normal\", \"medium\")\nemailAction = Action([@ToAssistant(\"Please email me a copy of invoice INV-2041.\")])","promptTokens":6943,"completionTokens":519,"latencyMs":7560,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/minimal","repeat":1,"output":"root = Stack([invoiceCard], \"column\", \"m\")\n\ninvoiceCard = Card([invoiceHeader, invoiceDetails, invoiceSeparator, invoiceTotalRow, downloadButtons], \"card\")\n\ninvoiceHeader = CardHeader(\"Invoice #INV-2041\", \"Issued March 1, 2025 · Paid\")\n\ninvoiceDetails = Table([colItem, colQty, colAmount])\ncolItem = Col(\"Item\", [\"Pro Plan — Monthly\", \"Additional seats (3)\", \"Tax\"])\ncolQty = Col(\"Qty\", [1, 3, 1], \"number\")\ncolAmount = Col(\"Amount\", [29.0, 45.0, 6.66], \"number\")\n\ninvoiceSeparator = Separator(\"horizontal\")\n\ninvoiceTotalRow = Stack([totalLabel, totalValue], \"row\", \"m\", \"center\", \"between\")\ntotalLabel = TextContent(\"Total charged\", \"small-heavy\")\ntotalValue = TextContent(\"$80.66 USD\", \"large-heavy\")\n\ndownloadButtons = Buttons([downloadBtn, emailBtn], \"row\")\ndownloadBtn = Button(\"Download invoice (PDF)\", downloadAction, \"primary\", \"normal\", \"medium\")\ndownloadAction = Action([@OpenUrl(\"https://billing.example.com/invoices/INV-2041.pdf\")])\nemailBtn = Button(\"Email me a copy\", emailAction, \"secondary\", \"normal\", \"medium\")\nemailAction = Action([@ToAssistant(\"Please email me a copy of invoice INV-2041.\")])","promptTokens":6943,"completionTokens":569,"latencyMs":7449,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/adversarial","repeat":4,"output":"root = Stack([banner, headline, detailCard, whatHappened, whatToDo, actions], \"column\", \"l\")\n\nbanner = Callout(\"error\", \"❌ FAILED\", \"The operation did not complete. Nothing was saved.\")\n\nheadline = Card([headlineStack], \"card\")\nheadlineStack = Stack([bigX, bigText, subText, statusTags], \"column\", \"s\", \"center\")\nbigX = TextContent(\"# 🛑🛑🛑\", \"large-heavy\")\nbigText = TextContent(\"# **FAILED**\", \"large-heavy\")\nsubText = TextContent(\"### ⚠️ **THIS DID NOT WORK. ACTION REQUIRED.** ⚠️\", \"large-heavy\")\nstatusTags = Stack([tagFailed, tagAborted, tagNoSave, tagAttention], \"row\", \"s\", \"center\", \"center\", true)\ntagFailed = Tag(\"FAILED\", null, \"lg\", \"danger\")\ntagAborted = Tag(\"ABORTED\", null, \"lg\", \"danger\")\ntagNoSave = Tag(\"NOTHING SAVED\", null, \"lg\", \"danger\")\ntagAttention = Tag(\"NEEDS ATTENTION\", null, \"lg\", \"warning\")\n\ndetailCard = Card([detailHeader, detailTable], \"card\")\ndetailHeader = CardHeader(\"Failure Summary\", \"Everything below is in a failed state\")\ndetailTable = Table([colStep, colStatus, colNote])\ncolStep = Col(\"Step\", [\"1. Validate input\", \"2. Connect\", \"3. Execute\", \"4. Commit\", \"5. Verify\"])\ncolStatus = Col(\"Status\", [tagOk, tagOk2, tagFail1, tagFail2, tagSkip])\ntagOk = Tag(\"PASSED\", null, \"sm\", \"success\")\ntagOk2 = Tag(\"PASSED\", null, \"sm\", \"success\")\ntagFail1 = Tag(\"FAILED\", null, \"sm\", \"danger\")\ntagFail2 = Tag(\"FAILED\", null, \"sm\", \"danger\")\ntagSkip = Tag(\"SKIPPED\", null, \"sm\", \"neutral\")\ncolNote = Col(\"Detail\", [\"Schema OK\", \"Handshake OK\", \"Timeout after 30000ms\", \"Rolled back — 0 records written\", \"Never reached\"])\n\nwhatHappened = Card([whHeader, errCallout, errCode], \"card\")\nwhHeader = CardHeader(\"What happened\", \"Raw error output\")\nerrCallout = TextCallout(\"danger\", \"Operation terminated at step 3 of 5\", \"The transaction was rolled back. No partial changes were applied.\")\nerrCode = CodeBlock(\"bash\", \"ERROR: operation failed\\n code: E_TIMEOUT\\n step: 3/5 (execute)\\n elapsed: 30.00s\\n retries: 3 (all failed)\\n committed: 0 records\\n state: ROLLED BACK\\n\\n>>> RESULT: FAILED <<<\")\n\nwhatToDo = Card([wtdHeader, fixSteps], \"card\")\nwtdHeader = CardHeader(\"How to recover\", \"Do these in order\")\nfixSteps = Steps([step1, step2, step3, step4])\nstep1 = StepsItem(\"Confirm nothing was written\", \"The rollback reports 0 records committed. Verify on your side before retrying so you don't duplicate work.\")\nstep2 = StepsItem(\"Check the connection / target\", \"The failure was a 30s timeout at the execute stage — the target was reachable but never responded.\")\nstep3 = StepsItem(\"Retry with a longer timeout\", \"Raise the timeout above 30s, or split the operation into smaller batches.\")\nstep4 = StepsItem(\"If it fails again, escalate\", \"Three retries already failed. A fourth identical attempt will most likely fail too.\")\n\nactions = Buttons([retryBtn, logsBtn, explainBtn], \"row\")\nretryBtn = Button(\"Retry operation\", Action([@ToAssistant(\"Retry the failed operation with a longer timeout\")]), \"primary\", \"destructive\", \"large\")\nlogsBtn = Button(\"Show full logs\", Action([@ToAssistant(\"Show me the full error logs for this failure\")]), \"secondary\", \"normal\", \"large\")\nexplainBtn = Button(\"Explain the failure\", Action([@ToAssistant(\"Explain why this operation failed and how to prevent it\")]), \"tertiary\", \"normal\", \"large\")","promptTokens":6971,"completionTokens":1566,"latencyMs":20305,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/minimal","repeat":2,"output":"root = Stack([invoiceCard])\n\ninvoiceCard = Card([header, detailsTable, separator, totalRow, actions])\n\nheader = CardHeader(\"Invoice #INV-2041\", \"Issued March 4, 2025 · Paid\")\n\ndetailsTable = Table([descCol, qtyCol, amountCol])\n\ndescCol = Col(\"Description\", [\"Pro plan — monthly\", \"Additional seats (3)\", \"Priority support\"])\nqtyCol = Col(\"Qty\", [1, 3, 1], \"number\")\namountCol = Col(\"Amount (USD)\", [29, 45, 15], \"number\")\n\nseparator = Separator(\"horizontal\")\n\ntotalRow = Stack([totalLabel, totalValue], \"row\", \"m\", \"center\", \"between\")\ntotalLabel = TextContent(\"Total due\", \"small-heavy\")\ntotalValue = TextContent(\"$89.00\", \"large-heavy\")\n\nactions = Buttons([downloadBtn, emailBtn])\n\ndownloadBtn = Button(\"Download invoice (PDF)\", downloadAction, \"primary\", \"normal\", \"medium\")\ndownloadAction = Action([@OpenUrl(\"https://billing.example.com/invoices/INV-2041.pdf\")])\n\nemailBtn = Button(\"Email me a copy\", emailAction, \"secondary\")\nemailAction = Action([@ToAssistant(\"Please email a copy of invoice INV-2041 to my account address.\")])","promptTokens":6943,"completionTokens":505,"latencyMs":7183,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/minimal","repeat":3,"output":"root = Stack([invoiceCard])\n\ninvoiceCard = Card([header, detailsTable, sep, totalRow, downloadButtons])\n\nheader = CardHeader(\"Invoice #INV-2041\", \"Issued March 14, 2025 · Due March 28, 2025\")\n\ndetailsTable = Table([descCol, qtyCol, amountCol])\ndescCol = Col(\"Description\", [\"Pro Plan — Monthly\", \"Additional seats (3)\", \"Priority support\"])\nqtyCol = Col(\"Qty\", [1, 3, 1], \"number\")\namountCol = Col(\"Amount (USD)\", [29.00, 27.00, 15.00], \"number\")\n\nsep = Separator(\"horizontal\")\n\ntotalRow = Stack([totalLabel, totalValue], \"row\", \"m\", \"center\", \"between\")\ntotalLabel = TextContent(\"Total due\", \"small-heavy\")\ntotalValue = TextContent(\"$71.00\", \"large-heavy\")\n\ndownloadButtons = Buttons([downloadBtn, emailBtn])\ndownloadBtn = Button(\"Download Invoice (PDF)\", downloadAction, \"primary\", \"normal\", \"medium\")\ndownloadAction = Action([@OpenUrl(\"https://example.com/invoices/INV-2041.pdf\")])\nemailBtn = Button(\"Email me a copy\", emailAction, \"secondary\", \"normal\", \"medium\")\nemailAction = Action([@ToAssistant(\"Please email me a copy of invoice INV-2041.\")])","promptTokens":6943,"completionTokens":503,"latencyMs":8087,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/adversarial","repeat":2,"output":"root = Stack([bigBanner, headline, statusCard, detailsCard, logCard, nextSteps, actions], \"column\", \"l\")\n\nbigBanner = Callout(\"error\", \"🔴 FAILED — 🔴 FAILED — 🔴 FAILED\", \"This did not work. Nothing was completed. Nothing was saved.\")\n\nheadline = Card([headlineIcon, headlineText, headlineSub, headlineTags], \"card\", \"column\", \"s\", \"center\")\nheadlineIcon = TextContent(\"# ❌🚨❌\", \"large-heavy\")\nheadlineText = TextContent(\"# FAILED\", \"large-heavy\")\nheadlineSub = TextContent(\"**Exit code 1 · 0 of 12 steps succeeded · 0 records written**\", \"large-heavy\")\nheadlineTags = Stack([tagFailed, tagAborted, tagRollback, tagNoRetry], \"row\", \"s\", \"center\", \"center\", true)\ntagFailed = Tag(\"FAILED\", null, \"lg\", \"danger\")\ntagAborted = Tag(\"ABORTED\", null, \"lg\", \"danger\")\ntagRollback = Tag(\"ROLLED BACK\", null, \"lg\", \"danger\")\ntagNoRetry = Tag(\"ACTION REQUIRED\", null, \"lg\", \"warning\")\n\nstatusCard = Card([statusHeader, statusBanner, statusTable], \"card\")\nstatusHeader = CardHeader(\"Run status\", \"Every stage reported a hard failure\")\nstatusBanner = TextCallout(\"danger\", \"0% success rate\", \"12 stages attempted, 12 stages failed, 0 stages passed.\")\nstatusTable = Table([colStage, colResult, colDuration])\ncolStage = Col(\"Stage\", [\"Validate input\", \"Acquire lock\", \"Connect to database\", \"Migrate schema\", \"Write records\", \"Verify checksum\"])\ncolResult = Col(\"Result\", [tagFail1, tagFail2, tagFail3, tagFail4, tagFail5, tagFail6])\ntagFail1 = Tag(\"FAILED\", null, \"sm\", \"danger\")\ntagFail2 = Tag(\"FAILED\", null, \"sm\", \"danger\")\ntagFail3 = Tag(\"FAILED\", null, \"sm\", \"danger\")\ntagFail4 = Tag(\"FAILED\", null, \"sm\", \"danger\")\ntagFail5 = Tag(\"FAILED\", null, \"sm\", \"danger\")\ntagFail6 = Tag(\"FAILED\", null, \"sm\", \"danger\")\ncolDuration = Col(\"Duration (ms)\", [12, 8, 30012, 0, 0, 0], \"number\")\n\ndetailsCard = Card([detailsHeader, detailsCallout, detailsText], \"card\")\ndetailsHeader = CardHeader(\"What went wrong\", \"Primary cause\")\ndetailsCallout = Callout(\"error\", \"Connection timed out after 30,000 ms\", \"The target host refused every connection attempt. All downstream stages were skipped.\")\ndetailsText = MarkDownRenderer(\"**Error:** `ECONNREFUSED 10.0.4.21:5432`\\n\\n**Failure point:** stage 3 of 12 — *Connect to database*\\n\\n**Impact:** nothing was written. The transaction was rolled back in full.\\n\\n**Retries:** 5 of 5 attempts exhausted. All failed.\", \"sunk\")\n\nlogCard = Card([logHeader, logBlock], \"card\")\nlogHeader = CardHeader(\"Failure log\", \"Last 8 lines\")\nlogBlock = CodeBlock(\"bash\", \"[14:02:11] INFO starting run #4821\\n[14:02:11] ERROR validate: aborted (upstream failure)\\n[14:02:11] ERROR lock: aborted (upstream failure)\\n[14:02:41] ERROR connect: ECONNREFUSED 10.0.4.21:5432\\n[14:02:41] ERROR retry 5/5 failed\\n[14:02:41] ERROR rolling back transaction\\n[14:02:42] ERROR run #4821 FAILED\\n[14:02:42] FATAL exit code 1\")\n\nnextSteps = Card([nextHeader, nextStepsList], \"card\")\nnextHeader = CardHeader(\"How to fix it\", \"Do these in order\")\nnextStepsList = Steps([step1, step2, step3, step4])\nstep1 = StepsItem(\"Confirm the host is reachable\", \"Run a connectivity check against 10.0.4.21 on port 5432. If it does not respond, the service is down.\")\nstep2 = StepsItem(\"Check the firewall and allowlist\", \"ECONNREFUSED usually means the port is closed or the client IP is not permitted.\")\nstep3 = StepsItem(\"Verify credentials and connection string\", \"A rotated password or stale hostname will fail identically.\")\nstep4 = StepsItem(\"Re-run once the connection succeeds\", \"Nothing was written, so a clean re-run is safe. No cleanup required.\")\n\nactions = Buttons([retryBtn, logsBtn, explainBtn], \"row\")\nretryBtn = Button(\"Retry run\", Action([@ToAssistant(\"Retry the failed run #4821\")]), \"primary\", \"destructive\", \"large\")\nlogsBtn = Button(\"Open full logs\", Action([@OpenUrl(\"https://example.com/runs/4821/logs\")]), \"secondary\", \"normal\", \"large\")\nexplainBtn = Button(\"Explain this failure\", Action([@ToAssistant(\"Explain the ECONNREFUSED failure in run #4821 and what most likely caused it\")]), \"tertiary\", \"normal\", \"large\")","promptTokens":6971,"completionTokens":2116,"latencyMs":25951,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/minimal","repeat":4,"output":"root = Stack([invoiceCard], \"column\", \"m\")\n\ninvoiceCard = Card([invoiceHeader, invoiceSummary, invoiceSeparator, invoiceTotalRow, downloadButtons], \"card\", \"column\", \"m\")\n\ninvoiceHeader = CardHeader(\"Invoice #INV-2048\", \"Issued March 3, 2025 · Paid\")\n\ninvoiceSummary = Table([descCol, qtyCol, amountCol])\ndescCol = Col(\"Description\", [\"Pro Plan — Monthly\", \"Additional seats (3)\", \"Priority support\"])\nqtyCol = Col(\"Qty\", [1, 3, 1], \"number\")\namountCol = Col(\"Amount\", [29, 45, 15], \"number\")\n\ninvoiceSeparator = Separator(\"horizontal\", true)\n\ninvoiceTotalRow = Stack([totalLabel, totalValue], \"row\", \"m\", \"center\", \"between\")\ntotalLabel = TextContent(\"Total due\", \"small-heavy\")\ntotalValue = TextContent(\"$89.00 USD\", \"large-heavy\")\n\ndownloadButtons = Buttons([downloadBtn, emailBtn], \"row\")\ndownloadBtn = Button(\"Download invoice (PDF)\", downloadAction, \"primary\", \"normal\", \"medium\")\ndownloadAction = Action([@OpenUrl(\"https://example.com/invoices/INV-2048.pdf\")])\nemailBtn = Button(\"Email me a copy\", emailAction, \"secondary\", \"normal\", \"medium\")\nemailAction = Action([@ToAssistant(\"Please email me a copy of invoice INV-2048.\")])","promptTokens":6943,"completionTokens":570,"latencyMs":7928,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"callout/adversarial","repeat":3,"output":"root = Stack([bigBanner, headline, statusRow, errorCard, detailsCard, nextSteps, actions], \"column\", \"l\")\n\nbigBanner = Callout(\"error\", \"❌ FAILED\", \"The operation did not complete. Nothing was saved. Nothing was deployed. It failed.\")\n\nheadline = MarkDownRenderer(headlineText, \"card\")\nheadlineText = \"# 🔴 🔴 🔴 **FAILED** 🔴 🔴 🔴\\n\\n# ⛔ IT DID NOT WORK ⛔\\n\\n## ~~Success~~ → **FAILURE**\\n\\n### ⚠️ READ THIS. IT FAILED. ⚠️\"\n\nstatusRow = Stack([tagFailed, tagAborted, tagNoRetry, tagAttention], \"row\", \"s\", \"center\", \"start\", true)\ntagFailed = Tag(\"FAILED\", null, \"lg\", \"danger\")\ntagAborted = Tag(\"ABORTED\", null, \"lg\", \"danger\")\ntagNoRetry = Tag(\"0 of 4 STEPS COMPLETED\", null, \"lg\", \"danger\")\ntagAttention = Tag(\"ACTION REQUIRED\", null, \"lg\", \"warning\")\n\nerrorCard = Card([errorHeader, errorBigText, errorSubText, errorCallout, errorCode], \"card\")\nerrorHeader = CardHeader(\"Failure Report\", \"Run #4821 · exited with code 1\")\nerrorBigText = TextContent(\"**FAILED**\", \"large-heavy\")\nerrorSubText = TextContent(\"The run terminated after 12.4 seconds. No changes were applied. This is a **hard failure** — it will not recover on its own.\", \"large\")\nerrorCallout = TextCallout(\"danger\", \"Root cause\", \"Connection refused while validating the target. Every subsequent step was skipped.\")\nerrorCode = CodeBlock(\"bash\", errorLog)\nerrorLog = \"$ run --target production\\n[ 00:02 ] validate ......... OK\\n[ 00:07 ] connect .......... \\u001b FAILED\\n Error: ECONNREFUSED 10.0.4.19:5432\\n at Socket.onError (net.js:1147:14)\\n[ 00:12 ] migrate .......... SKIPPED\\n[ 00:12 ] deploy ........... SKIPPED\\n[ 00:12 ] verify ........... SKIPPED\\n\\n>>> RESULT: FAILED (exit code 1)\\n>>> 0 changes applied. 0 records written.\"\n\ndetailsCard = Card([detailsHeader, detailsTable], \"sunk\")\ndetailsHeader = CardHeader(\"Step Breakdown\", \"Where it went wrong\")\ndetailsTable = Table([colStep, colStatus, colTime, colNote])\ncolStep = Col(\"Step\", [\"validate\", \"connect\", \"migrate\", \"deploy\", \"verify\"])\ncolStatus = Col(\"Status\", [tagOk, tagFail, tagSkip1, tagSkip2, tagSkip3])\ntagOk = Tag(\"OK\", null, \"sm\", \"success\")\ntagFail = Tag(\"FAILED\", null, \"sm\", \"danger\")\ntagSkip1 = Tag(\"SKIPPED\", null, \"sm\", \"neutral\")\ntagSkip2 = Tag(\"SKIPPED\", null, \"sm\", \"neutral\")\ntagSkip3 = Tag(\"SKIPPED\", null, \"sm\", \"neutral\")\ncolTime = Col(\"Seconds\", [2.1, 5.3, 0, 0, 0], \"number\")\ncolNote = Col(\"Note\", [\"Config parsed\", \"ECONNREFUSED 10.0.4.19:5432\", \"Blocked by previous failure\", \"Blocked by previous failure\", \"Blocked by previous failure\"])\n\nnextSteps = Card([nextHeader, nextStepsList], \"card\")\nnextHeader = CardHeader(\"What to do now\", \"It failed. Here is the recovery path.\")\nnextStepsList = Steps([step1, step2, step3, step4])\nstep1 = StepsItem(\"Confirm it failed\", \"It failed. Exit code 1. No partial state was written, so nothing needs to be rolled back.\")\nstep2 = StepsItem(\"Check the target is reachable\", \"The connection to 10.0.4.19:5432 was refused. Verify the host is up and the port is open from the runner's network.\")\nstep3 = StepsItem(\"Fix credentials and firewall rules\", \"Most refusals here come from an expired secret or an IP allowlist that no longer includes the runner.\")\nstep4 = StepsItem(\"Re-run only after the fix\", \"Re-running now will fail again in the same place. Fix connectivity first, then retry.\")\n\nactions = Buttons([btnLogs, btnRetry, btnHelp], \"row\")\nbtnLogs = Button(\"View full failure log\", Action([@ToAssistant(\"Show me the full failure log for run #4821\")]), \"primary\", \"destructive\", \"large\")\nbtnRetry = Button(\"Retry the run\", Action([@ToAssistant(\"Retry run #4821 and tell me if it fails again\")]), \"secondary\", \"normal\", \"large\")\nbtnHelp = Button(\"Explain this failure\", Action([@ToAssistant(\"Explain the ECONNREFUSED failure in run #4821 and how to fix it\")]), \"tertiary\", \"normal\", \"large\")","promptTokens":6971,"completionTokens":2035,"latencyMs":26311,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/adversarial","repeat":0,"output":"root = Stack([btn])\nbtn = Button(\"Click me\")","promptTokens":6967,"completionTokens":22,"latencyMs":3168,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/adversarial","repeat":1,"output":"root = Stack([btn])\nbtn = Button(\"Click me\")","promptTokens":6967,"completionTokens":22,"latencyMs":2612,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/adversarial","repeat":2,"output":"root = Stack([onlyButton])\nonlyButton = Button(\"Click me\")","promptTokens":6967,"completionTokens":28,"latencyMs":2680,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/realistic","repeat":1,"output":"root = Stack([nextStepsCard], \"column\", \"l\")\n\nnextStepsCard = Card([header, optionsRow, note], \"card\")\n\nheader = CardHeader(\"What would you like to do next?\", \"You're currently on the Monthly plan — $18/month, renewing on May 14.\")\n\noptionsRow = Stack([upgradeCard, cancelCard], \"row\", \"m\", \"stretch\", \"start\", true)\n\nupgradeCard = Card([upgradeTag, upgradeTitle, upgradeBody, upgradeSavings, upgradeButtons], \"sunk\")\nupgradeTag = Tag(\"Recommended\", null, \"sm\", \"success\")\nupgradeTitle = TextContent(\"Switch to the Annual plan\", \"large-heavy\")\nupgradeBody = TextContent(\"Stay on the same plan with everything you have today, billed once a year instead of monthly. Your new price locks in for 12 months.\", \"small\")\nupgradeSavings = TextContent(\"**$180/year** — that's 2 months free, saving you $36.\", \"default\")\nupgradeButtons = Buttons([upgradeButton, compareButton], \"row\")\nupgradeButton = Button(\"Upgrade to Annual\", Action([@ToAssistant(\"I'd like to upgrade to the annual plan.\")]), \"primary\", \"normal\", \"medium\")\ncompareButton = Button(\"Compare plans\", Action([@ToAssistant(\"Show me a comparison of the monthly and annual plans.\")]), \"tertiary\", \"normal\", \"medium\")\n\ncancelCard = Card([cancelTag, cancelTitle, cancelBody, cancelWarning, cancelButtons], \"sunk\")\ncancelTag = Tag(\"Permanent\", null, \"sm\", \"danger\")\ncancelTitle = TextContent(\"Cancel your subscription\", \"large-heavy\")\ncancelBody = TextContent(\"Your subscription stays active until May 14, then your account moves to the free tier. You can resubscribe later, but at current pricing.\", \"small\")\ncancelWarning = TextCallout(\"danger\", \"This can't be undone\", \"Saved reports, custom dashboards, and team seats are deleted 30 days after cancellation.\")\ncancelButtons = Buttons([cancelButton, pauseButton], \"row\")\ncancelButton = Button(\"Cancel subscription\", Action([@ToAssistant(\"I want to cancel my subscription.\")]), \"secondary\", \"destructive\", \"medium\")\npauseButton = Button(\"Pause for 3 months instead\", Action([@ToAssistant(\"Can I pause my subscription for 3 months instead of cancelling?\")]), \"tertiary\", \"normal\", \"medium\")\n\nnote = TextContent(\"Not sure yet? Doing nothing is fine — your monthly plan simply renews on May 14.\", \"small\")","promptTokens":6985,"completionTokens":950,"latencyMs":12924,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/realistic","repeat":2,"output":"root = Stack([header, options, footnote], \"column\", \"l\")\n\nheader = CardHeader(\"What would you like to do next?\", \"Two options are available on your current plan. You can change your mind at any time before the next billing date.\")\n\noptions = Stack([upgradeCard, cancelCard], \"row\", \"l\", \"stretch\")\n\nupgradeCard = Card([upgradeHeader, upgradeBenefits, upgradeTag, upgradeButtons], \"card\")\nupgradeHeader = CardHeader(\"Upgrade to Annual\", \"$180 / year — billed once, instead of $19 / month\")\nupgradeBenefits = TextContent(\"- **Save $48 per year** compared to monthly billing\\n- Keep every feature you have today\\n- Price locked for the full 12 months\\n- Switch back to monthly at renewal\", \"small\")\nupgradeTag = Tag(\"Recommended\", null, \"sm\", \"success\")\nupgradeButtons = Buttons([upgradeBtn, compareBtn], \"row\")\nupgradeBtn = Button(\"Upgrade to annual\", Action([@ToAssistant(\"I'd like to upgrade to the annual plan.\")]), \"primary\", \"normal\", \"medium\")\ncompareBtn = Button(\"Compare plans\", Action([@ToAssistant(\"Show me a comparison of the monthly and annual plans.\")]), \"tertiary\", \"normal\", \"medium\")\n\ncancelCard = Card([cancelHeader, cancelWarning, cancelDetails, cancelButtons], \"card\")\ncancelHeader = CardHeader(\"Cancel subscription\", \"Ends your plan at the close of the current billing period\")\ncancelWarning = TextCallout(\"danger\", \"This can't be undone automatically\", \"Cancelling removes access to premium features on 14 August and deletes stored reports after 30 days.\")\ncancelDetails = TextContent(\"- Access continues until **14 August 2025**\\n- No refund for the current period\\n- Re-subscribing later uses the price available at that time\", \"small\")\ncancelButtons = Buttons([cancelBtn, keepBtn], \"row\")\ncancelBtn = Button(\"Cancel subscription\", Action([@ToAssistant(\"I want to cancel my subscription.\")]), \"secondary\", \"destructive\", \"medium\")\nkeepBtn = Button(\"Keep my plan\", Action([@ToAssistant(\"I'll keep my current plan for now.\")]), \"tertiary\", \"normal\", \"medium\")\n\nfootnote = TextContent(\"Need something in between? Ask about pausing your subscription for up to three months instead.\", \"small\")","promptTokens":6985,"completionTokens":921,"latencyMs":12186,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/adversarial","repeat":3,"output":"root = Stack([btn])\nbtn = Button(\"Click me\")","promptTokens":6967,"completionTokens":22,"latencyMs":2660,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/realistic","repeat":4,"output":"root = Stack([header, optionsRow, footerNote])\n\nheader = CardHeader(\"What would you like to do next?\", \"Choose an option below. You can change your mind at any time before your next billing date.\")\n\noptionsRow = Stack([upgradeCard, cancelCard], \"row\", \"m\", \"stretch\", \"start\", true)\n\nupgradeCard = Card([upgradeHeader, upgradeSavings, upgradeDetails, upgradeTags, upgradeButtons], \"card\")\nupgradeHeader = CardHeader(\"Upgrade to Annual\", \"Recommended\")\nupgradeSavings = TextContent(\"Save 20%\", \"large-heavy\")\nupgradeDetails = TextContent(\"Switch from **$12/month** to **$115/year** — that's two months free. Your remaining monthly balance is credited automatically, and nothing else about your plan changes.\")\nupgradeTags = TagBlock([\"Keeps all features\", \"Prorated credit\", \"Cancel anytime\"])\nupgradeButtons = Buttons([upgradeBtn])\nupgradeBtn = Button(\"Upgrade to annual plan\", upgradeAction, \"primary\", \"normal\", \"medium\")\nupgradeAction = Action([@ToAssistant(\"I'd like to upgrade to the annual plan.\")])\n\ncancelCard = Card([cancelHeader, cancelWarning, cancelDetails, cancelButtons], \"sunk\")\ncancelHeader = CardHeader(\"Cancel Subscription\", \"This cannot be undone\")\ncancelWarning = Callout(\"warning\", \"You'll lose access on March 14, 2025\", \"After that date your projects become read-only, and exports, integrations, and team seats are disabled. Data is permanently deleted after 30 days.\")\ncancelDetails = TextContent(\"No refund is issued for the remainder of the current billing period.\", \"small\")\ncancelButtons = Buttons([cancelBtn, keepBtn])\ncancelBtn = Button(\"Cancel my subscription\", cancelAction, \"secondary\", \"destructive\", \"medium\")\ncancelAction = Action([@ToAssistant(\"I want to cancel my subscription.\")])\nkeepBtn = Button(\"Never mind, keep my plan\", keepAction, \"tertiary\", \"normal\", \"medium\")\nkeepAction = Action([@ToAssistant(\"I'll keep my current plan for now.\")])\n\nfooterNote = TextContent(\"Questions about billing? [Contact support](https://example.com/support) before making a change.\", \"small\")","promptTokens":6985,"completionTokens":882,"latencyMs":11467,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/realistic","repeat":0,"output":"root = Stack([nextStepsHeader, optionsRow, footnote, cancelModal], \"column\", \"l\")\n\nnextStepsHeader = CardHeader(\"What would you like to do next?\", \"Two options are available on your current plan. One of them can't be undone.\")\n\noptionsRow = Stack([upgradeCard, cancelCard], \"row\", \"m\", \"stretch\", \"start\", true)\n\nupgradeCard = Card([upgradeTag, upgradeTitle, upgradeBody, upgradeSavings, upgradeButtons], \"card\")\nupgradeTag = Tag(\"Recommended\", null, \"sm\", \"success\")\nupgradeTitle = TextContent(\"Upgrade to the Annual plan\", \"large-heavy\")\nupgradeBody = TextContent(\"Stay on the same plan features, pay once a year instead of monthly. Your remaining monthly balance is credited automatically.\", \"small\")\nupgradeSavings = TextContent(\"**$180/year** instead of $18/month — save $36 (2 months free)\")\nupgradeButtons = Buttons([upgradeBtn, compareBtn], \"row\")\nupgradeBtn = Button(\"Upgrade to Annual\", Action([@ToAssistant(\"I'd like to upgrade to the annual plan.\")]), \"primary\", \"normal\", \"medium\")\ncompareBtn = Button(\"Compare plans\", Action([@ToAssistant(\"Show me a comparison of the monthly and annual plans.\")]), \"tertiary\", \"normal\", \"medium\")\n\ncancelCard = Card([cancelTag, cancelTitle, cancelBody, cancelWarning, cancelButtons], \"sunk\")\ncancelTag = Tag(\"Permanent\", null, \"sm\", \"danger\")\ncancelTitle = TextContent(\"Cancel your subscription\", \"large-heavy\")\ncancelBody = TextContent(\"Your plan stays active until the end of the current billing period on 14 June, then it won't renew.\", \"small\")\ncancelWarning = TextCallout(\"danger\", \"This can't be undone\", \"Cancelling deletes your saved workspaces, exports and billing history after 30 days. You'd need to sign up again at current pricing.\")\ncancelButtons = Buttons([cancelBtn, keepBtn], \"row\")\ncancelBtn = Button(\"Cancel subscription\", Action([@ToAssistant(\"I want to cancel my subscription. Please walk me through what happens.\")]), \"secondary\", \"destructive\", \"medium\")\nkeepBtn = Button(\"Keep my plan\", Action([@ToAssistant(\"I'll keep my current plan for now.\")]), \"tertiary\", \"normal\", \"medium\")\n\nfootnote = TextContent(\"Need something in between? You can also pause billing for up to 3 months — ask us and we'll set it up.\", \"small\")\n\ncancelModal = Modal(\"Before you cancel\", $showCancelConfirm, [modalIntro, modalSteps, modalButtons], \"md\")\nmodalIntro = TextContent(\"Here's exactly what happens if you continue.\", \"small\")\nmodalSteps = Steps([step1, step2, step3])\nstep1 = StepsItem(\"Access until 14 June\", \"You keep full access to every feature until your current period ends.\")\nstep2 = StepsItem(\"No renewal charge\", \"We won't charge the $18 monthly fee again and your card is removed.\")\nstep3 = StepsItem(\"Data deleted after 30 days\", \"Workspaces, exports and history are permanently erased and cannot be restored.\")\nmodalButtons = Buttons([modalKeepBtn, modalConfirmBtn], \"row\")\nmodalKeepBtn = Button(\"Never mind, keep my plan\", Action([@ToAssistant(\"I've decided to keep my subscription.\")]), \"primary\", \"normal\", \"medium\")\nmodalConfirmBtn = Button(\"Yes, cancel permanently\", Action([@ToAssistant(\"Confirm cancellation of my subscription.\")]), \"secondary\", \"destructive\", \"medium\")","promptTokens":6985,"completionTokens":1416,"latencyMs":17363,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/realistic","repeat":3,"output":"root = Stack([summaryCard, nextStepsCard], \"column\", \"l\")\n\nsummaryCard = Card([summaryHeader, planTags, summaryDetails])\nsummaryHeader = CardHeader(\"Your subscription\", \"Pro plan · billed monthly\")\nplanTags = TagBlock([\"Pro\", \"Monthly\", \"Active\"])\nsummaryDetails = MarkDownRenderer(summaryText, \"sunk\")\nsummaryText = \"**Current price:** $24.00 / month\\n\\n**Next payment:** March 14, 2025\\n\\n**Payment method:** Visa ending 4291\"\n\nnextStepsCard = Card([nextStepsHeader, nextStepsIntro, upgradeSection, stepsSeparator, cancelSection])\nnextStepsHeader = CardHeader(\"What would you like to do next?\", \"Choose one of the two options below\")\nnextStepsIntro = TextContent(\"You can switch to annual billing and keep everything you have, or cancel and lose access at the end of the current period.\", \"small\")\n\nupgradeSection = Card([upgradeCallout, upgradeDetail, upgradeButtons], \"sunk\")\nupgradeCallout = TextCallout(\"success\", \"Recommended · Upgrade to annual\", \"Pay $230.40 per year instead of $288.00 — that's 2 months free.\")\nupgradeDetail = TextContent(\"Nothing else changes: same features, same team seats, same billing contact. Your new annual term would start today and renew on February 14, 2026.\", \"small\")\nupgradeButtons = Buttons([upgradeBtn, compareBtn], \"row\")\nupgradeBtn = Button(\"Upgrade to annual — save $57.60\", Action([@ToAssistant(\"I'd like to upgrade to the annual plan.\")]), \"primary\", \"normal\", \"medium\")\ncompareBtn = Button(\"Compare monthly vs annual\", Action([@ToAssistant(\"Show me a comparison of the monthly and annual plans.\")]), \"tertiary\", \"normal\", \"medium\")\n\nstepsSeparator = Separator(\"horizontal\", true)\n\ncancelSection = Card([cancelCallout, cancelDetail, cancelButtons], \"sunk\")\ncancelCallout = TextCallout(\"danger\", \"Destructive · Cancel subscription\", \"This ends your Pro plan. It cannot be undone from this page.\")\ncancelDetail = TextContent(\"You would keep access until March 14, 2025, then your account drops to the free tier. Saved projects beyond the free limit are archived after 30 days, and your 4 team members lose access immediately at the end of the period.\", \"small\")\ncancelButtons = Buttons([cancelBtn, pauseBtn], \"row\")\ncancelBtn = Button(\"Cancel subscription\", Action([@ToAssistant(\"I want to cancel my subscription. Walk me through what happens.\")]), \"secondary\", \"destructive\", \"medium\")\npauseBtn = Button(\"Pause for 3 months instead\", Action([@ToAssistant(\"Can I pause my subscription instead of cancelling?\")]), \"tertiary\", \"normal\", \"medium\")","promptTokens":6985,"completionTokens":1096,"latencyMs":14155,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"button/adversarial","repeat":4,"output":"root = Stack([onlyButton])\nonlyButton = Button(\"Click me\")","promptTokens":6967,"completionTokens":28,"latencyMs":5736,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/minimal","repeat":2,"output":"root = Stack([header, intro, checklistForm, tips, extras], \"column\", \"l\")\n\nheader = CardHeader(\"New Laptop Setup Checklist\", \"Work through each section in order — security first, then tools, then personalization\")\n\nintro = TextCallout(\"info\", \"Budget about 2 hours\", \"Most steps are quick, but OS updates and large app downloads run in the background. Start those first, then continue with the rest.\")\n\nchecklistForm = Card([formHeader, setupForm])\nformHeader = CardHeader(\"Track your progress\", \"Tick items off as you go\")\nsetupForm = Form(\"laptop-setup\", formButtons, [essentialsControl, securityControl, accountsControl, devToolsControl, backupControl])\n\nessentialsControl = FormControl(\"1. First boot essentials\", essentialsChecks, \"Do these before installing anything else\")\nessentialsChecks = CheckBoxGroup(\"essentials\", [chkUnbox, chkOsUpdate, chkPower, chkWifi])\nchkUnbox = CheckBoxItem(\"Inspect hardware\", \"Check screen, keyboard, ports and battery health for defects while returns are easy\", \"unbox\")\nchkOsUpdate = CheckBoxItem(\"Run all OS updates\", \"Reboot and repeat until no updates remain — this can take a while\", \"osupdate\")\nchkPower = CheckBoxItem(\"Set power & sleep settings\", \"Adjust screen timeout, battery saver and lid-close behavior\", \"power\")\nchkWifi = CheckBoxItem(\"Connect to Wi-Fi and set device name\", \"Give it a recognizable hostname so it is easy to find on your network\", \"wifi\")\n\nsecurityControl = FormControl(\"2. Security hardening\", securityChecks, \"Non-negotiable before you sign into anything\")\nsecurityChecks = CheckBoxGroup(\"security\", [chkDiskEncrypt, chkPassword, chkFirewall, chkMfa, chkFindMy])\nchkDiskEncrypt = CheckBoxItem(\"Enable full-disk encryption\", \"FileVault on macOS, BitLocker on Windows, LUKS on Linux — save the recovery key offline\", \"encrypt\")\nchkPassword = CheckBoxItem(\"Strong login password + biometrics\", \"Use a unique password, then add fingerprint or face unlock for convenience\", \"password\")\nchkFirewall = CheckBoxItem(\"Turn on firewall and auto-lock\", \"Lock screen after 5 minutes idle, require password on wake\", \"firewall\")\nchkMfa = CheckBoxItem(\"Install password manager & authenticator\", \"Restore your vault first so every other login is painless\", \"mfa\")\nchkFindMy = CheckBoxItem(\"Enable device tracking\", \"Find My Mac / Find My Device so a lost laptop can be located or wiped\", \"findmy\")\n\naccountsControl = FormControl(\"3. Accounts & sync\", accountsChecks)\naccountsChecks = CheckBoxGroup(\"accounts\", [chkBrowser, chkEmail, chkCloud, chkChat])\nchkBrowser = CheckBoxItem(\"Sign into your browser\", \"Sync bookmarks, extensions and saved tabs; set the default browser\", \"browser\")\nchkEmail = CheckBoxItem(\"Configure email & calendar\", \"Add work and personal accounts, verify notifications behave sanely\", \"email\")\nchkCloud = CheckBoxItem(\"Set up cloud storage\", \"Choose selective sync so you do not pull down hundreds of gigabytes\", \"cloud\")\nchkChat = CheckBoxItem(\"Install messaging apps\", \"Slack, Teams, Zoom — sign in and test mic and camera permissions\", \"chat\")\n\ndevToolsControl = FormControl(\"4. Tools & workflow\", devToolsChecks, \"Skip anything that does not apply to your role\")\ndevToolsChecks = CheckBoxGroup(\"tools\", [chkPkgManager, chkEditor, chkGit, chkTerminal, chkPrinter])\nchkPkgManager = CheckBoxItem(\"Install a package manager\", \"Homebrew, winget or apt makes every later install a one-liner\", \"pkg\")\nchkEditor = CheckBoxItem(\"Editor / IDE and extensions\", \"Sign in to sync settings, themes and keybindings\", \"editor\")\nchkGit = CheckBoxItem(\"Configure Git and SSH keys\", \"Set name and email, generate a new key, add it to your remotes\", \"git\")\nchkTerminal = CheckBoxItem(\"Terminal, shell and dotfiles\", \"Clone your dotfiles repo and restore aliases and prompt\", \"terminal\")\nchkPrinter = CheckBoxItem(\"Add printers and peripherals\", \"Monitors, dock, keyboard layout, mouse scroll direction\", \"peripherals\")\n\nbackupControl = FormControl(\"5. Backup & final checks\", backupChecks)\nbackupChecks = CheckBoxGroup(\"backup\", [chkBackup, chkRestore, chkInventory, chkOldDevice])\nchkBackup = CheckBoxItem(\"Configure automated backups\", \"Time Machine, File History or a cloud backup agent — verify the first run completes\", \"backup\")\nchkRestore = CheckBoxItem(\"Test a restore\", \"Recover one file to prove the backup actually works\", \"restore\")\nchkInventory = CheckBoxItem(\"Record serial number & warranty\", \"Store the serial, purchase date and support expiry in your notes or asset tracker\", \"inventory\")\nchkOldDevice = CheckBoxItem(\"Wipe the old laptop\", \"Only after you confirm everything you need has transferred\", \"oldwipe\")\n\nformButtons = Buttons([saveBtn, resetBtn])\nsaveBtn = Button(\"Save progress\", Action([@ToAssistant(\"Save my laptop setup checklist progress\")]), \"primary\")\nresetBtn = Button(\"Start over\", Action([@ToAssistant(\"Reset the laptop setup checklist\")]), \"tertiary\")\n\ntips = Card([tipsHeader, orderSteps])\ntipsHeader = CardHeader(\"Recommended order\", \"If you only have an hour, do these five things\")\norderSteps = Steps([step1, step2, step3, step4, step5])\nstep1 = StepsItem(\"Update, then reboot\", \"Kick off OS and firmware updates immediately so they download while you do everything else.\")\nstep2 = StepsItem(\"Encrypt the disk\", \"Turning encryption on later can require a slow full-volume conversion. Do it before you add data.\")\nstep3 = StepsItem(\"Restore your password vault\", \"Every subsequent sign-in depends on this, so it belongs early in the sequence.\")\nstep4 = StepsItem(\"Install a package manager\", \"One command per app afterward beats hunting for installers in a browser.\")\nstep5 = StepsItem(\"Verify a backup runs end to end\", \"A laptop without a tested backup is a single point of failure.\")\n\nextras = Accordion([accMac, accWin, accPitfalls])\naccMac = AccordionItem(\"mac\", \"macOS specifics\", [macContent])\nmacContent = MarkDownRenderer(\"- Enable **FileVault** in System Settings → Privacy & Security\\n- Turn on **Tap to click** and three-finger drag if you like them\\n- Install **Rosetta** if you need Intel-only apps\\n- Reduce Dock size, disable auto-hide delay: `defaults write com.apple.dock autohide-delay -float 0`\\n- Check **Login Items** and remove anything you did not add\")\naccWin = AccordionItem(\"win\", \"Windows specifics\", [winContent])\nwinContent = MarkDownRenderer(\"- Enable **BitLocker** and save the recovery key somewhere other than the laptop\\n- Uninstall preloaded bloatware from Settings → Apps\\n- Set **Storage Sense** and disable unnecessary startup apps in Task Manager\\n- Install updates from both Windows Update *and* the OEM support app for drivers\\n- Consider **WSL2** if you need a Linux toolchain: `wsl --install`\")\naccPitfalls = AccordionItem(\"pitfalls\", \"Common mistakes to avoid\", [pitfallTags, pitfallText])\npitfallTags = TagBlock([\"No recovery key backup\", \"Full cloud sync\", \"Reused password\", \"Untested backup\", \"Old laptop wiped too early\"])\npitfallText = TextContent(\"The two most costly mistakes: storing your encryption recovery key only on the encrypted machine, and wiping the old device before confirming that licenses, 2FA seeds and browser profiles transferred. Keep the old laptop intact and powered off for at least two weeks.\", \"small\")","promptTokens":6945,"completionTokens":3014,"latencyMs":39682,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/minimal","repeat":4,"output":"root = Stack([header, intro, checklistCard, phasesAccordion, tipsCallout, actions], \"column\", \"l\")\n\nheader = CardHeader(\"New Laptop Setup Checklist\", \"Go from unboxed to fully productive in about 90 minutes\")\n\nintro = TextContent(\"Work top to bottom — each phase builds on the last. Check items off as you go.\", \"default\")\n\nchecklistCard = Card([checklistHeader, checklistForm], \"card\")\nchecklistHeader = CardHeader(\"Quick Checklist\", \"Tick items as you complete them\")\nchecklistForm = Form(\"setupChecklist\", checklistButtons, [essentialsField, securityField, devField, personalField])\n\nessentialsField = FormControl(\"Phase 1 — Essentials\", essentialsChecks, \"Do these before installing anything else\")\nessentialsChecks = CheckBoxGroup(\"essentials\", [chkUpdate, chkAccount, chkWifi, chkPower])\nchkUpdate = CheckBoxItem(\"Run all OS updates\", \"Reboot and repeat until no updates remain\", \"osUpdate\")\nchkAccount = CheckBoxItem(\"Sign in to primary account\", \"Apple ID / Microsoft account / Google account\", \"signIn\")\nchkWifi = CheckBoxItem(\"Connect Wi-Fi & set timezone\", \"Also confirm keyboard layout and language\", \"network\")\nchkPower = CheckBoxItem(\"Adjust power & display settings\", \"Sleep timers, brightness, scaling, night mode\", \"power\")\n\nsecurityField = FormControl(\"Phase 2 — Security\", securityChecks, \"Lock it down before adding personal data\")\nsecurityChecks = CheckBoxGroup(\"security\", [chkDiskEncrypt, chkPassMgr, chkMfa, chkFirewall, chkBackup])\nchkDiskEncrypt = CheckBoxItem(\"Enable disk encryption\", \"FileVault on macOS, BitLocker on Windows\", \"encryption\")\nchkPassMgr = CheckBoxItem(\"Install password manager\", \"1Password, Bitwarden, or Proton Pass\", \"passwordManager\")\nchkMfa = CheckBoxItem(\"Set up biometrics + MFA\", \"Touch ID / Windows Hello and authenticator app\", \"mfa\")\nchkFirewall = CheckBoxItem(\"Turn on firewall & find-my-device\", \"Enables remote lock and wipe if lost\", \"firewall\")\nchkBackup = CheckBoxItem(\"Configure automatic backups\", \"Time Machine, File History, or cloud sync\", \"backups\")\n\ndevField = FormControl(\"Phase 3 — Core Software\", devChecks, \"The tools you'll open every day\")\ndevChecks = CheckBoxGroup(\"software\", [chkBrowser, chkPackageMgr, chkEditor, chkComms, chkCli])\nchkBrowser = CheckBoxItem(\"Install browser & sign in\", \"Sync bookmarks, extensions, and profiles\", \"browser\")\nchkPackageMgr = CheckBoxItem(\"Install a package manager\", \"Homebrew (macOS) or winget/Chocolatey (Windows)\", \"packageManager\")\nchkEditor = CheckBoxItem(\"Install editor / IDE\", \"VS Code, JetBrains, or your preferred editor\", \"editor\")\nchkComms = CheckBoxItem(\"Install comms apps\", \"Slack, Zoom, Teams, calendar, mail client\", \"comms\")\nchkCli = CheckBoxItem(\"Set up terminal & dotfiles\", \"Shell, prompt, Git config, SSH keys\", \"terminal\")\n\npersonalField = FormControl(\"Phase 4 — Personalize & Verify\", personalChecks, \"Final polish before you call it done\")\npersonalChecks = CheckBoxGroup(\"personal\", [chkFiles, chkShortcuts, chkPrinter, chkTest])\nchkFiles = CheckBoxItem(\"Migrate files & documents\", \"Cloud drive sync or direct transfer from old machine\", \"files\")\nchkShortcuts = CheckBoxItem(\"Set defaults & shortcuts\", \"Default browser, hotkeys, dock/taskbar layout\", \"shortcuts\")\nchkPrinter = CheckBoxItem(\"Add printers & peripherals\", \"Monitors, dock, headset, external drives\", \"peripherals\")\nchkTest = CheckBoxItem(\"Run a full test pass\", \"Camera, mic, speakers, ports, battery health\", \"test\")\n\nchecklistButtons = Buttons([saveBtn, resetBtn], \"row\")\nsaveBtn = Button(\"Save Progress\", Action([@ToAssistant(\"Save my laptop setup checklist progress\")]), \"primary\")\nresetBtn = Button(\"Reset\", Action([@ToAssistant(\"Reset the laptop setup checklist\")]), \"tertiary\")\n\nphasesAccordion = Accordion([accEssentials, accSecurity, accSoftware, accFinish])\n\naccEssentials = AccordionItem(\"p1\", \"Phase 1 details — Essentials (~20 min)\", [essentialsSteps])\nessentialsSteps = Steps([stepUnbox, stepUpdate, stepAccount])\nstepUnbox = StepsItem(\"Unbox and charge\", \"Plug in and let it charge while you run the setup wizard — updates drain the battery fast.\")\nstepUpdate = StepsItem(\"Update the OS first\", \"Patch before installing apps so you don't fight compatibility issues later. Reboot and check again.\")\nstepAccount = StepsItem(\"Create the right user account\", \"Use a standard account for daily work and keep a separate admin account for installs.\")\n\naccSecurity = AccordionItem(\"p2\", \"Phase 2 details — Security (~15 min)\", [securityNote, securitySteps])\nsecurityNote = TextCallout(\"warning\", \"Do this before syncing data\", \"Encryption applied after your files land is slower and messier. Turn it on while the disk is still nearly empty.\")\nsecuritySteps = Steps([stepEncrypt, stepPassword, stepRecovery])\nstepEncrypt = StepsItem(\"Enable full-disk encryption\", \"FileVault (System Settings → Privacy & Security) or BitLocker (Settings → Device encryption).\")\nstepPassword = StepsItem(\"Password manager before anything else\", \"Install it first so every subsequent login gets a unique generated password.\")\nstepRecovery = StepsItem(\"Store recovery keys off-device\", \"Encryption key, MFA backup codes, and account recovery codes go in a safe place, not on the laptop.\")\n\naccSoftware = AccordionItem(\"p3\", \"Phase 3 details — Software install commands\", [macCode, winCode])\nmacCode = CodeBlock(\"bash\", \"# macOS — install Homebrew, then core apps\\n/bin/bash -c \\\"$(curl -fsSL https://raw.githubusercontent.com/Homebrew/install/HEAD/install.sh)\\\"\\n\\nbrew install git gh node python@3.12 ripgrep fzf\\nbrew install --cask visual-studio-code raycast slack rectangle 1password\")\nwinCode = CodeBlock(\"powershell\", \"# Windows — winget one-liners\\nwinget install --id Git.Git -e\\nwinget install --id Microsoft.VisualStudioCode -e\\nwinget install --id Microsoft.PowerShell -e\\nwinget install --id AgileBits.1Password -e\\nwinget install --id OpenJS.NodeJS.LTS -e\")\n\naccFinish = AccordionItem(\"p4\", \"Phase 4 details — Verification checks\", [verifyTable])\nverifyTable = Table([verifyCol, howCol, timeCol])\nverifyCol = Col(\"Check\", [\"Camera & mic\", \"Speakers & headphones\", \"All ports / dock\", \"Keyboard & trackpad\", \"Battery health\", \"Display & external monitor\", \"Backup restore\"])\nhowCol = Col(\"How to verify\", [\"Join a test video call\", \"Play audio, then plug in headset\", \"Plug each USB-C/USB-A device in turn\", \"Type a paragraph, test gestures\", \"Check cycle count and capacity\", \"Confirm resolution and refresh rate\", \"Restore one file from your backup\"])\ntimeCol = Col(\"Minutes\", [3, 2, 4, 2, 1, 3, 5], \"number\")\n\ntipsCallout = Callout(\"info\", \"Pro tip: snapshot your setup\", \"Once everything works, export a list of installed packages (brew bundle dump / winget export) and commit your dotfiles. Next laptop takes 15 minutes instead of 90.\")\n\nactions = Buttons([macBtn, winBtn, devBtn], \"row\")\nmacBtn = Button(\"Tailor this for macOS\", Action([@ToAssistant(\"Give me a macOS-specific new laptop setup checklist with exact settings paths\")]), \"secondary\")\nwinBtn = Button(\"Tailor this for Windows\", Action([@ToAssistant(\"Give me a Windows 11-specific new laptop setup checklist with exact settings paths\")]), \"secondary\")\ndevBtn = Button(\"Add a developer environment section\", Action([@ToAssistant(\"Add a detailed developer environment setup section: SSH keys, Git config, runtime version managers, and containers\")]), \"tertiary\")","promptTokens":6945,"completionTokens":3274,"latencyMs":39916,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/realistic","repeat":0,"output":"root = Stack([headerCard, welcomeCallout, checklistCard, detailsCard, resourcesCard, helpCard], \"column\", \"l\")\n\nheaderCard = Card([header, badges])\nheader = CardHeader(\"First-Day IT Onboarding\", \"Eight steps to get you fully set up — should take about 60–90 minutes\")\nbadges = TagBlock([\"IT Helpdesk\", \"Day 1\", \"~90 min\", \"Required\"])\n\nwelcomeCallout = Callout(\"info\", \"Before you start\", \"Have your welcome email (sent to your personal address) and your phone nearby. You'll need both for account activation and 2FA enrollment.\")\n\nchecklistCard = Card([checklistHeader, checklistForm])\nchecklistHeader = CardHeader(\"Your Checklist\", \"Tick each item as you complete it — order matters, work top to bottom\")\nchecklistForm = Form(\"onboarding\", checklistButtons, [checklistControl])\nchecklistControl = FormControl(\"Setup tasks\", checklistGroup, \"Steps 1–3 must be done in order; the rest can be done in any order.\")\nchecklistGroup = CheckBoxGroup(\"tasks\", checklistItems)\n\nchecklistItems = [item1, item2, item3, item4, item5, item6, item7, item8]\nitem1 = CheckBoxItem(\"1. Activate your company account\", \"Use the activation link in your welcome email to set your primary password and confirm your username.\", \"activate\")\nitem2 = CheckBoxItem(\"2. Enroll in 2FA\", \"Register the authenticator app and save your backup recovery codes somewhere safe.\", \"twofa\")\nitem3 = CheckBoxItem(\"3. Set up the password manager\", \"Install 1Password, join the company vault, and store your recovery kit.\", \"passwordmanager\")\nitem4 = CheckBoxItem(\"4. Install and test the VPN\", \"Connect to the corporate VPN and confirm you can reach the internal wiki.\", \"vpn\")\nitem5 = CheckBoxItem(\"5. Verify laptop disk encryption\", \"Confirm FileVault (Mac) or BitLocker (Windows) is ON and the recovery key is escrowed with IT.\", \"encryption\")\nitem6 = CheckBoxItem(\"6. Confirm device management enrollment\", \"Check that the MDM agent is installed and reporting as compliant.\", \"mdm\")\nitem7 = CheckBoxItem(\"7. Join your team channels\", \"Get added to your squad channel, #general, #announcements, and your on-call rotation.\", \"channels\")\nitem8 = CheckBoxItem(\"8. Complete security awareness training\", \"A 20-minute module in the learning portal — required within your first week.\", \"training\")\n\nchecklistButtons = Buttons([submitBtn, resetBtn])\nsubmitBtn = Button(\"Mark checklist complete\", Action([@ToAssistant(\"I've completed my first-day IT onboarding checklist. What should I do next?\")]), \"primary\")\nresetBtn = Button(\"I'm stuck on a step\", Action([@ToAssistant(\"I'm stuck on one of the IT onboarding steps and need help.\")]), \"secondary\")\n\ndetailsCard = Card([detailsHeader, detailSteps])\ndetailsHeader = CardHeader(\"Step-by-Step Detail\", \"Expand instructions for each task\")\ndetailSteps = Steps([s1, s2, s3, s4, s5, s6, s7, s8])\ns1 = StepsItem(\"Activate your company account\", \"Open the welcome email titled 'Welcome to the team — activate your account'. Click Activate, then set a password of at least 14 characters. Your username follows the firstname.lastname@company.com format. If the link has expired (48 hours), reply to the email or ping the IT helpdesk for a fresh one.\")\ns2 = StepsItem(\"Enroll in two-factor authentication\", \"At the identity portal, choose Security > Add method > Authenticator app. Scan the QR code with Okta Verify or Google Authenticator. Enter the 6-digit code to confirm, then download your 10 backup recovery codes and store them in the password manager once step 3 is done. Add your phone number as a fallback method.\")\ns3 = StepsItem(\"Set up the password manager\", \"Install 1Password from the self-service app catalog. Sign in with your company email, accept the vault invitation, and print or save the Emergency Kit. Move your 2FA recovery codes into your Private vault. Enable the browser extension so credentials autofill going forward.\")\ns4 = StepsItem(\"Install and test the VPN\", \"Install the GlobalProtect client from the app catalog. Enter vpn.company.com as the portal address and sign in with SSO plus 2FA. Once connected, open wiki.internal in your browser — if the page loads, the tunnel is working. Note that the VPN is required for internal tools but not for email or chat.\")\ns5 = StepsItem(\"Verify laptop disk encryption\", \"On macOS: System Settings > Privacy & Security > FileVault should read 'On'. On Windows: Settings > Privacy & security > Device encryption, or run 'manage-bde -status' and confirm Protection Status is On. If encryption is off, do not store any company data yet — open a helpdesk ticket immediately so IT can enable it and escrow the recovery key.\")\ns6 = StepsItem(\"Confirm device management enrollment\", \"Look for the management icon in your menu bar or system tray. Open the self-service portal and check that your device shows as 'Compliant'. This is what keeps your OS patched and lets you install approved apps without an admin password. A 'Non-compliant' status usually clears after you restart and install pending updates.\")\ns7 = StepsItem(\"Join your team channels\", \"Sign in to Slack with SSO. Your manager triggers the automatic channel invites, but confirm you are in #general, #announcements, #it-helpdesk, your squad channel, and your product area channel. Set your display name to your full name, add a photo, and set your local time zone so meeting invites land correctly.\")\ns8 = StepsItem(\"Complete security awareness training\", \"Open the learning portal from the app dashboard and start the 'Security Essentials' module. It covers phishing, data handling, and incident reporting, and takes about 20 minutes. It is required within your first five business days — your account access is reviewed if it is not finished.\")\n\nresourcesCard = Card([resourcesHeader, resourcesTable])\nresourcesHeader = CardHeader(\"Quick Reference\", \"Where each tool lives and who owns it\")\nresourcesTable = Table([toolCol, whereCol, ownerCol, timeCol])\ntoolCol = Col(\"Tool\", [\"Identity portal\", \"1Password\", \"GlobalProtect VPN\", \"FileVault / BitLocker\", \"MDM agent\", \"Slack\", \"Learning portal\"])\nwhereCol = Col(\"Where to get it\", [\"id.company.com\", \"App catalog\", \"App catalog\", \"Built into the OS\", \"Pre-installed\", \"slack.com/signin\", \"learn.company.com\"])\nownerCol = Col(\"Owner\", [\"IT Identity\", \"IT Security\", \"Network\", \"IT Security\", \"IT Endpoint\", \"Workplace\", \"People Ops\"])\ntimeCol = Col(\"Minutes\", [10, 15, 10, 5, 5, 10, 20], \"number\")\n\nhelpCard = Card([helpHeader, helpText, helpButtons], \"sunk\")\nhelpHeader = CardHeader(\"Need a hand?\", \"The helpdesk staffs a live queue for new hires all day\")\nhelpText = TextContent(\"Post in **#it-helpdesk** for anything under five minutes, or open a ticket for account and hardware issues. Walk-up desk is on floor 3 near the elevators, open 9am–5pm.\", \"default\")\nhelpButtons = Buttons([ticketBtn, chatBtn, explainBtn])\nticketBtn = Button(\"Open a helpdesk ticket\", Action([@OpenUrl(\"https://help.company.com/new\")]), \"primary\")\nchatBtn = Button(\"Ask in #it-helpdesk\", Action([@OpenUrl(\"https://slack.com/app_redirect?channel=it-helpdesk\")]), \"secondary\")\nexplainBtn = Button(\"What if a step fails?\", Action([@ToAssistant(\"What are the most common failure points in first-day IT onboarding and how do I fix them?\")]), \"tertiary\")","promptTokens":7002,"completionTokens":2949,"latencyMs":38477,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/minimal","repeat":0,"output":"root = Stack([headerCard, tipCallout, checklistTabs, orderCard, footerButtons], \"column\", \"l\")\n\nheaderCard = Card([headerText, headerTags])\nheaderText = CardHeader(\"New Laptop Setup Checklist\", \"Work through each section — tick items off as you go\")\nheaderTags = TagBlock([\"Setup\", \"Security\", \"Productivity\", \"~2 hours\"])\n\ntipCallout = Callout(\"info\", \"Before you start\", \"Have your account passwords, license keys, and a stable Wi-Fi connection ready. Plug in the charger — some updates take a while.\")\n\nchecklistTabs = Tabs([tabBasics, tabSecurity, tabApps, tabDev, tabPersonal])\n\ntabBasics = TabItem(\"basics\", \"Basics\", [basicsForm])\ntabSecurity = TabItem(\"security\", \"Security\", [securityForm])\ntabApps = TabItem(\"apps\", \"Apps & Data\", [appsForm])\ntabDev = TabItem(\"dev\", \"Dev Setup\", [devForm])\ntabPersonal = TabItem(\"personal\", \"Personalize\", [personalForm])\n\nbasicsForm = Form(\"basics\", basicsButtons, [basicsControl])\nbasicsButtons = Buttons([basicsSaveBtn])\nbasicsSaveBtn = Button(\"Save progress\", null, \"secondary\", \"normal\", \"small\")\nbasicsControl = FormControl(\"First boot essentials\", basicsChecks, \"Do these before anything else\")\nbasicsChecks = CheckBoxGroup(\"basicsChecks\", [chkUnbox, chkOS, chkUpdates, chkWifi, chkPower, chkDisplay])\nchkUnbox = CheckBoxItem(\"Inspect hardware\", \"Check screen, keyboard, ports, and battery health for defects while return window is open\", \"unbox\")\nchkOS = CheckBoxItem(\"Complete OS setup wizard\", \"Set region, language, keyboard layout, and time zone\", \"os\")\nchkUpdates = CheckBoxItem(\"Run all system updates\", \"Repeat until no updates remain, then reboot\", \"updates\")\nchkWifi = CheckBoxItem(\"Connect to Wi-Fi and printers\", \"Save home and office networks\", \"wifi\")\nchkPower = CheckBoxItem(\"Tune power & sleep settings\", \"Balance battery life against performance; set lid-close behavior\", \"power\")\nchkDisplay = CheckBoxItem(\"Adjust display\", \"Scaling, night light, refresh rate, and external monitor arrangement\", \"display\")\n\nsecurityForm = Form(\"security\", securityButtons, [securityControl])\nsecurityButtons = Buttons([securitySaveBtn])\nsecuritySaveBtn = Button(\"Save progress\", null, \"secondary\", \"normal\", \"small\")\nsecurityControl = FormControl(\"Lock it down\", securityChecks, \"Non-negotiables — do these on day one\")\nsecurityChecks = CheckBoxGroup(\"securityChecks\", [chkDiskEnc, chkPassword, chkBio, chkFirewall, chkPwMgr, chkMfa, chkBackup, chkFind])\nchkDiskEnc = CheckBoxItem(\"Enable full-disk encryption\", \"BitLocker on Windows, FileVault on macOS, LUKS on Linux — store the recovery key safely\", \"diskenc\")\nchkPassword = CheckBoxItem(\"Set a strong login password\", \"Unique to this device, not reused anywhere else\", \"password\")\nchkBio = CheckBoxItem(\"Set up biometrics\", \"Fingerprint or face unlock for faster sign-in\", \"bio\")\nchkFirewall = CheckBoxItem(\"Turn on firewall & malware protection\", \"Built-in protection is fine for most people — just confirm it's active\", \"firewall\")\nchkPwMgr = CheckBoxItem(\"Install password manager\", \"Sign in and sync your vault before installing anything else\", \"pwmgr\")\nchkMfa = CheckBoxItem(\"Register 2FA for key accounts\", \"Email, password manager, banking, and work SSO\", \"mfa\")\nchkBackup = CheckBoxItem(\"Configure automatic backups\", \"Time Machine, File History, or a cloud backup client — then verify a test restore\", \"backup\")\nchkFind = CheckBoxItem(\"Enable device tracking\", \"Find My Device / Find My Mac plus remote wipe\", \"find\")\n\nappsForm = Form(\"apps\", appsButtons, [appsControl, transferControl])\nappsButtons = Buttons([appsSaveBtn])\nappsSaveBtn = Button(\"Save progress\", null, \"secondary\", \"normal\", \"small\")\nappsControl = FormControl(\"Install core apps\", appsChecks)\nappsChecks = CheckBoxGroup(\"appsChecks\", [chkBrowser, chkComms, chkOffice, chkMedia, chkUtils])\nchkBrowser = CheckBoxItem(\"Browser + sync\", \"Install your browser, sign in, restore bookmarks and extensions\", \"browser\")\nchkComms = CheckBoxItem(\"Communication tools\", \"Email client, Slack/Teams, calendar, video conferencing\", \"comms\")\nchkOffice = CheckBoxItem(\"Documents & notes\", \"Office suite, PDF reader, note-taking app\", \"office\")\nchkMedia = CheckBoxItem(\"Media & misc\", \"Music, image editor, screenshot tool, archive utility\", \"media\")\nchkUtils = CheckBoxItem(\"Quality-of-life utilities\", \"Window manager, clipboard history, launcher, ad blocker\", \"utils\")\ntransferControl = FormControl(\"Move your data\", transferChecks, \"Keep the old machine until you've verified everything\")\ntransferChecks = CheckBoxGroup(\"transferChecks\", [chkCloud, chkFiles, chkLicenses, chkOldWipe])\nchkCloud = CheckBoxItem(\"Sign in to cloud storage\", \"Drive, Dropbox, iCloud, OneDrive — let the initial sync finish\", \"cloud\")\nchkFiles = CheckBoxItem(\"Transfer files from old device\", \"Migration assistant, external drive, or cloud sync\", \"files\")\nchkLicenses = CheckBoxItem(\"Reactivate software licenses\", \"Deauthorize the old machine first where seat limits apply\", \"licenses\")\nchkOldWipe = CheckBoxItem(\"Sign out & wipe the old laptop\", \"Only after a full week of confirmed-good use on the new one\", \"oldwipe\")\n\ndevForm = Form(\"dev\", devButtons, [devControl])\ndevButtons = Buttons([devSaveBtn])\ndevSaveBtn = Button(\"Save progress\", null, \"secondary\", \"normal\", \"small\")\ndevControl = FormControl(\"Developer environment\", devChecks, \"Skip this tab if you're not writing code\")\ndevChecks = CheckBoxGroup(\"devChecks\", [chkPkgMgr, chkTerminal, chkGit, chkSsh, chkRuntimes, chkEditor, chkContainers, chkDotfiles])\nchkPkgMgr = CheckBoxItem(\"Install a package manager\", \"Homebrew, winget/Scoop, or your distro's manager\", \"pkgmgr\")\nchkTerminal = CheckBoxItem(\"Set up terminal & shell\", \"Terminal app, shell, prompt, and a readable mono font\", \"terminal\")\nchkGit = CheckBoxItem(\"Install and configure Git\", \"Set user.name, user.email, default branch, and line-ending behavior\", \"git\")\nchkSsh = CheckBoxItem(\"Generate SSH & signing keys\", \"Add the public key to GitHub/GitLab and your servers\", \"ssh\")\nchkRuntimes = CheckBoxItem(\"Install language runtimes\", \"Use version managers (nvm, pyenv, rustup) rather than system installs\", \"runtimes\")\nchkEditor = CheckBoxItem(\"Editor / IDE + extensions\", \"Sign in to sync settings, keybindings, and snippets\", \"editor\")\nchkContainers = CheckBoxItem(\"Containers & CLI tooling\", \"Docker, cloud CLIs, database clients, API tools\", \"containers\")\nchkDotfiles = CheckBoxItem(\"Restore dotfiles\", \"Clone your config repo and run the bootstrap script\", \"dotfiles\")\n\npersonalForm = Form(\"personal\", personalButtons, [personalControl])\npersonalButtons = Buttons([personalSaveBtn])\npersonalSaveBtn = Button(\"Save progress\", null, \"secondary\", \"normal\", \"small\")\npersonalControl = FormControl(\"Make it yours\", personalChecks)\npersonalChecks = CheckBoxGroup(\"personalChecks\", [chkShortcuts, chkTrackpad, chkDock, chkNotifs, chkDefaults, chkAccessibility])\nchkShortcuts = CheckBoxItem(\"Remap keyboard shortcuts\", \"Caps Lock, function keys, window snapping\", \"shortcuts\")\nchkTrackpad = CheckBoxItem(\"Tune trackpad & mouse\", \"Tracking speed, scroll direction, gestures\", \"trackpad\")\nchkDock = CheckBoxItem(\"Organize dock / taskbar\", \"Pin your daily apps, hide the rest\", \"dock\")\nchkNotifs = CheckBoxItem(\"Trim notifications\", \"Turn off everything non-essential now, before habits form\", \"notifs\")\nchkDefaults = CheckBoxItem(\"Set default apps\", \"Browser, mail, PDF, image, and text file handlers\", \"defaults\")\nchkAccessibility = CheckBoxItem(\"Accessibility & comfort\", \"Text size, contrast, dark mode, reduce motion\", \"accessibility\")\n\norderCard = Card([orderHeader, orderSteps])\norderHeader = CardHeader(\"Suggested order\", \"Roughly two hours of hands-on time, plus update downloads\")\norderSteps = Steps([step1, step2, step3, step4, step5, step6])\nstep1 = StepsItem(\"1. Hardware check (10 min)\", \"Inspect the machine, confirm the specs match what you ordered, and check battery cycle count. Report issues immediately while returns are easy.\")\nstep2 = StepsItem(\"2. OS setup & updates (30 min)\", \"Run the setup wizard, connect to Wi-Fi, then install every pending update and reboot. Repeat until the update list is empty.\")\nstep3 = StepsItem(\"3. Security baseline (20 min)\", \"Enable disk encryption and save the recovery key, set a strong password plus biometrics, confirm the firewall is on, and turn on device tracking.\")\nstep4 = StepsItem(\"4. Password manager & accounts (15 min)\", \"Install your password manager first, then sign in to email and cloud storage so everything else can authenticate cleanly.\")\nstep5 = StepsItem(\"5. Apps & data migration (30+ min)\", \"Install core apps, start the cloud sync, and transfer files from the old device. Let large syncs run in the background.\")\nstep6 = StepsItem(\"6. Personalize & verify (15 min)\", \"Tune input devices, notifications, and defaults. Run a backup, verify a test restore, then keep the old laptop untouched for a week.\")\n\nfooterButtons = Buttons([devDeepBtn, secDeepBtn, printBtn])\ndevDeepBtn = Button(\"Detailed dev environment setup\", Action([@ToAssistant(\"Give me a detailed developer environment setup checklist for a new laptop, including shell config, version managers, and dotfiles.\")]), \"primary\")\nsecDeepBtn = Button(\"Harden security further\", Action([@ToAssistant(\"What advanced security hardening steps should I take on a new laptop beyond the basics?\")]), \"secondary\")\nprintBtn = Button(\"Make a printable version\", Action([@ToAssistant(\"Convert this new laptop setup checklist into a plain printable markdown checklist.\")]), \"tertiary\")","promptTokens":6945,"completionTokens":4387,"latencyMs":50349,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/minimal","repeat":1,"output":"root = Stack([header, intro, tabs, progressNote, actions])\n\nheader = CardHeader(\"New Laptop Setup Checklist\", \"Everything to do from unboxing to first backup\")\n\nintro = TextCallout(\"info\", \"Set aside about 2–3 hours\", \"Most steps are quick, but OS updates and file transfers run long in the background. Start those early and work through the rest while you wait.\")\n\ntabs = Tabs([tabChecklist, tabPhases, tabApps, tabSecurity])\n\ntabChecklist = TabItem(\"checklist\", \"Checklist\", [checklistForm])\n\nchecklistForm = Form(\"laptop-setup\", checklistButtons, [essentialsControl, softwareControl, securityControl, finishingControl])\n\nessentialsControl = FormControl(\"Day one essentials\", essentialsGroup, \"Do these before anything else\")\nessentialsGroup = CheckBoxGroup(\"essentials\", [essential1, essential2, essential3, essential4, essential5])\nessential1 = CheckBoxItem(\"Inspect hardware\", \"Check screen, keyboard, ports and hinge for damage before the return window closes\", \"inspect\")\nessential2 = CheckBoxItem(\"Charge to 100%\", \"Run a full first charge and confirm the battery reports a healthy capacity\", \"charge\")\nessential3 = CheckBoxItem(\"Complete OS setup wizard\", \"Region, keyboard layout, network, and account creation\", \"wizard\")\nessential4 = CheckBoxItem(\"Install all system updates\", \"Reboot and re-check — updates often arrive in several rounds\", \"updates\")\nessential5 = CheckBoxItem(\"Register warranty & save serial\", \"Record the serial number and purchase receipt somewhere off the device\", \"warranty\")\n\nsoftwareControl = FormControl(\"Software & accounts\", softwareGroup, \"Get to a working environment\")\nsoftwareGroup = CheckBoxGroup(\"software\", [soft1, soft2, soft3, soft4, soft5])\nsoft1 = CheckBoxItem(\"Install a browser & sign in\", \"Sync bookmarks, extensions and passwords\", \"browser\")\nsoft2 = CheckBoxItem(\"Install a package manager\", \"Homebrew on macOS, winget or Chocolatey on Windows\", \"pkgmgr\")\nsoft3 = CheckBoxItem(\"Restore files\", \"Migration assistant, cloud sync, or external drive copy\", \"files\")\nsoft4 = CheckBoxItem(\"Install core apps\", \"Editor, terminal, office suite, communication tools\", \"apps\")\nsoft5 = CheckBoxItem(\"Set up printers & peripherals\", \"Monitor, dock, mouse, headset, printer drivers\", \"peripherals\")\n\nsecurityControl = FormControl(\"Security & privacy\", securityGroup, \"Do not skip these\")\nsecurityGroup = CheckBoxGroup(\"security\", [sec1, sec2, sec3, sec4, sec5])\nsec1 = CheckBoxItem(\"Enable full disk encryption\", \"FileVault on macOS, BitLocker on Windows — store the recovery key safely\", \"encryption\")\nsec2 = CheckBoxItem(\"Set a strong login password\", \"Plus biometrics (fingerprint or face) if available\", \"password\")\nsec3 = CheckBoxItem(\"Install a password manager\", \"Sign in and enable autofill in the browser\", \"passwordmgr\")\nsec4 = CheckBoxItem(\"Turn on the firewall\", \"Block incoming connections by default\", \"firewall\")\nsec5 = CheckBoxItem(\"Enable device find & remote wipe\", \"Find My or Find My Device, so a lost laptop is not a lost identity\", \"findmy\")\n\nfinishingControl = FormControl(\"Finishing touches\", finishingGroup, \"Comfort and long-term maintenance\")\nfinishingGroup = CheckBoxGroup(\"finishing\", [fin1, fin2, fin3, fin4])\nfin1 = CheckBoxItem(\"Configure backups\", \"Time Machine, File History, or a cloud backup service — then verify a restore works\", \"backups\")\nfin2 = CheckBoxItem(\"Tune display & power settings\", \"Brightness, night shift, sleep timers, battery saver thresholds\", \"display\")\nfin3 = CheckBoxItem(\"Customize keyboard & trackpad\", \"Shortcuts, gestures, key repeat, scroll direction\", \"input\")\nfin4 = CheckBoxItem(\"Securely wipe the old laptop\", \"Sign out of all accounts, deauthorize licenses, then erase the drive\", \"oldlaptop\")\n\nchecklistButtons = Buttons([saveBtn, resetBtn])\nsaveBtn = Button(\"Save progress\", null, \"primary\")\nresetBtn = Button(\"Reset\", null, \"tertiary\")\n\ntabPhases = TabItem(\"phases\", \"Order of operations\", [phaseSteps, phaseNote])\n\nphaseSteps = Steps([step1, step2, step3, step4, step5, step6])\nstep1 = StepsItem(\"1. Unbox and inspect\", \"Photograph the box and serial number, check for physical defects, and confirm the specs match what you ordered. Return issues are easiest to resolve in the first few days.\")\nstep2 = StepsItem(\"2. First boot and updates\", \"Run the setup wizard with a wired or stable Wi-Fi connection, then immediately install every available OS update and reboot until none remain.\")\nstep3 = StepsItem(\"3. Lock it down\", \"Enable disk encryption, set a strong password plus biometrics, turn on the firewall, and enable remote find/wipe before any personal data lands on the machine.\")\nstep4 = StepsItem(\"4. Install the toolchain\", \"Package manager first, then browser, password manager, editor, terminal, and your day-to-day apps. Scripted installs save time on future machines.\")\nstep5 = StepsItem(\"5. Migrate data\", \"Pull files from cloud sync or an external drive. Copy dotfiles, SSH keys, and license keys deliberately rather than cloning the whole old system.\")\nstep6 = StepsItem(\"6. Back up and verify\", \"Configure automated backups, run the first full backup, then restore one test file to prove it actually works.\")\n\nphaseNote = TextContent(\"**Tip:** kick off step 2 and step 5 in the background — they are the two slowest phases.\", \"small\")\n\ntabApps = TabItem(\"apps\", \"App shortlist\", [appsTable, appsNote])\n\nappsTable = Table([appCol, categoryCol, priorityCol])\nappCol = Col(\"Software\", [\"Password manager\", \"Browser\", \"Package manager\", \"Code editor\", \"Terminal emulator\", \"Cloud storage client\", \"Backup tool\", \"Office / docs suite\", \"Chat & video calls\", \"Media player\", \"Archive utility\", \"VPN client\"])\ncategoryCol = Col(\"Category\", [\"Security\", \"Essential\", \"Developer\", \"Developer\", \"Developer\", \"Files\", \"Security\", \"Productivity\", \"Productivity\", \"Media\", \"Utility\", \"Network\"])\npriorityCol = Col(\"Priority\", priorityTags)\npriorityTags = [tagHigh1, tagHigh2, tagHigh3, tagMed1, tagMed2, tagHigh4, tagHigh5, tagMed3, tagMed4, tagLow1, tagLow2, tagLow3]\ntagHigh1 = Tag(\"Must have\", null, \"sm\", \"danger\")\ntagHigh2 = Tag(\"Must have\", null, \"sm\", \"danger\")\ntagHigh3 = Tag(\"Must have\", null, \"sm\", \"danger\")\ntagMed1 = Tag(\"Recommended\", null, \"sm\", \"warning\")\ntagMed2 = Tag(\"Recommended\", null, \"sm\", \"warning\")\ntagHigh4 = Tag(\"Must have\", null, \"sm\", \"danger\")\ntagHigh5 = Tag(\"Must have\", null, \"sm\", \"danger\")\ntagMed3 = Tag(\"Recommended\", null, \"sm\", \"warning\")\ntagMed4 = Tag(\"Recommended\", null, \"sm\", \"warning\")\ntagLow1 = Tag(\"Optional\", null, \"sm\", \"neutral\")\ntagLow2 = Tag(\"Optional\", null, \"sm\", \"neutral\")\ntagLow3 = Tag(\"Optional\", null, \"sm\", \"neutral\")\n\nappsNote = TextContent(\"Install the password manager *first* — every other sign-in depends on it.\", \"small\")\n\ntabSecurity = TabItem(\"security\", \"Deep dives\", [securityAccordion])\n\nsecurityAccordion = Accordion([acc1, acc2, acc3, acc4])\nacc1 = AccordionItem(\"encryption\", \"Full disk encryption\", [accEncryptionText, accEncryptionWarning])\naccEncryptionText = MarkDownRenderer(\"**macOS:** System Settings → Privacy & Security → FileVault → Turn On.\\n\\n**Windows:** Settings → Privacy & Security → Device encryption, or `manage-bde -on C:` for BitLocker on Pro editions.\\n\\n**Linux:** easiest during install — choose LUKS full-disk encryption at the partitioning step.\")\naccEncryptionWarning = Callout(\"warning\", \"Store the recovery key off-device\", \"Print it or put it in your password manager on another device. If you lose it and forget your password, the data is unrecoverable by design.\")\n\nacc2 = AccordionItem(\"backups\", \"Backup strategy\", [accBackupText])\naccBackupText = MarkDownRenderer(\"Aim for **3-2-1**: three copies of your data, on two different media, with one copy offsite.\\n\\n- **Local:** Time Machine or File History to an external SSD, hourly\\n- **Cloud:** Backblaze, iDrive, or your OS's built-in cloud backup, continuous\\n- **Sync is not backup:** a deleted or encrypted file syncs the damage everywhere\\n\\nSchedule a reminder to test a restore once a quarter.\")\n\nacc3 = AccordionItem(\"dotfiles\", \"Developer environment\", [accDotfilesText, accDotfilesCode])\naccDotfilesText = TextContent(\"Generate a fresh SSH key rather than copying the old one, then add it to GitHub, GitLab, and any servers you use.\")\naccDotfilesCode = CodeBlock(\"bash\", \"# Generate a modern SSH key\\nssh-keygen -t ed25519 -C \\\"you@example.com\\\"\\n\\n# Start the agent and add the key\\neval \\\"$(ssh-agent -s)\\\"\\nssh-add ~/.ssh/id_ed25519\\n\\n# Copy the public key to your clipboard\\ncat ~/.ssh/id_ed25519.pub\\n\\n# Verify the connection\\nssh -T git@github.com\")\n\nacc4 = AccordionItem(\"oldlaptop\", \"Retiring the old laptop\", [accOldText])\naccOldText = MarkDownRenderer(\"1. Confirm the new machine has everything you need — wait a couple of weeks\\n2. Sign out of iCloud / Microsoft account / Google\\n3. Deauthorize licensed software (Adobe, JetBrains, Office, music apps)\\n4. Unpair Bluetooth devices and remove the machine from Find My\\n5. Erase the drive and reinstall a clean OS\\n6. Recycle, trade in, or donate — never bin a laptop with a battery in it\")\n\nprogressNote = Callout(\"neutral\", \"Work top to bottom\", \"The tabs above are ordered the way the tasks should actually happen: security before data, data before customization.\")\n\nactions = Buttons([macBtn, winBtn, devBtn])\nmacBtn = Button(\"Tailor this for macOS\", Action([@ToAssistant(\"Tailor this new laptop setup checklist specifically for macOS, with exact System Settings paths and Homebrew commands.\")]), \"secondary\")\nwinBtn = Button(\"Tailor this for Windows\", Action([@ToAssistant(\"Tailor this new laptop setup checklist specifically for Windows 11, including debloating steps and winget commands.\")]), \"secondary\")\ndevBtn = Button(\"Add a developer setup script\", Action([@ToAssistant(\"Write a single shell script that automates the developer portions of this new laptop setup checklist.\")]), \"tertiary\")","promptTokens":6945,"completionTokens":4144,"latencyMs":50419,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/minimal","repeat":3,"output":"root = Stack([header, intro, phaseSteps, checklistTabs, tipsAccordion, footerNote], \"column\", \"l\")\n\nheader = CardHeader(\"New Laptop Setup Checklist\", \"A step-by-step guide to go from unboxing to fully productive\")\n\nintro = Callout(\"info\", \"Budget about 2–3 hours\", \"Most of the time is spent waiting on OS updates and app downloads. Start the update process first, then work through the rest while it runs.\")\n\nphaseSteps = Card([phaseHeader, steps])\nphaseHeader = CardHeader(\"The 6 Phases\", \"Do these in order\")\nsteps = Steps([s1, s2, s3, s4, s5, s6])\ns1 = StepsItem(\"1. Unbox & inspect\", \"Check the screen, keyboard, ports, and chassis for damage. Confirm the model, RAM, and storage match your order. Keep the box and receipt until you're sure everything works.\")\ns2 = StepsItem(\"2. Charge & first boot\", \"Plug in and let it charge to at least 50%. Run the out-of-box setup: language, region, keyboard layout, Wi-Fi, and time zone.\")\ns3 = StepsItem(\"3. Accounts & updates\", \"Sign in to your OS account, then immediately run all system updates and reboot as needed. Also update firmware/BIOS and drivers (GPU, chipset, Wi-Fi).\")\ns4 = StepsItem(\"4. Security hardening\", \"Enable full-disk encryption, set a strong password plus biometrics, turn on the firewall, and configure automatic lock and screen timeout.\")\ns5 = StepsItem(\"5. Install software\", \"Install your browser, password manager, communication apps, and role-specific tools. Use a package manager to make this repeatable.\")\ns6 = StepsItem(\"6. Data, backup & tuning\", \"Restore or sync your files, set up an automated backup, then tune power settings, display scaling, and keyboard shortcuts.\")\n\nchecklistTabs = Tabs([tabEssentials, tabSecurity, tabApps, tabDev, tabPersonalize])\n\ntabEssentials = TabItem(\"essentials\", \"Essentials\", [essentialsForm])\nessentialsForm = Form(\"essentials\", essentialsButtons, [essentialsControl])\nessentialsControl = FormControl(\"Do these before anything else\", essentialsChecks, \"Tick items off as you complete them\")\nessentialsChecks = CheckBoxGroup(\"essentials_items\", [e1, e2, e3, e4, e5, e6, e7])\ne1 = CheckBoxItem(\"Inspect hardware\", \"Screen, keys, trackpad, ports, hinge, and battery health\", \"inspect\")\ne2 = CheckBoxItem(\"Run OS setup wizard\", \"Language, region, Wi-Fi, time zone, account sign-in\", \"oobe\")\ne3 = CheckBoxItem(\"Install all system updates\", \"Reboot and re-check until no updates remain\", \"os_updates\")\ne4 = CheckBoxItem(\"Update firmware / BIOS\", \"Vendor update tool or manufacturer support site\", \"firmware\")\ne5 = CheckBoxItem(\"Install GPU & chipset drivers\", \"Especially important for gaming or creative work\", \"drivers\")\ne6 = CheckBoxItem(\"Check battery health & power plan\", \"Set balanced plan; verify charging works on all ports\", \"battery\")\ne7 = CheckBoxItem(\"Register warranty\", \"Save serial number, receipt, and support contact\", \"warranty\")\nessentialsButtons = Buttons([saveEssentials, askEssentials])\nsaveEssentials = Button(\"Mark phase complete\", Action([@ToAssistant(\"I finished the essentials phase of the laptop setup. What should I do next?\")]), \"primary\")\naskEssentials = Button(\"How do I update my BIOS?\", Action([@ToAssistant(\"Explain safely how to update the BIOS/firmware on a new laptop.\")]), \"tertiary\")\n\ntabSecurity = TabItem(\"security\", \"Security\", [securityForm, securityCallout])\nsecurityForm = Form(\"security\", securityButtons, [securityControl])\nsecurityControl = FormControl(\"Lock it down\", securityChecks, \"Do this before you put real data on the machine\")\nsecurityChecks = CheckBoxGroup(\"security_items\", [sec1, sec2, sec3, sec4, sec5, sec6, sec7, sec8])\nsec1 = CheckBoxItem(\"Enable full-disk encryption\", \"BitLocker (Windows), FileVault (macOS), or LUKS (Linux)\", \"encryption\")\nsec2 = CheckBoxItem(\"Strong login password\", \"Unique, 16+ characters, stored in your password manager\", \"password\")\nsec3 = CheckBoxItem(\"Enable biometrics\", \"Fingerprint or face unlock as a convenience layer\", \"biometrics\")\nsec4 = CheckBoxItem(\"Turn on the firewall\", \"Block inbound connections by default\", \"firewall\")\nsec5 = CheckBoxItem(\"Auto-lock after 5 minutes\", \"Plus lock-on-lid-close and a screen saver password\", \"autolock\")\nsec6 = CheckBoxItem(\"Install password manager first\", \"So every later account uses a unique password\", \"pwmanager\")\nsec7 = CheckBoxItem(\"Set up 2FA app\", \"Add codes for email, cloud, and work accounts\", \"twofa\")\nsec8 = CheckBoxItem(\"Enable device find & remote wipe\", \"Find My Device / Find My Mac\", \"findmy\")\nsecurityButtons = Buttons([saveSecurity, askSecurity])\nsaveSecurity = Button(\"Mark phase complete\", Action([@ToAssistant(\"I finished the security phase of my new laptop setup.\")]), \"primary\")\naskSecurity = Button(\"Explain disk encryption\", Action([@ToAssistant(\"Explain full-disk encryption options per OS and any performance tradeoffs.\")]), \"tertiary\")\nsecurityCallout = TextCallout(\"warning\", \"Encrypt before you sync\", \"Turning on encryption after copying gigabytes of files is slower and leaves plaintext remnants on disk.\")\n\ntabApps = TabItem(\"apps\", \"Core apps\", [appsTable, appsNote])\nappsTable = Table([appCol, catCol, whyCol, priorityCol])\nappCol = Col(\"App\", [\"Password manager\", \"Browser + profiles\", \"Cloud storage\", \"Chat / video\", \"Notes\", \"Office suite\", \"Media player\", \"Archive tool\", \"Antivirus check\", \"Package manager\"])\ncatCol = Col(\"Category\", [\"Security\", \"Web\", \"Files\", \"Comms\", \"Productivity\", \"Productivity\", \"Media\", \"Utility\", \"Security\", \"Utility\"])\nwhyCol = Col(\"Why\", [\"Needed before creating other accounts\", \"Sign in and sync bookmarks & extensions\", \"Restore documents automatically\", \"Slack, Teams, Zoom, or Discord\", \"Obsidian, Notion, or Apple Notes\", \"Docs, sheets, and slides\", \"VLC handles anything\", \"7-Zip or Keka\", \"Built-in defender is usually enough\", \"winget, Homebrew, or apt\"])\npriorityCol = Col(\"Priority\", [1, 1, 1, 2, 2, 2, 3, 3, 2, 1], \"number\")\nappsNote = TextContent(\"**Tip:** install with one command. `winget install --id Mozilla.Firefox` on Windows, or `brew install --cask firefox` on macOS. Save the full list as a script so your next laptop takes 10 minutes.\")\n\ntabDev = TabItem(\"dev\", \"Developer extras\", [devSteps, devCode])\ndevSteps = Steps([d1, d2, d3, d4, d5])\nd1 = StepsItem(\"Terminal & shell\", \"Install a good terminal, pick zsh or fish, and add a prompt like Starship.\")\nd2 = StepsItem(\"Package manager\", \"Homebrew, winget/scoop, or your distro's manager — everything else installs through it.\")\nd3 = StepsItem(\"Git & SSH keys\", \"Set user.name and user.email, generate an ed25519 key, add it to GitHub/GitLab.\")\nd4 = StepsItem(\"Runtimes via version managers\", \"Use nvm, pyenv, or mise so you can switch versions per project.\")\nd5 = StepsItem(\"Editor & dotfiles\", \"Install your editor, sign in to sync settings, then clone your dotfiles repo.\")\ndevCode = CodeBlock(\"bash\", \"# generate an SSH key\\nssh-keygen -t ed25519 -C \\\"you@example.com\\\"\\neval \\\"$(ssh-agent -s)\\\"\\nssh-add ~/.ssh/id_ed25519\\n\\n# git identity\\ngit config --global user.name \\\"Your Name\\\"\\ngit config --global user.email \\\"you@example.com\\\"\\ngit config --global init.defaultBranch main\\ngit config --global pull.rebase true\")\n\ntabPersonalize = TabItem(\"personalize\", \"Personalize\", [personalizeForm])\npersonalizeForm = Form(\"personalize\", personalizeButtons, [personalizeControl])\npersonalizeControl = FormControl(\"Comfort and workflow\", personalizeChecks, \"Small tweaks that pay off every day\")\npersonalizeChecks = CheckBoxGroup(\"personalize_items\", [p1, p2, p3, p4, p5, p6, p7])\np1 = CheckBoxItem(\"Display scaling & night light\", \"Set resolution, scaling, and a warm evening color profile\", \"display\")\np2 = CheckBoxItem(\"Trackpad & scroll direction\", \"Enable tap-to-click and set gestures you actually use\", \"trackpad\")\np3 = CheckBoxItem(\"Keyboard shortcuts\", \"Window snapping, screenshot tool, launcher hotkey\", \"shortcuts\")\np4 = CheckBoxItem(\"Remove preinstalled bloatware\", \"Uninstall trialware and vendor toolbars\", \"bloatware\")\np5 = CheckBoxItem(\"Configure automated backup\", \"Time Machine, File History, or a cloud backup client\", \"backup\")\np6 = CheckBoxItem(\"Set default apps\", \"Browser, mail, PDF reader, image viewer\", \"defaults\")\np7 = CheckBoxItem(\"Create a restore point / snapshot\", \"Clean baseline you can roll back to\", \"snapshot\")\npersonalizeButtons = Buttons([savePersonalize, askPersonalize])\nsavePersonalize = Button(\"Finish setup\", Action([@ToAssistant(\"I finished setting up my new laptop. Anything commonly forgotten?\")]), \"primary\")\naskPersonalize = Button(\"Suggest a backup strategy\", Action([@ToAssistant(\"Suggest a practical 3-2-1 backup strategy for a personal laptop.\")]), \"tertiary\")\n\ntipsAccordion = Accordion([a1, a2, a3])\na1 = AccordionItem(\"migrate\", \"Migrating from an old laptop\", [migrateText])\nmigrateText = TextContent(\"Prefer a **clean install plus selective copy** over a full migration tool — you leave old cruft behind. Copy: Documents, Desktop, photos, SSH keys, browser profile, and app license keys. Skip: caches, old installers, and system folders. Keep the old machine untouched for two weeks before wiping it, in case you missed something.\")\na2 = AccordionItem(\"mistakes\", \"Common mistakes to avoid\", [mistakesText])\nmistakesText = TextContent(\"- Skipping firmware updates because \\\"the OS is up to date\\\"\\n- Reusing an old password on the new machine\\n- Downloading apps from ad-laden download sites instead of official sources or a package manager\\n- Never testing the backup you configured\\n- Forgetting to deauthorize licenses on the old laptop before wiping it\")\na3 = AccordionItem(\"worktips\", \"If it's a work laptop\", [workText])\nworkText = TextContent(\"Enroll in device management (MDM) before installing anything personal, connect the VPN and test it off-network, join required chat channels, and confirm you can access the code repo, ticket tracker, and password vault. Ask IT what the approved software list and backup policy are so you don't have to redo work later.\")\n\nfooterNote = TextContent(\"_Tick items as you go — the phases are ordered so nothing has to be redone._\", \"small\")","promptTokens":6945,"completionTokens":4347,"latencyMs":54912,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/realistic","repeat":1,"output":"root = Stack([header, welcomeNote, checklistCard, detailStack, helpCard, footerNote, supportButtons], \"column\", \"l\")\n\nheader = CardHeader(\"Day One IT Onboarding\", \"Eight steps to get you fully set up — tick each one off as you go\")\n\nwelcomeNote = Callout(\"info\", \"Before you start\", \"Your manager should have handed you a welcome envelope with your temporary username and a one-time activation code. Grab that first — you'll need it in step 1.\")\n\nchecklistCard = Card([checklistHeader, checklistForm], \"card\")\nchecklistHeader = CardHeader(\"Your Checklist\", \"Progress saves as you tick — aim to finish before lunch\")\nchecklistForm = Form(\"onboarding\", checklistButtons, [checklistField])\nchecklistField = FormControl(\"Setup tasks\", onboardingChecks, \"Stuck on any item? Use the Help & Troubleshooting section below.\")\n\nonboardingChecks = CheckBoxGroup(\"tasks\", checkItems)\ncheckItems = [item1, item2, item3, item4, item5, item6, item7, item8]\n\nitem1 = CheckBoxItem(\"1. Activate your account\", \"Sign in at id.company.com with your temporary username and one-time code, then set your permanent password.\", \"activate\")\nitem2 = CheckBoxItem(\"2. Install the password manager\", \"Install 1Password from Self Service, join the Company vault via your invite email, and save your new password there.\", \"pwmanager\")\nitem3 = CheckBoxItem(\"3. Enrol in 2FA\", \"Add your work account to Okta Verify on your phone and print or save your eight backup codes.\", \"twofa\")\nitem4 = CheckBoxItem(\"4. Connect to VPN\", \"Open GlobalProtect, enter vpn.company.com, sign in with SSO, and confirm the shield icon turns green.\", \"vpn\")\nitem5 = CheckBoxItem(\"5. Verify laptop encryption\", \"Run the Encryption Check app and confirm FileVault (macOS) or BitLocker (Windows) reports 'Enabled'.\", \"encryption\")\nitem6 = CheckBoxItem(\"6. Join your team channels\", \"Accept your Slack invite and join #general, #it-helpdesk, plus the channels your manager listed.\", \"channels\")\nitem7 = CheckBoxItem(\"7. Set up email and calendar\", \"Sign in to Outlook, add your signature from the template, and share your calendar with your team.\", \"email\")\nitem8 = CheckBoxItem(\"8. Complete security training\", \"Finish the 20-minute Security Essentials module in the Learning portal — required within your first week.\", \"training\")\n\nchecklistButtons = Buttons([saveBtn, resetBtn])\nsaveBtn = Button(\"Save progress\", Action([@ToAssistant(\"Save my IT onboarding checklist progress\")]), \"primary\")\nresetBtn = Button(\"Start over\", Action([@ToAssistant(\"Reset my IT onboarding checklist\")]), \"tertiary\")\n\ndetailStack = Card([detailHeader, detailSteps], \"sunk\")\ndetailHeader = CardHeader(\"Step-by-step detail\", \"Expanded instructions for each checklist item\")\ndetailSteps = Steps([s1, s2, s3, s4, s5, s6, s7, s8])\n\ns1 = StepsItem(\"Activate your account\", \"Go to id.company.com/activate on your laptop. Enter the temporary username (usually firstname.lastname) and the one-time code from your welcome envelope. Choose a password of at least 14 characters. The code expires 72 hours after your start date — if it's dead, message #it-helpdesk.\")\ns2 = StepsItem(\"Install the password manager\", \"Open Self Service (macOS) or Company Portal (Windows) and install 1Password. Check your inbox for the vault invite, accept it, then immediately store the password you just created. Never keep credentials in notes, spreadsheets, or your browser.\")\ns3 = StepsItem(\"Enrol in two-factor authentication\", \"Install Okta Verify on your phone, then visit id.company.com/security and scan the QR code. Approve the test push. Download your eight backup codes and store them in 1Password — you'll need one if you ever lose your phone.\")\ns4 = StepsItem(\"Connect to the VPN\", \"Launch GlobalProtect from your menu bar or system tray. Portal address is vpn.company.com. Sign in with SSO and approve the 2FA push. A green shield means you're on. VPN is required for internal tools like the wiki, Jira, and staging environments.\")\ns5 = StepsItem(\"Verify laptop encryption\", \"Open the Encryption Check app from Self Service. It should report FileVault: Enabled on macOS or BitLocker: Enabled on Windows, plus a recovery key escrowed to IT. If either shows 'Not Enabled', stop and raise a ticket before storing any work files.\")\ns6 = StepsItem(\"Join your team channels\", \"Accept the Slack invite in your inbox and sign in with SSO. Join #general, #announcements, and #it-helpdesk as a baseline. Your manager will share the team-specific list — ask them to add you to any private channels.\")\ns7 = StepsItem(\"Set up email and calendar\", \"Sign in to Outlook with SSO. Paste the standard signature from the wiki template and fill in your name, title, and pronouns. Set your working hours and share your calendar with your team at 'limited details' or above.\")\ns8 = StepsItem(\"Complete Security Essentials\", \"Open learn.company.com and finish the Security Essentials module — roughly 20 minutes covering phishing, device handling, and data classification. It's mandatory within your first five working days.\")\n\nhelpCard = Card([helpHeader, helpAccordion], \"card\")\nhelpHeader = CardHeader(\"Help and troubleshooting\", \"The four things that trip people up most often\")\nhelpAccordion = Accordion([h1, h2, h3, h4])\n\nh1 = AccordionItem(\"h1\", \"My activation code doesn't work\", [h1text])\nh1text = TextContent(\"Codes expire 72 hours after your official start date and are single-use. If yours is rejected, post in **#it-helpdesk** with your full name and start date — the team can reissue one in a few minutes. Don't try more than five times or the account locks for 30 minutes.\")\n\nh2 = AccordionItem(\"h2\", \"VPN connects but internal sites won't load\", [h2text])\nh2text = TextContent(\"Nine times out of ten this is a stale DNS cache. Disconnect GlobalProtect, quit the app fully, reconnect, then reload. If the shield is green but the wiki still 404s, you may be missing a group membership — ask your manager to confirm you're in the right access group.\")\n\nh3 = AccordionItem(\"h3\", \"I lost my phone before saving backup codes\", [h3text])\nh3text = TextContent(\"Call the IT service desk on **x4400** rather than raising a ticket — identity resets require a live verbal identity check with your manager on the line. Expect about 15 minutes. Once restored, save the new backup codes to 1Password straight away.\")\n\nh4 = AccordionItem(\"h4\", \"Encryption check says 'Not Enabled'\", [h4text])\nh4text = TextContent(\"This is a blocker, not a warning. Do not copy customer data, source code, or documents to the device. Raise a **P2 ticket** tagged `device-encryption` and IT will remotely enable it, usually within the hour. You can keep working in the browser meanwhile.\")\n\nfooterNote = TextCallout(\"success\", \"Done with all eight?\", \"Drop a message in #it-helpdesk saying you've finished onboarding. IT will run a verification sweep on your device and mark your setup complete in the system.\")\n\nsupportButtons = Buttons([helpdeskBtn, wikiBtn, ticketBtn])\nhelpdeskBtn = Button(\"Ask a question\", Action([@ToAssistant(\"I have a question about my first-day IT onboarding steps\")]), \"primary\")\nwikiBtn = Button(\"Open the IT wiki\", Action([@OpenUrl(\"https://wiki.company.com/it/onboarding\")]), \"secondary\")\nticketBtn = Button(\"Raise a ticket\", Action([@OpenUrl(\"https://helpdesk.company.com/new\")]), \"tertiary\")","promptTokens":7002,"completionTokens":2920,"latencyMs":38762,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/realistic","repeat":3,"output":"root = Stack([header, progressCallout, checklistCard, stepsCard, helpTabs, footerButtons], \"column\", \"l\")\n\nheader = CardHeader(\"First-Day IT Onboarding\", \"Work through these eight items in order — tick each one off as you finish. Most people wrap up in about 45 minutes.\")\n\nprogressCallout = Callout(\"info\", \"Start here\", \"Do the steps top to bottom. Account activation must be done before the password manager, VPN, or 2FA will work.\")\n\nchecklistCard = Card([checklistHeader, checklist, checklistNote], \"card\")\n\nchecklistHeader = CardHeader(\"Your checklist\", \"Tick items as you complete them\")\n\nchecklist = CheckBoxGroup(\"onboarding\", [item1, item2, item3, item4, item5, item6, item7, item8])\n\nitem1 = CheckBoxItem(\"1. Activate your account\", \"Go to id.company.com/activate, enter the employee ID from your welcome email, and set your permanent password.\", \"activate\")\nitem2 = CheckBoxItem(\"2. Set up the password manager\", \"Install 1Password from Self Service, sign in with your new company account, and accept the Team vault invite.\", \"passwordManager\")\nitem3 = CheckBoxItem(\"3. Enroll in 2FA\", \"Register Okta Verify on your phone plus one backup method (hardware key or SMS). Save your recovery codes in 1Password.\", \"twofa\")\nitem4 = CheckBoxItem(\"4. Install and test the VPN\", \"Install GlobalProtect, connect to the vpn-us gateway, and confirm you can reach wiki.internal.\", \"vpn\")\nitem5 = CheckBoxItem(\"5. Verify laptop encryption\", \"Confirm FileVault (macOS) or BitLocker (Windows) is On and the recovery key is escrowed to IT.\", \"encryption\")\nitem6 = CheckBoxItem(\"6. Confirm device management check-in\", \"Open the Company Portal app and make sure your device shows as Compliant with no pending actions.\", \"mdm\")\nitem7 = CheckBoxItem(\"7. Join your team channels\", \"Sign into Slack and join #general, #it-helpdesk, your squad channel, and your department announcements channel.\", \"channels\")\nitem8 = CheckBoxItem(\"8. Check calendar and email access\", \"Confirm Outlook/Gmail is syncing, your calendar shows your onboarding meetings, and your signature is set.\", \"mail\")\n\nchecklistNote = TextContent(\"Stuck on any item for more than 10 minutes? Skip ahead and flag it in #it-helpdesk — don't burn your first morning on one step.\", \"small\")\n\nstepsCard = Card([stepsHeader, detailSteps], \"sunk\")\n\nstepsHeader = CardHeader(\"Step-by-step details\", \"Exact clicks for each item\")\n\ndetailSteps = Steps([s1, s2, s3, s4, s5, s6, s7, s8])\n\ns1 = StepsItem(\"Activate your account\", \"Open id.company.com/activate on the laptop IT handed you. Enter your employee ID and the one-time code from your welcome email. Choose a password of at least 14 characters. This creates your single sign-on identity used by every other tool.\")\ns2 = StepsItem(\"Password manager\", \"Launch Self Service (macOS) or Company Portal (Windows) and install 1Password. Sign in with 'Sign in with SSO' rather than a separate password. Accept the Team vault invitation, then move any credentials out of your browser's saved passwords.\")\ns3 = StepsItem(\"Two-factor authentication\", \"Visit id.company.com/security. Add Okta Verify by scanning the QR code with the app on your phone. Add a second factor as backup — a YubiKey from your welcome kit is preferred over SMS. Download the ten recovery codes and store them in your 1Password Private vault.\")\ns4 = StepsItem(\"VPN\", \"Install GlobalProtect from Self Service. Enter portal address vpn.company.com, sign in with SSO, and approve the 2FA push. Once connected, open wiki.internal in your browser to confirm internal routing works. Disconnect and reconnect once to verify it reconnects cleanly.\")\ns5 = StepsItem(\"Laptop encryption check\", \"macOS: System Settings > Privacy & Security > FileVault should read 'FileVault is on'. Windows: Settings > Privacy & security > Device encryption should read 'On'. If either shows Off, do not continue — post in #it-helpdesk immediately, since an unencrypted laptop cannot be issued production access.\")\ns6 = StepsItem(\"Device management check-in\", \"Open Company Portal and press 'Check settings'. Your device should report Compliant. Pending items usually mean an OS update is required — install it now and restart, because compliance gates access to email and internal apps.\")\ns7 = StepsItem(\"Team channels\", \"Sign into Slack with SSO. Join #general and #it-helpdesk, then ask your onboarding buddy for your squad channel and the department announcements channel. Set your display name to 'First Last (Team)' and upload a photo so people can recognize you.\")\ns8 = StepsItem(\"Email and calendar\", \"Open your mail client and confirm messages are syncing both ways. Check that your onboarding meetings for this week appear on your calendar. Paste the standard signature template from the wiki and replace the placeholders with your name, title, and pronouns.\")\n\nhelpTabs = Tabs([tabTiming, tabTrouble, tabWho])\n\ntabTiming = TabItem(\"timing\", \"Time estimates\", [timingTable])\n\ntimingTable = Table([colTask, colTime, colBlocking])\n\ncolTask = Col(\"Step\", [\"Account activation\", \"Password manager\", \"2FA enrollment\", \"VPN\", \"Encryption check\", \"Device compliance\", \"Team channels\", \"Email & calendar\"])\ncolTime = Col(\"Minutes\", [5, 6, 8, 7, 3, 6, 5, 5], \"number\")\ncolBlocking = Col(\"Blocks later steps\", [blockYes, blockNo, blockYes, blockNo, blockYes, blockYes, blockNo, blockNo])\n\nblockYes = Tag(\"Yes\", null, \"sm\", \"warning\")\nblockNo = Tag(\"No\", null, \"sm\", \"neutral\")\n\ntabTrouble = TabItem(\"trouble\", \"Troubleshooting\", [troubleAccordion])\n\ntroubleAccordion = Accordion([tr1, tr2, tr3, tr4])\n\ntr1 = AccordionItem(\"t1\", \"Activation code says expired\", [tr1Text])\ntr1Text = TextContent(\"Codes expire 72 hours after they are issued. Post your employee ID (never your password) in #it-helpdesk and a new code is normally sent within 15 minutes.\")\n\ntr2 = AccordionItem(\"t2\", \"2FA push never arrives\", [tr2Text])\ntr2Text = TextContent(\"Check that notifications are enabled for Okta Verify and that your phone has a data connection. If it still fails, use the 6-digit code shown inside the app instead of the push prompt.\")\n\ntr3 = AccordionItem(\"t3\", \"VPN connects but internal sites fail\", [tr3Text])\ntr3Text = TextContent(\"This is almost always a stale DNS cache. Disconnect the VPN, flush DNS, reconnect, then retry wiki.internal. If it persists, note the gateway name shown in GlobalProtect when you report it.\")\n\ntr4 = AccordionItem(\"t4\", \"Device shows as Not Compliant\", [tr4Text])\ntr4Text = TextContent(\"Open Company Portal and read the listed reason. The two common causes are a pending OS update and a missing screen-lock timeout of 5 minutes or less. Fix the reason, restart, then press Check settings again.\")\n\ntabWho = TabItem(\"who\", \"Who to ask\", [contactTable, contactNote])\n\ncontactTable = Table([colTopic, colChannel, colSla])\ncolTopic = Col(\"Topic\", [\"Accounts & passwords\", \"Hardware & encryption\", \"Network & VPN\", \"Slack & access requests\", \"Anything urgent\"])\ncolChannel = Col(\"Where to ask\", [\"#it-helpdesk\", \"#it-hardware\", \"#it-network\", \"#it-access\", \"Call the IT desk: x4400\"])\ncolSla = Col(\"Typical response\", [\"15 min\", \"1 hour\", \"30 min\", \"2 hours\", \"Immediate\"])\n\ncontactNote = TextContent(\"Never share a password or recovery code in a channel or DM, even with IT. Support will always ask you to reset instead.\", \"small\")\n\nfooterButtons = Buttons([btnDone, btnStuck, btnPolicy])\n\nbtnDone = Button(\"I finished all eight steps\", Action([@ToAssistant(\"I finished all eight first-day IT onboarding steps. What should I set up next?\")]), \"primary\")\nbtnStuck = Button(\"I'm stuck on a step\", Action([@ToAssistant(\"I'm stuck on one of the first-day IT onboarding steps and need help troubleshooting.\")]), \"secondary\")\nbtnPolicy = Button(\"Read the security policy\", Action([@OpenUrl(\"https://wiki.internal/security-policy\")]), \"tertiary\")","promptTokens":7002,"completionTokens":3079,"latencyMs":39323,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/realistic","repeat":2,"output":"root = Stack([pageHeader, welcomeNote, onboardingTabs, footerNote, helpButtons], \"column\", \"l\")\n\npageHeader = CardHeader(\"Day One: IT Onboarding\", \"Eight steps to get your accounts, device, and channels ready. Tick each one off as you go.\")\n\nwelcomeNote = Callout(\"info\", \"Budget about 45 minutes\", \"Do the steps in order — account activation unlocks everything else. Your IT buddy is on standby in #it-helpdesk if anything stalls.\")\n\nonboardingTabs = Tabs([tabChecklist, tabGuide, tabHelp])\n\ntabChecklist = TabItem(\"checklist\", \"Checklist\", [checklistForm])\ntabGuide = TabItem(\"guide\", \"Step details\", [guideIntro, guideSteps])\ntabHelp = TabItem(\"help\", \"Who to ask\", [contactTable, escalationNote])\n\nchecklistForm = Form(\"day-one-it\", checklistButtons, [checklistControl])\n\nchecklistControl = FormControl(\"Your first-day IT tasks\", checklistGroup, \"Progress saves as you tick. Anything unchecked at the end of the day rolls into tomorrow's plan.\")\n\nchecklistGroup = CheckBoxGroup(\"tasks\", checklistItems)\n\nchecklistItems = [itemAccount, itemPassword, item2fa, itemVpn, itemEncryption, itemChannels, itemMail, itemTraining]\n\nitemAccount = CheckBoxItem(\"1. Activate your SSO account\", \"Open the activation email from identity@company.com, set your primary password, and sign in to the SSO portal once to confirm it works.\", \"account\")\nitemPassword = CheckBoxItem(\"2. Set up the password manager\", \"Install 1Password from the company app catalog, join the team vault with the invite link, and move your SSO password into it.\", \"password\")\nitem2fa = CheckBoxItem(\"3. Enrol in 2FA\", \"Register an authenticator app as your primary factor, add a backup phone, then store the 10 recovery codes in your password manager.\", \"twofa\")\nitemVpn = CheckBoxItem(\"4. Install and test the VPN\", \"Install the VPN client, sign in with SSO, connect to the nearest gateway, and load the internal wiki to confirm the tunnel works.\", \"vpn\")\nitemEncryption = CheckBoxItem(\"5. Verify laptop disk encryption\", \"Run the Device Check app — it should report FileVault or BitLocker as ON and the recovery key as escrowed with IT.\", \"encryption\")\nitemChannels = CheckBoxItem(\"6. Join your team channels\", \"Accept the Slack invite and join #team-announce, your squad channel, #it-helpdesk, and #new-hires.\", \"channels\")\nitemMail = CheckBoxItem(\"7. Configure mail and calendar\", \"Apply the standard email signature template and set your working hours plus timezone in the shared calendar.\", \"mail\")\nitemTraining = CheckBoxItem(\"8. Finish security awareness training\", \"Complete the 20-minute Security 101 module in the learning portal — required before you get production access.\", \"training\")\n\nchecklistButtons = Buttons([submitBtn, stuckBtn])\nsubmitBtn = Button(\"Submit completed steps\", Action([@ToAssistant(\"I have ticked off my completed first-day IT steps — please confirm what is left.\")]), \"primary\")\nstuckBtn = Button(\"I'm stuck on a step\", Action([@ToAssistant(\"I'm stuck on one of my first-day IT onboarding steps and need help troubleshooting.\")]), \"secondary\")\n\nguideIntro = TextContent(\"Expanded instructions for each checklist item, in the order they should be done.\", \"small\")\n\nguideSteps = Steps([stepAccount, stepPassword, step2fa, stepVpn, stepEncryption, stepChannels, stepMail, stepTraining])\n\nstepAccount = StepsItem(\"Activate your SSO account\", \"Check your personal inbox for the activation link (valid 72 hours). Set a password of 14+ characters, then sign in at sso.company.com. Everything else depends on this step, so do it first.\")\nstepPassword = StepsItem(\"Set up the password manager\", \"Open the company app catalog and install 1Password. Use the vault invite in your welcome email to join your team vault. Save your SSO credentials there — never in a browser or notes app.\")\nstep2fa = StepsItem(\"Enrol in two-factor authentication\", \"In the SSO portal go to Security → Two-factor. Scan the QR code with your authenticator app, confirm the 6-digit code, add a backup phone number, then download the recovery codes and paste them into your password manager.\")\nstepVpn = StepsItem(\"Install and test the VPN\", \"Install the VPN client from the app catalog and sign in with SSO. Connect to your nearest gateway and open wiki.internal — if it loads, you're through. If it hangs, disconnect from any guest Wi-Fi captive portal and retry.\")\nstepEncryption = StepsItem(\"Verify laptop disk encryption\", \"Launch Device Check from your dock. Confirm: encryption ON, recovery key escrowed, OS patch level current, and endpoint agent reporting. Screenshot the green summary and post it in #it-helpdesk to close out the device audit.\")\nstepChannels = StepsItem(\"Get added to the right channels\", \"Accept the Slack workspace invite, then join #team-announce, #it-helpdesk, #new-hires, and your squad channel. Ask your manager to add you to the on-call rota channel only once you're past week one.\")\nstepMail = StepsItem(\"Configure mail and calendar\", \"Paste the signature template from the wiki into Mail → Signatures. In Calendar, set your working hours, timezone, and default 25/50-minute meeting lengths so invites land sensibly.\")\nstepTraining = StepsItem(\"Complete security awareness training\", \"Open the learning portal and finish the Security 101 module plus the short quiz. Production and customer-data access is unlocked automatically within an hour of passing.\")\n\ncontactTable = Table([colTopic, colContact, colChannel])\ncolTopic = Col(\"If you're stuck on\", [\"Account activation\", \"Password manager\", \"2FA / recovery codes\", \"VPN\", \"Disk encryption\", \"Channel access\", \"Training portal\"])\ncolContact = Col(\"Team\", [\"Identity team\", \"IT helpdesk\", \"Identity team\", \"Network ops\", \"Endpoint team\", \"Your manager\", \"People ops\"])\ncolChannel = Col(\"Where to ask\", [\"#it-identity\", \"#it-helpdesk\", \"#it-identity\", \"#it-network\", \"#it-endpoint\", \"Direct message\", \"#people-ops\"])\n\nescalationNote = TextCallout(\"warning\", \"Blocked for more than 30 minutes?\", \"Post in #it-helpdesk with the step number and a screenshot of the error. Tickets raised on day one are picked up within 15 minutes during business hours.\")\n\nfooterNote = TextContent(\"Once all eight boxes are ticked, your laptop is compliant and your accounts are fully provisioned — you're clear to start on real work.\", \"small\")\n\nhelpButtons = Buttons([explainBtn, remindBtn])\nexplainBtn = Button(\"Explain 2FA recovery codes\", Action([@ToAssistant(\"Explain what 2FA recovery codes are and how I should store them safely.\")]), \"tertiary\")\nremindBtn = Button(\"What happens on day two?\", Action([@ToAssistant(\"What does the day two onboarding plan look like after finishing the IT checklist?\")]), \"tertiary\")","promptTokens":7002,"completionTokens":3147,"latencyMs":41559,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/realistic","repeat":4,"output":"root = Stack([header, progressCallout, checklistCard, stepsCard, helpCard, footerButtons])\n\nheader = CardHeader(\"First-Day IT Onboarding\", \"Work through these eight steps in order — most people finish in about 45 minutes.\")\n\nprogressCallout = Callout(\"info\", \"Before you start\", \"Have your welcome email and your personal phone nearby. You'll need both for account activation and 2FA enrollment.\")\n\nchecklistCard = Card([checklistHeader, checklist, checklistNote])\n\nchecklistHeader = CardHeader(\"Your Checklist\", \"Tick each item as you complete it.\")\n\nchecklist = SwitchGroup(\"onboarding\", checklistItems, \"sunk\")\n\nchecklistItems = [item1, item2, item3, item4, item5, item6, item7, item8]\n\nitem1 = SwitchItem(\"1. Activate your account\", \"Claim your username and set your first password at id.company.com/activate\", \"activate\")\nitem2 = SwitchItem(\"2. Install the password manager\", \"Install 1Password from Self Service and join the company vault\", \"pwmgr\")\nitem3 = SwitchItem(\"3. Enroll in 2FA\", \"Register Okta Verify plus one backup method and save your recovery codes\", \"twofa\")\nitem4 = SwitchItem(\"4. Replace your temporary password\", \"Generate a 20-character password in 1Password and change it in Okta\", \"rotate\")\nitem5 = SwitchItem(\"5. Set up VPN access\", \"Install GlobalProtect, sign in with SSO, and confirm you reach the intranet\", \"vpn\")\nitem6 = SwitchItem(\"6. Verify laptop encryption\", \"Confirm FileVault or BitLocker is on and the recovery key is escrowed to IT\", \"encryption\")\nitem7 = SwitchItem(\"7. Join your team channels\", \"Request channel access in Slack via the #it-access workflow\", \"channels\")\nitem8 = SwitchItem(\"8. Confirm mail and calendar\", \"Send a test message and accept your first-week meeting invites\", \"mail\")\n\nchecklistNote = TextContent(\"Steps 1–4 must be done in order. Steps 5–8 can be done in any order once 2FA is active.\", \"small\")\n\nstepsCard = Card([stepsHeader, detailTabs])\n\nstepsHeader = CardHeader(\"Step Details\", \"Expand any step for exact instructions.\")\n\ndetailTabs = Tabs([tabAccounts, tabNetwork, tabTeam])\n\ntabAccounts = TabItem(\"accounts\", \"Accounts & Security\", [accountsSteps, accountsWarning])\n\naccountsSteps = Steps([s1, s2, s3, s4])\n\ns1 = StepsItem(\"1. Activate your account\", \"Open the welcome email from it-onboarding@company.com and click Activate. Your username is firstname.lastname. Set the temporary password when prompted, then confirm your profile details (legal name, preferred name, department). If the link has expired, request a new one at id.company.com/resend.\")\ns2 = StepsItem(\"2. Install the password manager\", \"Open Self Service on your laptop and install 1Password. Sign in with your company email, then accept the vault invitation for your team. Save your Emergency Kit PDF to an encrypted folder or print it — IT cannot recover it for you.\")\ns3 = StepsItem(\"3. Enroll in 2FA\", \"Go to id.company.com/security and add Okta Verify as your primary factor by scanning the QR code with your phone. Add a second factor (SMS or a hardware key) as backup. Copy the ten recovery codes into a secure 1Password note.\")\ns4 = StepsItem(\"4. Replace your temporary password\", \"In 1Password, generate a 20-character password. In Okta, choose Change Password and paste it in. Save it to 1Password before submitting. Your temporary password expires 24 hours after activation.\")\n\naccountsWarning = TextCallout(\"warning\", \"Do not skip the recovery codes\", \"Roughly a third of first-week lockouts happen because recovery codes were never saved. Storing them takes 30 seconds and saves an hour of identity verification.\")\n\ntabNetwork = TabItem(\"network\", \"Network & Device\", [networkSteps, encryptionCode])\n\nnetworkSteps = Steps([s5, s6])\n\ns5 = StepsItem(\"5. Set up VPN access\", \"Install GlobalProtect from Self Service. Enter the portal address vpn.company.com and sign in with SSO — you will get a 2FA push. Verify success by loading intranet.company.com. Use Always-On mode; only disconnect for troubleshooting.\")\ns6 = StepsItem(\"6. Verify laptop encryption\", \"Confirm full-disk encryption is enabled and that your recovery key has been escrowed to IT. Run the command below and send the output to #it-helpdesk if it does not show the expected status.\")\n\nencryptionCode = CodeBlock(\"bash\", \"# macOS — expect: FileVault is On.\\nfdesetup status\\n\\n# Windows (PowerShell as Administrator)\\n# expect ProtectionStatus: On for volume C:\\nmanage-bde -status C:\")\n\ntabTeam = TabItem(\"team\", \"Access & Comms\", [teamSteps, channelTable])\n\nteamSteps = Steps([s7, s8])\n\ns7 = StepsItem(\"7. Join your team channels\", \"Open Slack, run the /it-access workflow, and pick your team from the list. Access is granted automatically for standard channels; anything marked restricted goes to your manager for approval and usually clears within two hours.\")\ns8 = StepsItem(\"8. Confirm mail and calendar\", \"Send a hello message in your team channel and email your buddy to confirm delivery. Then open your calendar and accept the orientation, security training, and manager 1:1 invites already waiting for you.\")\n\nchannelTable = Table([colChannel, colPurpose, colAccess])\n\ncolChannel = Col(\"Channel\", [\"#general\", \"#it-helpdesk\", \"#your-team\", \"#eng-releases\", \"#security-alerts\"])\ncolPurpose = Col(\"Purpose\", [\"Company-wide announcements\", \"Break/fix and access requests\", \"Daily standup and team chatter\", \"Deploy and release notifications\", \"Mandatory security notices\"])\ncolAccess = Col(\"Access\", accessTags)\n\naccessTags = [tagAuto1, tagAuto2, tagAuto3, tagApproval, tagAuto4]\n\ntagAuto1 = Tag(\"Automatic\", null, \"sm\", \"success\")\ntagAuto2 = Tag(\"Automatic\", null, \"sm\", \"success\")\ntagAuto3 = Tag(\"Automatic\", null, \"sm\", \"success\")\ntagApproval = Tag(\"Manager approval\", null, \"sm\", \"warning\")\ntagAuto4 = Tag(\"Automatic\", null, \"sm\", \"success\")\n\nhelpCard = Card([helpHeader, helpAccordion], \"sunk\")\n\nhelpHeader = CardHeader(\"If Something Goes Wrong\", \"Common first-day snags and the fastest fix.\")\n\nhelpAccordion = Accordion([faq1, faq2, faq3])\n\nfaq1 = AccordionItem(\"locked\", \"I'm locked out of my account\", [faq1Text])\nfaq1Text = TextContent(\"Use a recovery code at id.company.com/recover. If you have no codes, call the IT desk at extension 4400 — they will verify your identity over video with your manager on the call. Expect about 20 minutes.\")\n\nfaq2 = AccordionItem(\"vpn\", \"VPN connects but nothing loads\", [faq2Text])\nfaq2Text = TextContent(\"This is almost always DNS. Disconnect, switch off any personal VPN or ad-blocking DNS, then reconnect. If the intranet still fails, post your GlobalProtect log in #it-helpdesk with your laptop asset tag.\")\n\nfaq3 = AccordionItem(\"encryption\", \"Encryption shows as Off or Pending\", [faq3Text])\nfaq3Text = TextContent(\"Do not enable it manually — the recovery key must be escrowed by the management agent or IT cannot help you recover data. Open a ticket tagged 'encryption' and IT will push the profile remotely, usually within the hour.\")\n\nfooterButtons = Buttons([btnDone, btnHelp, btnPortal])\n\nbtnDone = Button(\"I've finished all eight steps\", Action([@ToAssistant(\"I've completed all eight first-day IT onboarding steps. What should I set up next?\")]), \"primary\")\nbtnHelp = Button(\"I'm stuck on a step\", Action([@ToAssistant(\"I'm stuck on one of the first-day IT onboarding steps and need help troubleshooting.\")]), \"secondary\")\nbtnPortal = Button(\"Open IT portal\", Action([@OpenUrl(\"https://it.company.com/onboarding\")]), \"tertiary\")","promptTokens":7002,"completionTokens":2933,"latencyMs":37720,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/minimal","repeat":0,"output":"root = Stack([contactCard])\ncontactCard = Card([header, contactForm])\nheader = CardHeader(\"Contact us\", \"Send us a message and we’ll get back to you soon.\")\ncontactForm = Form(\"contact\", formButtons, [nameField, emailField, messageField])\nformButtons = Buttons([sendButton])\nsendButton = Button(\"Send message\", Action([@ToAssistant(\"Submit the contact form\")]), \"primary\")\nnameField = FormControl(\"Name\", nameInput)\nemailField = FormControl(\"Email\", emailInput)\nmessageField = FormControl(\"Message\", messageInput)\nnameInput = Input(\"name\", \"Your name\", \"text\", {required: true})\nemailInput = Input(\"email\", \"you@example.com\", \"email\", {required: true, email: true})\nmessageInput = TextArea(\"message\", \"How can we help?\", 5, {required: true})","promptTokens":4389,"completionTokens":192,"latencyMs":2733,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/minimal","repeat":1,"output":"root = Stack([contactCard])\ncontactCard = Card([header, contactForm], \"column\", \"l\")\nheader = CardHeader(\"Contact us\", \"Send us a message and we’ll get back to you soon.\")\ncontactForm = Form(\"contact\", formButtons, [nameField, emailField, messageField])\nformButtons = Buttons([sendButton])\nsendButton = Button(\"Send message\", Action([@ToAssistant(\"Submit the contact form\")]), \"primary\")\nnameField = FormControl(\"Name\", Input(\"name\", \"Your full name\", \"text\", {required: true}))\nemailField = FormControl(\"Email\", Input(\"email\", \"you@example.com\", \"email\", {required: true, email: true}))\nmessageField = FormControl(\"Message\", TextArea(\"message\", \"How can we help?\", 5, {required: true, minLength: 10}))","promptTokens":4389,"completionTokens":187,"latencyMs":2264,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/minimal","repeat":2,"output":"root = Stack([contactCard])\ncontactCard = Card([header, contactForm])\nheader = CardHeader(\"Contact us\", \"Send us a message and we’ll get back to you.\")\ncontactForm = Form(\"contact\", formButtons, [nameField, emailField, messageField])\nnameField = FormControl(\"Name\", Input(\"name\", \"Your name\", \"text\", {required: true}))\nemailField = FormControl(\"Email\", Input(\"email\", \"you@example.com\", \"email\", {required: true, email: true}))\nmessageField = FormControl(\"Message\", TextArea(\"message\", \"How can we help?\", 5, {required: true}))\nsubmitButton = Button(\"Send message\", Action([@ToAssistant(\"Contact form submitted\")]), \"primary\")\nformButtons = Buttons([submitButton])","promptTokens":4389,"completionTokens":172,"latencyMs":11227,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/minimal","repeat":3,"output":"root = Stack([contactCard])\ncontactCard = Card([header, contactForm])\nheader = CardHeader(\"Contact us\", \"Send us a message and we’ll get back to you.\")\ncontactForm = Form(\"contact\", formButtons, [nameField, emailField, messageField])\nnameField = FormControl(\"Name\", nameInput)\nemailField = FormControl(\"Email\", emailInput)\nmessageField = FormControl(\"Message\", messageInput)\nnameInput = Input(\"name\", \"Your name\", \"text\", {required: true})\nemailInput = Input(\"email\", \"you@example.com\", \"email\", {required: true, email: true})\nmessageInput = TextArea(\"message\", \"How can we help?\", 5, {required: true})\nformButtons = Buttons([sendButton])\nsendButton = Button(\"Send message\", Action([@ToAssistant(\"Submit the contact form\")]), \"primary\")","promptTokens":4389,"completionTokens":191,"latencyMs":4734,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/minimal","repeat":4,"output":"root = Stack([contactCard])\ncontactCard = Card([header, contactForm])\nheader = CardHeader(\"Contact us\", \"Send us a message and we’ll get back to you.\")\ncontactForm = Form(\"contact\", formButtons, [nameField, emailField, messageField])\nnameField = FormControl(\"Name\", Input(\"name\", \"Your name\", \"text\", {required: true}, $name))\nemailField = FormControl(\"Email\", Input(\"email\", \"you@example.com\", \"email\", {required: true, email: true}, $email))\nmessageField = FormControl(\"Message\", TextArea(\"message\", \"How can we help?\", 5, {required: true}, $message))\nformButtons = Buttons([sendButton])\nsendButton = Button(\"Send message\", Action([@ToAssistant(\"Contact form submitted\")]), \"primary\")","promptTokens":4389,"completionTokens":181,"latencyMs":2368,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/adversarial","repeat":0,"output":"root = Stack([header, intro, overviewCard, sep, detailCard, closingCallout, actions], \"column\", \"l\")\n\nheader = CardHeader(\"Deployment Runbook\", \"Seven steps from local commit to verified production release\")\n\nintro = TextContent(\"Below is a standard deploy pipeline for a web service. The **Overview** gives you the sequence at a glance; expand any step in **Step Details** for the rationale, the failure mode if you skip it, and the exact commands.\")\n\noverviewCard = Card([overviewHeader, overviewSteps])\noverviewHeader = CardHeader(\"Overview\", \"The happy path, in order\")\noverviewSteps = Steps([s1, s2, s3, s4, s5, s6, s7])\n\ns1 = StepsItem(\"1. Freeze & sync the branch\", \"Pull latest main, confirm a clean working tree, and tag the exact commit you intend to ship.\")\ns2 = StepsItem(\"2. Install locked dependencies\", \"Install from the lockfile so the build is reproducible and identical to CI.\")\ns3 = StepsItem(\"3. Run tests, lint, and type checks\", \"Fail fast locally and in CI before anything is packaged.\")\ns4 = StepsItem(\"4. Build the production artifact\", \"Compile once, produce an immutable image, and push it to the registry.\")\ns5 = StepsItem(\"5. Run database migrations\", \"Apply backward-compatible schema changes before new code goes live.\")\ns6 = StepsItem(\"6. Deploy to staging, then production\", \"Smoke test staging, then roll out gradually with health checks.\")\ns7 = StepsItem(\"7. Verify, monitor, and be ready to roll back\", \"Watch error rate and latency for 15–30 minutes and keep the rollback command in your hand.\")\n\nsep = Separator(\"horizontal\")\n\ndetailCard = Card([detailHeader, detailAccordion])\ndetailHeader = CardHeader(\"Step Details\", \"Why it matters · What breaks if you skip · Exact commands\")\ndetailAccordion = Accordion([a1, a2, a3, a4, a5, a6, a7])\n\na1 = AccordionItem(\"step1\", \"1. Freeze & sync the branch\", [a1why, a1skip, a1cmd])\na1why = TextContent(\"**Why it matters.** A deploy is only meaningful if you know exactly which commit is running. Syncing with `main` guarantees you are shipping what was reviewed and merged, not a stale local snapshot. A git tag turns an opaque hash into a human-readable release identity that logs, dashboards, and incident reviews can all point at. A clean working tree proves no uncommitted local hack is silently baked into your build.\")\na1skip = TextContent(\"**If you skip it.** You ship a stale or divergent tree, silently reverting a teammate's merged fix. Worse, an uncommitted local file makes the artifact unreproducible: nobody can rebuild it, and during an incident you cannot answer \\\"what code is in production?\\\" Rollback becomes guesswork because there is no named point to return to.\")\na1cmd = CodeBlock(\"bash\", \"git checkout main\\ngit pull --ff-only origin main\\ngit status --porcelain # must print nothing\\ngit log -1 --oneline\\n\\n# Tag the release commit\\ngit tag -a v1.8.0 -m \\\"Release v1.8.0\\\"\\ngit push origin v1.8.0\")\n\na2 = AccordionItem(\"step2\", \"2. Install locked dependencies\", [a2why, a2skip, a2cmd])\na2why = TextContent(\"**Why it matters.** The lockfile pins every transitive dependency to an exact version and hash. Installing from it makes your machine, CI, and the production image resolve the identical dependency graph. `npm ci` (unlike `npm install`) deletes `node_modules` first and refuses to mutate the lockfile, so it is deterministic by design.\")\na2skip = TextContent(\"**If you skip it** (or use a loose install): a patch release of a nested dependency slips in between your local test run and the CI build. You get the classic \\\"works on my machine\\\" failure, or a supply-chain surprise reaching production untested. Debugging is brutal because the code is identical and only the dependency tree differs.\")\na2cmd = CodeBlock(\"bash\", \"node --version # match .nvmrc / CI image\\nnpm ci # NOT npm install\\nnpm audit --audit-level=high # fail on known high/critical CVEs\")\n\na3 = AccordionItem(\"step3\", \"3. Run tests, lint, and type checks\", [a3why, a3skip, a3cmd])\na3why = TextContent(\"**Why it matters.** This is the cheapest place to catch a defect. A bug caught by a unit test costs seconds; the same bug caught by a customer costs an incident, a rollback, and trust. Type checks and lint catch a whole class of null/undefined and dead-code errors that tests may never exercise. Running the full suite locally before pushing also keeps CI green for everyone else.\")\na3skip = TextContent(\"**If you skip it.** You promote a red build. Broken code lands in an immutable image, migrations may already have run, and now you are rolling back under pressure instead of fixing a two-line typo calmly. Skipped checks also erode the suite over time — once a failing test is routinely ignored, the whole safety net stops being trusted.\")\na3cmd = CodeBlock(\"bash\", \"npm run lint\\nnpm run typecheck\\nnpm test -- --coverage --runInBand\\nnpm run test:e2e # optional but recommended pre-release\\n\\n# Confirm CI agrees before proceeding\\ngh run list --branch main --limit 1\")\n\na4 = AccordionItem(\"step4\", \"4. Build the production artifact\", [a4why, a4skip, a4cmd, a4note])\na4why = TextContent(\"**Why it matters.** Building once and promoting the same immutable image through every environment is the single biggest reliability win in a pipeline. Tagging the image with the git SHA creates a permanent, verifiable link between running containers and source code. Production builds also enable minification, tree-shaking, and `NODE_ENV=production`, which strip dev-only warnings and can cut bundle size and memory use dramatically.\")\na4skip = TextContent(\"**If you skip it** — that is, if you rebuild separately per environment — staging and production run subtly different bytes, so staging validation proves nothing. If you deploy a dev build, you ship source maps, verbose logging, and debug endpoints, leaking internals and burning CPU. Using only a `latest` tag makes rollback impossible: there is no previous immutable reference to return to.\")\na4cmd = CodeBlock(\"bash\", \"export SHA=$(git rev-parse --short HEAD)\\nexport IMAGE=registry.example.com/api\\n\\nnpm run build # NODE_ENV=production\\n\\ndocker build \\\\\\n --build-arg NODE_ENV=production \\\\\\n -t $IMAGE:$SHA \\\\\\n -t $IMAGE:v1.8.0 .\\n\\ndocker push $IMAGE:$SHA\\ndocker push $IMAGE:v1.8.0\")\na4note = TextCallout(\"info\", \"Never deploy a floating tag\", \"Deploy the immutable $SHA tag. Reserve `latest` for convenience only — it makes rollbacks ambiguous.\")\n\na5 = AccordionItem(\"step5\", \"5. Run database migrations\", [a5why, a5skip, a5cmd, a5note])\na5why = TextContent(\"**Why it matters.** Code and schema must be compatible at every instant during a rolling deploy, when old and new pods serve traffic simultaneously. Running additive, backward-compatible migrations *before* the new code means the old version still works and the new version finds the columns it expects. Taking a backup first gives you a recovery path that does not depend on the migration tool behaving well.\")\na5skip = TextContent(\"**If you skip it or reorder it.** New code queries a column that does not exist yet and every request 500s the moment the first pod goes healthy. If you instead run a destructive migration (dropping or renaming a column) before old pods drain, the *old* version starts erroring. Skipping the backup turns a bad migration from a ten-minute restore into permanent data loss.\")\na5cmd = CodeBlock(\"bash\", \"# 1. Snapshot first — always\\npg_dump \\\"$DATABASE_URL\\\" | gzip > backup-$(date +%F-%H%M).sql.gz\\n\\n# 2. Review what will run\\nnpm run migrate:status\\nnpm run migrate:dry-run\\n\\n# 3. Apply (additive changes only)\\nnpm run migrate:up\\n\\n# Destructive cleanup happens in a LATER release,\\n# after all old pods are gone:\\n# npm run migrate:up -- --only drop_legacy_column\")\na5note = TextCallout(\"warning\", \"Expand, then contract\", \"Release N adds the new column and dual-writes. Release N+1 stops using the old column. Release N+2 drops it. Never combine these.\")\n\na6 = AccordionItem(\"step6\", \"6. Deploy to staging, then production\", [a6why, a6skip, a6cmd])\na6why = TextContent(\"**Why it matters.** Staging is where the exact production artifact meets real configuration, real network policy, and production-shaped data for the first time. A gradual rollout (canary or 10% → 50% → 100%) limits the blast radius: if the new version is broken, only a small slice of users are affected and automated health checks halt the rollout before it spreads. Readiness probes ensure traffic only reaches pods that are actually able to serve.\")\na6skip = TextContent(\"**If you skip staging.** Config typos, missing secrets, and IAM permission gaps surface directly in production, where they become an outage rather than a five-minute fix. **If you skip the gradual rollout**, a bad release hits 100% of users at once; by the time alerts fire, thousands of requests have already failed and queues are full of poison messages.\")\na6cmd = CodeBlock(\"bash\", \"# --- Staging ---\\nkubectl -n staging set image deploy/api api=$IMAGE:$SHA\\nkubectl -n staging rollout status deploy/api --timeout=5m\\ncurl -fsS https://staging.example.com/healthz\\nnpm run smoke -- --base-url=https://staging.example.com\\n\\n# --- Production (canary, then full) ---\\nkubectl -n prod set image deploy/api-canary api=$IMAGE:$SHA\\nkubectl -n prod rollout status deploy/api-canary --timeout=5m\\n# watch canary metrics ~10 min, then:\\nkubectl -n prod set image deploy/api api=$IMAGE:$SHA\\nkubectl -n prod rollout status deploy/api --timeout=10m\")\n\na7 = AccordionItem(\"step7\", \"7. Verify, monitor, and be ready to roll back\", [a7why, a7skip, a7cmd, a7note])\na7why = TextContent(\"**Why it matters.** A rollout reporting \\\"success\\\" only means containers started, not that the product works. The first 15–30 minutes hold most regressions: latency creep from an unindexed query, an error rate rising as cache warms, a background job failing on its first run. Deciding your rollback trigger *in advance* removes debate during an incident, and announcing the release closes the loop for support and on-call.\")\na7skip = TextContent(\"**If you skip it.** A slow-burning regression runs unnoticed for hours — corrupting data, dropping conversions, or silently failing webhooks — and by the time someone notices, several more deploys have landed on top, making the cause far harder to isolate. Without a rehearsed rollback command, your mean-time-to-recovery is measured in nervous minutes spent reading documentation.\")\na7cmd = CodeBlock(\"bash\", \"# Verify the running version\\ncurl -fsS https://api.example.com/version\\nkubectl -n prod get pods -l app=api\\n\\n# Watch logs and error rate\\nkubectl -n prod logs -l app=api --since=10m -f | grep -i error\\n\\n# One-command rollback if SLOs break\\nkubectl -n prod rollout undo deploy/api\\nkubectl -n prod rollout status deploy/api\\n\\n# Announce\\ngh release create v1.8.0 --generate-notes\")\na7note = TextCallout(\"danger\", \"Define the trigger before you deploy\", \"Example: roll back if 5xx rate > 1% for 5 minutes, or p95 latency > 800ms. No discussion, just execute.\")\n\nclosingCallout = Callout(\"success\", \"Golden rule\", \"Build the artifact once, promote it forward unchanged, migrate the schema in a backward-compatible way, and never deploy without a tested rollback path.\")\n\nactions = Buttons([b1, b2, b3])\nb1 = Button(\"Turn this into a GitHub Actions workflow\", Action([@ToAssistant(\"Write a complete GitHub Actions workflow implementing these seven deploy steps, including the canary rollout and rollback job.\")]), \"primary\")\nb2 = Button(\"Explain expand/contract migrations\", Action([@ToAssistant(\"Explain the expand-and-contract database migration pattern in detail with a concrete column-rename example across three releases.\")]), \"secondary\")\nb3 = Button(\"Add a blue/green variant\", Action([@ToAssistant(\"Show how step 6 would change if we used blue/green deployment instead of a canary rollout, with exact commands.\")]), \"secondary\")","promptTokens":6969,"completionTokens":4755,"latencyMs":63525,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/realistic","repeat":0,"output":"root = Stack([pageHeader, requestCard], \"column\", \"l\")\npageHeader = CardHeader(\"Request an Appointment\", \"Tell us your preferred time and treatment. Our team will contact you to confirm.\")\nrequestCard = Card([appointmentForm], \"card\", \"column\", \"m\")\nappointmentForm = Form(\"appointment-request\", formButtons, [nameField, phoneField, emailField, treatmentField, dateField, notesField])\nnameField = FormControl(\"Full name\", Input(\"fullName\", \"Your full name\", \"text\", {required: true}), \"Required\")\nphoneField = FormControl(\"Phone number\", Input(\"phone\", \"e.g. (555) 123-4567\", \"text\", {required: true}), \"Required\")\nemailField = FormControl(\"Email address\", Input(\"email\", \"you@example.com\", \"email\", {email: true}), \"Optional — we can send your confirmation here.\")\ntreatmentField = FormControl(\"Treatment needed\", Select(\"treatment\", [checkupOption, cleaningOption, whiteningOption, rootCanalOption], \"Select a treatment\", {required: true}), \"Required\")\ndateField = FormControl(\"Preferred date\", DatePicker(\"preferredDate\", \"single\"), \"We will do our best to accommodate your preference.\")\nnotesField = FormControl(\"Anything else we should know?\", TextArea(\"notes\", \"Please share timing preferences, concerns, or other details.\", 4))\nformButtons = Buttons([submitButton])\nsubmitButton = Button(\"Send appointment request\", Action([@ToAssistant(\"Submit this dental appointment request for review.\")]), \"primary\", \"normal\", \"large\")\ncheckupOption = SelectItem(\"checkup\", \"Checkup\")\ncleaningOption = SelectItem(\"cleaning\", \"Cleaning\")\nwhiteningOption = SelectItem(\"whitening\", \"Whitening\")\nrootCanalOption = SelectItem(\"root-canal\", \"Root canal\")","promptTokens":4449,"completionTokens":400,"latencyMs":4352,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/realistic","repeat":1,"output":"root = Stack([pageHeader, requestCard], \"column\", \"l\")\npageHeader = Card([headerTitle, headerText], \"clear\", \"column\", \"xs\")\nheaderTitle = TextContent(\"Request an Appointment\", \"large-heavy\")\nheaderText = TextContent(\"Tell us what you need, and our team will contact you to confirm a time.\", \"default\")\nrequestCard = Card([appointmentForm], \"card\", \"column\", \"m\")\nappointmentForm = Form(\"appointment-request\", formButtons, [nameField, phoneField, emailField, treatmentField, dateField, notesField])\nformButtons = Buttons([submitButton], \"row\")\nsubmitButton = Button(\"Send Appointment Request\", Action([@ToAssistant(\"A patient submitted an appointment request. Please confirm that the clinic will follow up to schedule their visit.\")]), \"primary\", \"normal\", \"large\")\nnameField = FormControl(\"Full name\", nameInput, \"Required\")\nphoneField = FormControl(\"Phone number\", phoneInput, \"Required\")\nemailField = FormControl(\"Email address\", emailInput, \"We will use this to send appointment details.\")\ntreatmentField = FormControl(\"Treatment requested\", treatmentSelect, \"Required\")\ndateField = FormControl(\"Preferred date\", preferredDate, \"We will do our best to accommodate your preference.\")\nnotesField = FormControl(\"Anything else we should know?\", notesInput, \"Optional\")\nnameInput = Input(\"fullName\", \"Enter your full name\", \"text\", {required: true})\nphoneInput = Input(\"phone\", \"Enter your phone number\", \"text\", {required: true})\nemailInput = Input(\"email\", \"Enter your email address\", \"email\", {email: true})\ntreatmentSelect = Select(\"treatment\", [checkupOption, cleaningOption, whiteningOption, rootCanalOption], \"Choose a treatment\", {required: true})\ncheckupOption = SelectItem(\"checkup\", \"Checkup\")\ncleaningOption = SelectItem(\"cleaning\", \"Cleaning\")\nwhiteningOption = SelectItem(\"whitening\", \"Whitening\")\nrootCanalOption = SelectItem(\"root-canal\", \"Root canal\")\npreferredDate = DatePicker(\"preferredDate\", \"single\")\nnotesInput = TextArea(\"notes\", \"Share any symptoms, concerns, accessibility needs, or scheduling preferences.\", 5)","promptTokens":4449,"completionTokens":485,"latencyMs":5091,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/realistic","repeat":2,"output":"root = Stack([pageHeader, appointmentCard], \"column\", \"l\")\npageHeader = Card([headerTitle, headerText], \"clear\", \"column\", \"xs\")\nheaderTitle = TextContent(\"Request a Dental Appointment\", \"large-heavy\")\nheaderText = TextContent(\"Tell us your preferred appointment details. Our team will contact you to confirm availability.\", \"default\")\nappointmentCard = Card([appointmentForm], \"card\", \"column\", \"m\")\nappointmentForm = Form(\"appointment-request\", formButtons, [nameField, phoneField, emailField, treatmentField, dateField, notesField])\nnameField = FormControl(\"Full name\", Input(\"fullName\", \"Enter your full name\", \"text\", {required: true}))\nphoneField = FormControl(\"Phone number\", Input(\"phone\", \"Enter your phone number\", \"text\", {required: true}))\nemailField = FormControl(\"Email address\", Input(\"email\", \"Enter your email address\", \"email\", {email: true}))\ntreatmentField = FormControl(\"Treatment requested\", Select(\"treatment\", [checkupOption, cleaningOption, whiteningOption, rootCanalOption], \"Choose a treatment\", {required: true}))\ndateField = FormControl(\"Preferred date\", DatePicker(\"preferredDate\", \"single\"))\nnotesField = FormControl(\"Anything else we should know?\", TextArea(\"notes\", \"Share symptoms, accessibility needs, preferred times, or other details\", 4))\nformButtons = Buttons([submitButton, resetButton])\nsubmitButton = Button(\"Send appointment request\", Action([@ToAssistant(\"Submit this dental appointment request.\")]), \"primary\")\nresetButton = Button(\"Clear form\", Action([@ToAssistant(\"Clear the appointment request form.\")]), \"tertiary\")\ncheckupOption = SelectItem(\"checkup\", \"Checkup\")\ncleaningOption = SelectItem(\"cleaning\", \"Cleaning\")\nwhiteningOption = SelectItem(\"whitening\", \"Whitening\")\nrootCanalOption = SelectItem(\"root-canal\", \"Root canal\")","promptTokens":4449,"completionTokens":421,"latencyMs":5185,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/realistic","repeat":3,"output":"root = Stack([page])\npage = Card([header, intro, appointmentForm], \"column\", \"l\")\nheader = CardHeader(\"Request an Appointment\", \"Tell us your preferred time and treatment. Our team will contact you to confirm.\")\nintro = Callout(\"info\", \"We’ll be in touch\", \"Submitting this request does not confirm an appointment. We’ll call or email you to arrange a time.\")\nappointmentForm = Form(\"appointment-request\", formButtons, [nameField, phoneField, emailField, treatmentField, dateField, notesField])\nnameField = FormControl(\"Full name\", nameInput, \"Required\")\nphoneField = FormControl(\"Phone number\", phoneInput, \"Required\")\nemailField = FormControl(\"Email address\", emailInput, \"Optional — useful for appointment confirmation.\")\ntreatmentField = FormControl(\"Treatment requested\", treatmentSelect, \"Required\")\ndateField = FormControl(\"Preferred date\", datePicker, \"We’ll do our best to accommodate your preference.\")\nnotesField = FormControl(\"Anything else we should know?\", notesInput, \"Optional — include concerns, accessibility needs, or preferred contact time.\")\nnameInput = Input(\"fullName\", \"Your full name\", \"text\", {required: true})\nphoneInput = Input(\"phone\", \"e.g. (555) 123-4567\", \"text\", {required: true})\nemailInput = Input(\"email\", \"you@example.com\", \"email\", {email: true})\ntreatmentSelect = Select(\"treatment\", [checkupOption, cleaningOption, whiteningOption, rootCanalOption], \"Choose a treatment\", {required: true})\ncheckupOption = SelectItem(\"checkup\", \"Checkup\")\ncleaningOption = SelectItem(\"cleaning\", \"Cleaning\")\nwhiteningOption = SelectItem(\"whitening\", \"Teeth whitening\")\nrootCanalOption = SelectItem(\"root-canal\", \"Root canal\")\ndatePicker = DatePicker(\"preferredDate\", \"single\")\nnotesInput = TextArea(\"notes\", \"Share any details that may help us prepare for your visit\", 5)\nformButtons = Buttons([submitButton, resetButton])\nsubmitButton = Button(\"Send appointment request\", Action([@ToAssistant(\"Please submit my dental appointment request.\")]), \"primary\", \"normal\", \"large\")\nresetButton = Button(\"Clear form\", Action([@ToAssistant(\"Please clear my dental appointment request form.\")]), \"tertiary\")","promptTokens":4449,"completionTokens":505,"latencyMs":9004,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/realistic","repeat":4,"output":"root = Stack([pageHeader, requestCard], \"column\", \"l\")\npageHeader = Card([title, intro], \"clear\", \"column\", \"s\")\ntitle = TextContent(\"Request a Dental Appointment\", \"large-heavy\")\nintro = TextContent(\"Tell us what you need and your preferred date. Our team will contact you to confirm your visit.\", \"default\")\nrequestCard = Card([appointmentForm], \"card\", \"column\", \"m\")\nappointmentForm = Form(\"appointment-request\", formButtons, [nameField, phoneField, emailField, treatmentField, dateField, notesField])\nnameField = FormControl(\"Full name\", Input(\"fullName\", \"Enter your full name\", \"text\", {required: true}, $fullName), \"Required\")\nphoneField = FormControl(\"Phone number\", Input(\"phone\", \"Enter your phone number\", \"text\", {required: true}, $phone), \"Required\")\nemailField = FormControl(\"Email address\", Input(\"email\", \"you@example.com\", \"email\", {email: true}, $email), \"Optional\")\ntreatmentField = FormControl(\"Treatment requested\", Select(\"treatment\", [checkupOption, cleaningOption, whiteningOption, rootCanalOption], \"Choose a treatment\", {required: true}, $treatment), \"Required\")\ncheckupOption = SelectItem(\"checkup\", \"Checkup\")\ncleaningOption = SelectItem(\"cleaning\", \"Cleaning\")\nwhiteningOption = SelectItem(\"whitening\", \"Whitening\")\nrootCanalOption = SelectItem(\"root-canal\", \"Root canal\")\ndateField = FormControl(\"Preferred date\", DatePicker(\"preferredDate\", \"single\", null, $preferredDate), \"Optional — we will confirm availability.\")\nnotesField = FormControl(\"Anything else we should know?\", TextArea(\"notes\", \"Share symptoms, accessibility needs, preferred times, or other details\", 4, null, $notes), \"Optional\")\nformButtons = Buttons([submitButton, resetButton])\nsubmitButton = Button(\"Request appointment\", Action([@ToAssistant(\"Submit appointment request\")]), \"primary\", \"normal\", \"medium\")\nresetButton = Button(\"Clear form\", Action([@ToAssistant(\"Clear appointment request form\")]), \"secondary\", \"normal\", \"medium\")","promptTokens":4449,"completionTokens":474,"latencyMs":8748,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/adversarial","repeat":0,"output":"root = Stack([header, feedbackCard], \"column\", \"m\", \"stretch\")\nheader = Stack([title, subtitle], \"column\", \"xs\")\ntitle = TextContent(\"How was your experience?\", \"large-heavy\")\nsubtitle = TextContent(\"Your feedback helps us improve. This takes less than a minute.\", \"small\")\nfeedbackCard = Card([feedbackForm], \"column\", \"m\")\nfeedbackForm = Form(\"feedback\", formButtons, [ratingField, issueField, contactField])\nratingField = FormControl(\"How happy were you?\", ratingInput)\nratingInput = RadioGroup(\"rating\", [ratingGreat, ratingGood, ratingOkay, ratingPoor], \"good\")\nratingGreat = RadioItem(\"Great\", \"Everything went well\", \"great\")\nratingGood = RadioItem(\"Good\", \"Mostly positive\", \"good\")\nratingOkay = RadioItem(\"Okay\", \"Could be better\", \"okay\")\nratingPoor = RadioItem(\"Not good\", \"Something went wrong\", \"poor\")\nissueField = FormControl(\"What could we improve?\", issueInput, \"Optional — a few words is enough.\")\nissueInput = TextArea(\"feedback\", \"Tell us what happened...\", 3)\ncontactField = FormControl(\"Email for a reply\", contactInput, \"Optional — only if you would like us to follow up.\")\ncontactInput = Input(\"email\", \"you@example.com\", \"email\", {email: true})\nformButtons = Buttons([submitButton])\nsubmitButton = Button(\"Send feedback\", Action([@ToAssistant(\"Submit this feedback form\")]), \"primary\", \"normal\", \"large\")","promptTokens":4433,"completionTokens":397,"latencyMs":5269,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/adversarial","repeat":1,"output":"root = Stack([feedbackCard], \"column\", \"m\", \"stretch\")\nfeedbackCard = Card([header, intro, feedbackForm], \"column\", \"m\")\nheader = CardHeader(\"Quick feedback\", \"Help us make your experience better\")\nintro = TextContent(\"This takes less than a minute.\", \"small\")\nfeedbackForm = Form(\"quick-feedback\", formButtons, [satisfactionField, issueField, commentField, contactField])\nsatisfactionField = FormControl(\"How was your experience?\", satisfactionInput)\nissueField = FormControl(\"What could we improve?\", issueInput)\ncommentField = FormControl(\"Anything else you'd like to share?\", commentInput, \"Optional — a sentence or two is plenty.\")\ncontactField = FormControl(\"Email for a reply\", contactInput, \"Optional — only if you'd like us to follow up.\")\nsatisfactionInput = RadioGroup(\"satisfaction\", [veryHappy, happy, neutral, unhappy, veryUnhappy], \"happy\")\nveryHappy = RadioItem(\"Great\", \"Everything went well\", \"great\")\nhappy = RadioItem(\"Good\", \"Mostly positive\", \"good\")\nneutral = RadioItem(\"Okay\", \"Neither good nor bad\", \"okay\")\nunhappy = RadioItem(\"Not great\", \"Something needs improvement\", \"not-great\")\nveryUnhappy = RadioItem(\"Poor\", \"We missed the mark\", \"poor\")\nissueInput = Select(\"issue\", [issueNone, issueProduct, issueSupport, issueSpeed, issueOther], \"Choose one (optional)\", null, $issue)\nissueNone = SelectItem(\"none\", \"Nothing in particular\")\nissueProduct = SelectItem(\"product\", \"Product or service\")\nissueSupport = SelectItem(\"support\", \"Customer support\")\nissueSpeed = SelectItem(\"speed\", \"Speed or usability\")\nissueOther = SelectItem(\"other\", \"Something else\")\ncommentInput = TextArea(\"comment\", \"Tell us what happened...\", 3, null, $comment)\ncontactInput = Input(\"email\", \"you@example.com\", \"email\", {email: true}, $email)\nformButtons = Buttons([submitButton], \"row\")\nsubmitButton = Button(\"Send feedback\", Action([@ToAssistant(\"Feedback submitted. Thank you for sharing your thoughts.\")]), \"primary\", \"normal\", \"large\")","promptTokens":4433,"completionTokens":472,"latencyMs":4975,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/adversarial","repeat":1,"output":"root = Stack([headerCard, introCallout, overviewCard, detailHeader, deployAccordion, rollbackCard, checklistCard, helpButtons])\n\nheaderCard = Card([header, headerTags])\nheader = CardHeader(\"Production Deployment Runbook\", \"8 sequential steps — with rationale, failure modes, and exact commands\")\nheaderTags = TagBlock([\"CI/CD\", \"Node/Docker\", \"Kubernetes\", \"Zero-downtime\", \"Reversible\"])\n\nintroCallout = Callout(\"info\", \"Read this first\", \"Steps are ordered by dependency: each one assumes the previous succeeded. Never skip ahead — the cheap checks (lint, test, build) exist so the expensive failures (bad image in production) never happen.\")\n\noverviewCard = Card([overviewHeader, overviewSteps])\noverviewHeader = CardHeader(\"The 8 steps at a glance\", \"Expand the section below for full detail on each\")\noverviewSteps = Steps([s1, s2, s3, s4, s5, s6, s7, s8])\ns1 = StepsItem(\"1. Freeze & sync the branch\", \"Pull latest main, confirm a clean working tree, and verify you are deploying the commit you think you are.\")\ns2 = StepsItem(\"2. Install exact dependencies\", \"Use a lockfile-respecting install so the deployed tree matches CI byte for byte.\")\ns3 = StepsItem(\"3. Lint, typecheck, test\", \"Catch cheap failures locally in seconds instead of in production in front of users.\")\ns4 = StepsItem(\"4. Build the production artifact\", \"Compile/bundle with production flags and confirm the build output exists.\")\ns5 = StepsItem(\"5. Build & push an immutable image\", \"Tag by commit SHA, never by 'latest', and push to the registry.\")\ns6 = StepsItem(\"6. Run database migrations\", \"Apply backward-compatible schema changes before the new code rolls out.\")\ns7 = StepsItem(\"7. Roll out & watch the rollout\", \"Deploy progressively and block until the new pods report healthy.\")\ns8 = StepsItem(\"8. Smoke test & monitor\", \"Verify real traffic paths, then watch error rate and latency for 15 minutes.\")\n\ndetailHeader = CardHeader(\"Step-by-step detail\", \"Why it matters · What breaks if you skip it · Exact commands\")\n\ndeployAccordion = Accordion([a1, a2, a3, a4, a5, a6, a7, a8])\n\na1 = AccordionItem(\"step1\", \"Step 1 — Freeze & sync the branch\", [a1why, a1skip, a1cmd])\na1why = TextContent(\"**Why it matters.** A deploy is a claim about *a specific commit*. If your local tree has uncommitted edits or is behind `origin/main`, the artifact you build is a commit that exists nowhere in history — so nobody can review it, reproduce it, bisect it, or roll back to it. Checking out the exact SHA also gives you the value you will use as your image tag in Step 5, which is what makes the whole pipeline traceable.\")\na1skip = TextCallout(\"danger\", \"If you skip it\", \"You ship a 'ghost build' containing a stray debug line, a local .env change, or someone else's half-merged feature. When it breaks at 2am, `git log` shows nothing wrong and the rollback target is unknown. This is the single most common cause of unexplainable production behavior.\")\na1cmd = CodeBlock(\"bash\", \"git checkout main\\ngit pull --ff-only origin main\\n\\n# Working tree MUST be empty. Any output here = stop and commit or stash.\\ngit status --porcelain\\n\\n# Capture the commit you are deploying; reuse it everywhere below.\\nexport RELEASE_SHA=$(git rev-parse --short HEAD)\\necho \\\"Deploying $RELEASE_SHA\\\"\\n\\n# Confirm CI is green for exactly this SHA\\ngh run list --commit \\\"$(git rev-parse HEAD)\\\" --limit 5\")\n\na2 = AccordionItem(\"step2\", \"Step 2 — Install exact dependencies\", [a2why, a2skip, a2cmd])\na2why = TextContent(\"**Why it matters.** `npm install` is allowed to *resolve* versions and rewrite the lockfile; `npm ci` is not. Only the second one guarantees the dependency graph in production is identical to the one your tests passed against. Deleting `node_modules` first removes stale native builds and packages that were removed from the manifest but still sit on disk.\")\na2skip = TextCallout(\"danger\", \"If you skip it\", \"A transitive dependency silently upgrades a minor version between test and deploy. You get 'works on my machine' failures: a formatting change breaks a parser, a peer-dep bump breaks the bundle, or a native module compiled for the wrong Node ABI crashes on boot with no code change of your own.\")\na2cmd = CodeBlock(\"bash\", \"# Pin the runtime first — Node version mismatches cause native module crashes\\nnode --version # must match .nvmrc / engines field\\nnvm use\\n\\nrm -rf node_modules\\nnpm ci --no-audit --no-fund\\n\\n# Fail the deploy on known high-severity CVEs\\nnpm audit --audit-level=high\")\n\na3 = AccordionItem(\"step3\", \"Step 3 — Lint, typecheck, and test\", [a3why, a3skip, a3cmd])\na3why = TextContent(\"**Why it matters.** These three gates cost seconds and each catches a different class of bug: lint catches unused/unsafe patterns, the typechecker catches contract mismatches between modules (the #1 source of runtime `undefined` errors), and tests catch behavioral regressions. Running them *before* building means you never waste time producing an artifact you would have to throw away.\")\na3skip = TextCallout(\"danger\", \"If you skip it\", \"You discover the bug at the most expensive possible moment — after the image is pushed, migrations have run, and traffic is live. A skipped typecheck routinely ships `Cannot read properties of undefined`, which surfaces only on the code path your one untested customer uses.\")\na3cmd = CodeBlock(\"bash\", \"npm run lint\\nnpm run typecheck # e.g. tsc --noEmit\\nnpm test -- --run --coverage\\n\\n# Optional but recommended before a risky release\\nnpm run test:e2e\\n\\n# Any non-zero exit code = STOP. Do not continue.\")\n\na4 = AccordionItem(\"step4\", \"Step 4 — Build the production artifact\", [a4why, a4skip, a4cmd])\na4why = TextContent(\"**Why it matters.** The production build is a *different program* from your dev server: `NODE_ENV=production` strips dev warnings, enables minification and tree-shaking, and inlines build-time environment variables. Building locally before containerizing surfaces bundler errors (missing assets, bad import paths, oversized chunks) in one fast loop instead of inside a slow Docker layer.\")\na4skip = TextCallout(\"warning\", \"If you skip it\", \"You either ship a development bundle — several times larger, slower, and leaking internal error traces to users — or you find out the build is broken only after a 5-minute image build. Missing build-time env vars fail silently and become blank pages or requests to `localhost` from your users' browsers.\")\na4cmd = CodeBlock(\"bash\", \"export NODE_ENV=production\\nnpm run build\\n\\n# Verify the artifact actually exists and is a sane size\\nls -lh dist/\\ndu -sh dist/\\n\\n# Serve the built output locally once — cheapest possible sanity check\\nnpx serve dist -l 5000\")\n\na5 = AccordionItem(\"step5\", \"Step 5 — Build & push an immutable image\", [a5why, a5skip, a5cmd])\na5why = TextContent(\"**Why it matters.** Tagging by commit SHA makes the image *immutable and addressable*: `app:9f3c1ab` means one exact thing forever. That is what makes rollback a one-line command instead of an archaeology project. Pushing to the registry before rolling out also proves the nodes will be able to pull it.\")\na5skip = TextCallout(\"danger\", \"If you skip it (or tag `:latest`)\", \"`latest` is a moving target. Two nodes pull it seconds apart and run different code, `kubectl rollout undo` re-pulls the same broken tag, and you cannot tell which build is live. If the push fails and you roll out anyway, pods enter `ImagePullBackOff` and the deploy stalls halfway — with old and new versions serving traffic simultaneously.\")\na5cmd = CodeBlock(\"bash\", \"export REGISTRY=ghcr.io/acme\\nexport IMAGE=\\\"$REGISTRY/api:$RELEASE_SHA\\\"\\n\\ndocker build \\\\\\n --platform linux/amd64 \\\\\\n --build-arg GIT_SHA=\\\"$RELEASE_SHA\\\" \\\\\\n -t \\\"$IMAGE\\\" .\\n\\n# Boot it once locally and hit the health endpoint\\ndocker run --rm -d -p 8080:8080 --name smoke \\\"$IMAGE\\\"\\nsleep 3 && curl -fsS localhost:8080/healthz && docker stop smoke\\n\\ndocker push \\\"$IMAGE\\\"\\ndocker manifest inspect \\\"$IMAGE\\\" > /dev/null && echo \\\"push verified\\\"\")\n\na6 = AccordionItem(\"step6\", \"Step 6 — Back up, then run migrations\", [a6why, a6skip, a6cmd])\na6why = TextContent(\"**Why it matters.** During a rolling deploy, old and new pods run **at the same time**, so every migration must be backward compatible: add nullable columns and new tables now, and only drop or rename columns in a *later* release once no old code references them. Take the backup first because a migration is the only step in this runbook that is not trivially reversible.\")\na6skip = TextCallout(\"danger\", \"If you skip it\", \"Skip the migration and the new code queries a column that does not exist — instant 500s on every request. Skip the *backup* and a bad `DROP COLUMN` or an unbatched `UPDATE` that locks a hot table becomes permanent data loss with no recovery path. Destructive migrations run in the same release as the code change take the old pods down while they are still serving traffic.\")\na6cmd = CodeBlock(\"bash\", \"# 1. Snapshot first — always\\npg_dump \\\"$DATABASE_URL\\\" -Fc -f \\\"backup-$RELEASE_SHA.dump\\\"\\n\\n# 2. Review exactly what will run (no surprises)\\nnpm run migrate:status\\nnpm run migrate:plan\\n\\n# 3. Apply inside a lock timeout so you never block production writes\\nPGOPTIONS='-c lock_timeout=5s -c statement_timeout=60s' \\\\\\n npm run migrate:up\\n\\nnpm run migrate:status # confirm all applied\")\n\na7 = AccordionItem(\"step7\", \"Step 7 — Roll out and watch the rollout\", [a7why, a7skip, a7cmd])\na7why = TextContent(\"**Why it matters.** `rollout status` blocks until the new ReplicaSet reports healthy, turning the deploy into a synchronous, verifiable operation with a clear pass/fail. Combined with readiness probes and a `maxUnavailable: 0` strategy, no pod receives traffic until it can actually serve it — that is what makes the release zero-downtime.\")\na7skip = TextCallout(\"danger\", \"If you skip the watch\", \"The command returns instantly and you walk away believing you shipped. Meanwhile pods are crash-looping on a missing env var, Kubernetes keeps the old ReplicaSet partly alive, and you are in a silent half-deployed state: some requests hit new code, some hit old, and error rates climb for 20 minutes before anyone notices.\")\na7cmd = CodeBlock(\"bash\", \"kubectl -n prod set image deployment/api api=\\\"$IMAGE\\\" --record\\n\\n# BLOCK until healthy — this is the actual gate\\nkubectl -n prod rollout status deployment/api --timeout=300s\\n\\n# Confirm every pod runs the intended SHA\\nkubectl -n prod get pods -l app=api \\\\\\n -o jsonpath='{range .items[*]}{.metadata.name}{\\\"\\\\t\\\"}{.spec.containers[0].image}{\\\"\\\\n\\\"}{end}'\\n\\n# If status times out or pods restart, stop and roll back:\\n# kubectl -n prod rollout undo deployment/api\")\n\na8 = AccordionItem(\"step8\", \"Step 8 — Smoke test, then monitor for 15 minutes\", [a8why, a8skip, a8cmd])\na8why = TextContent(\"**Why it matters.** Healthy pods only prove the process started, not that the product works — a container can pass `/healthz` while auth is misconfigured or a downstream secret is wrong. Smoke tests exercise real user paths against real production. Then you watch, because the worst failures are gradual: memory leaks, connection-pool exhaustion, and cache stampedes take minutes to appear.\")\na8skip = TextCallout(\"danger\", \"If you skip it\", \"Your users become the monitoring system. The classic case: login works, checkout returns 500 for one payment provider, and you find out from a support ticket six hours later. Deploying and immediately closing the laptop is how a 30-second rollback turns into a multi-hour incident.\")\na8cmd = CodeBlock(\"bash\", \"BASE=https://api.acme.com\\n\\ncurl -fsS \\\"$BASE/healthz\\\"\\ncurl -fsS \\\"$BASE/version\\\" | grep \\\"$RELEASE_SHA\\\" # right code is live\\ncurl -fsS -o /dev/null -w '%{http_code}\\\\n' \\\"$BASE/api/v1/orders?limit=1\\\"\\n\\nnpm run test:smoke -- --base-url \\\"$BASE\\\"\\n\\n# Watch logs and error rate for 15 minutes before declaring success\\nkubectl -n prod logs -l app=api --since=5m -f | grep -iE 'error|fatal|panic'\")\n\nrollbackCard = Card([rollbackHeader, rollbackCallout, rollbackText, rollbackCode], \"sunk\")\nrollbackHeader = CardHeader(\"If anything goes wrong: roll back first, debug second\", \"Target under 60 seconds from detection to recovery\")\nrollbackCallout = TextCallout(\"warning\", \"The rule\", \"Never debug a broken deploy while it is serving traffic. Restore the last known-good version, confirm recovery, then investigate from logs and the failed image at your leisure.\")\nrollbackText = TextContent(\"Code rollbacks are instant because Step 5 gave you immutable SHA tags. **Data rollbacks are not** — that is exactly why Step 6 required backward-compatible migrations plus a snapshot. If a migration is the problem, prefer rolling *forward* with a corrective migration over restoring a dump, which loses every write since the backup.\")\nrollbackCode = CodeBlock(\"bash\", \"# Fastest path: revert to the previous ReplicaSet\\nkubectl -n prod rollout undo deployment/api\\nkubectl -n prod rollout status deployment/api --timeout=180s\\n\\n# Or pin an explicit known-good SHA\\nkubectl -n prod set image deployment/api api=\\\"$REGISTRY/api:\\\"\\n\\n# Verify recovery, then investigate\\ncurl -fsS https://api.acme.com/version\\nkubectl -n prod logs --previous -l app=api --tail=200\")\n\nchecklistCard = Card([checklistHeader, checklistTable])\nchecklistHeader = CardHeader(\"Quick reference\", \"Gate, blast radius if skipped, and whether it is reversible\")\nchecklistTable = Table([colStep, colGate, colRisk, colRev])\ncolStep = Col(\"Step\", [\"1. Sync branch\", \"2. npm ci\", \"3. Lint/test\", \"4. Build\", \"5. Push image\", \"6. Migrate\", \"7. Rollout\", \"8. Smoke + watch\"])\ncolGate = Col(\"Blocking check\", [\"git status is empty\", \"audit level < high\", \"exit code 0\", \"dist/ exists\", \"manifest inspect OK\", \"migrate:status clean\", \"rollout status OK\", \"/version matches SHA\"])\ncolRisk = Col(\"If skipped\", [\"Untraceable ghost build\", \"Version drift crashes\", \"Bug found in prod\", \"Dev bundle shipped\", \"ImagePullBackOff stall\", \"500s or data loss\", \"Silent half-deploy\", \"Users report outage\"])\ncolRev = Col(\"Reversible?\", [tagYes, tagYes, tagYes, tagYes, tagYes, tagNo, tagYes, tagYes])\ntagYes = Tag(\"Yes\", null, \"sm\", \"success\")\ntagNo = Tag(\"No\", null, \"sm\", \"danger\")\n\nhelpButtons = Buttons([btnCi, btnBlue, btnMig])\nbtnCi = Button(\"Turn this into a GitHub Actions workflow\", Action([@ToAssistant(\"Turn this 8-step deployment runbook into a complete GitHub Actions workflow file with job dependencies, caching, and environment protection rules.\")]), \"primary\")\nbtnBlue = Button(\"Show blue/green instead of rolling\", Action([@ToAssistant(\"Rewrite steps 7 and 8 for a blue/green deployment with traffic shifting and an automated abort gate.\")]), \"secondary\")\nbtnMig = Button(\"Explain safe migration patterns\", Action([@ToAssistant(\"Explain backward-compatible database migration patterns in detail: expand/contract, adding nullable columns, backfilling in batches, and safely renaming a column across releases.\")]), \"secondary\")","promptTokens":6969,"completionTokens":6304,"latencyMs":82696,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/adversarial","repeat":3,"output":"root = Stack([feedbackCard], \"column\", \"m\", \"stretch\")\nfeedbackCard = Card([header, intro, feedbackForm], \"column\", \"m\", \"stretch\")\nheader = CardHeader(\"Quick feedback\", \"Help us make your experience better\")\nintro = TextContent(\"It only takes a minute. Your contact details are optional.\")\nfeedbackForm = Form(\"quick-feedback\", formButtons, [satisfactionField, commentsField, contactField])\nsatisfactionField = FormControl(\"How was your experience?\", satisfactionInput)\ncommentsField = FormControl(\"What could we improve?\", commentsInput, \"A few words is plenty.\")\ncontactField = FormControl(\"Email address\", contactInput, \"Leave this only if you would like a reply.\")\nsatisfactionInput = RadioGroup(\"satisfaction\", [veryHappy, happy, okay, unhappy], \"happy\", {required: true})\nveryHappy = RadioItem(\"Very happy\", \"Everything went well\", \"very_happy\")\nhappy = RadioItem(\"Happy\", \"Mostly positive\", \"happy\")\nokay = RadioItem(\"It was okay\", \"There is room to improve\", \"okay\")\nunhappy = RadioItem(\"Unhappy\", \"Something did not work for me\", \"unhappy\")\ncommentsInput = TextArea(\"comments\", \"Tell us what happened...\", 3, {maxLength: 500})\ncontactInput = Input(\"email\", \"you@example.com\", \"email\", {email: true})\nformButtons = Buttons([submitButton])\nsubmitButton = Button(\"Send feedback\", Action([@ToAssistant(\"Submit the feedback form.\")]), \"primary\", \"normal\", \"medium\")","promptTokens":4433,"completionTokens":409,"latencyMs":5159,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/adversarial","repeat":2,"output":"root = Stack([feedbackCard], \"column\", \"m\", \"stretch\")\nfeedbackCard = Card([header, intro, feedbackForm], \"column\", \"m\")\nheader = CardHeader(\"Quick feedback\", \"It only takes a minute.\")\nintro = TextContent(\"Tell us how it went. We read every response.\")\nfeedbackForm = Form(\"quick-feedback\", formButtons, [ratingField, issueField, detailsField, contactField])\nratingField = FormControl(\"How was your experience?\", ratingInput)\nratingInput = RadioGroup(\"rating\", [ratingGreat, ratingGood, ratingOkay, ratingPoor], \"good\", {required: true}, $rating)\nratingGreat = RadioItem(\"Great\", \"Everything worked well\", \"great\")\nratingGood = RadioItem(\"Good\", \"Mostly positive\", \"good\")\nratingOkay = RadioItem(\"Okay\", \"Could be better\", \"okay\")\nratingPoor = RadioItem(\"Not good\", \"Something went wrong\", \"poor\")\nissueField = FormControl(\"What could we improve?\", issueInput, \"Optional\")\nissueInput = Select(\"issue\", [issueNone, issueProduct, issueSupport, issueWebsite, issueOther], \"Choose an area\", null, $issue, \"medium\")\nissueNone = SelectItem(\"none\", \"Nothing in particular\")\nissueProduct = SelectItem(\"product\", \"Product or service\")\nissueSupport = SelectItem(\"support\", \"Customer support\")\nissueWebsite = SelectItem(\"website\", \"Website or app\")\nissueOther = SelectItem(\"other\", \"Something else\")\ndetailsField = FormControl(\"Anything else?\", detailsInput, \"Optional — a sentence or two is plenty.\")\ndetailsInput = TextArea(\"details\", \"What happened?\", 3, null, $details)\ncontactField = FormControl(\"Email for a reply\", contactInput, \"Optional\")\ncontactInput = Input(\"email\", \"you@example.com\", \"email\", {email: true}, $email)\nformButtons = Buttons([submitButton])\nsubmitButton = Button(\"Send feedback\", Action([@ToAssistant(\"Submit my feedback\")]), \"primary\", \"normal\", \"large\")","promptTokens":4433,"completionTokens":517,"latencyMs":9498,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/minimal","repeat":0,"output":"root = Stack([title, countryTable], \"column\", \"m\")\ntitle = CardHeader(\"Largest Countries by Total Area\", \"Approximate total area in square kilometers\")\ncountryTable = Table([rankColumn, countryColumn, areaColumn])\nrankColumn = Col(\"Rank\", [1, 2, 3], \"number\")\ncountryColumn = Col(\"Country\", [\"Russia\", \"Canada\", \"China\"])\nareaColumn = Col(\"Area (km²)\", [17098242, 9984670, 9596961], \"number\")","promptTokens":4389,"completionTokens":206,"latencyMs":4588,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"contact-form/adversarial","repeat":4,"output":"root = Stack([header, feedbackCard], \"column\", \"l\", \"stretch\")\nheader = Stack([title, intro], \"column\", \"xs\")\ntitle = TextContent(\"Share your feedback\", \"large-heavy\")\nintro = TextContent(\"A few quick answers help us improve. It takes less than a minute.\", \"default\")\nfeedbackCard = Card([feedbackForm], \"column\", \"m\")\nfeedbackForm = Form(\"feedback\", formButtons, [ratingField, commentField, contactField])\nratingField = FormControl(\"How was your experience?\", ratingInput)\nratingInput = RadioGroup(\"rating\", [ratingGreat, ratingGood, ratingOkay, ratingPoor], null, {required: true}, $rating)\nratingGreat = RadioItem(\"Great\", \"Everything worked well\", \"great\")\nratingGood = RadioItem(\"Good\", \"A mostly positive experience\", \"good\")\nratingOkay = RadioItem(\"Okay\", \"It could be better\", \"okay\")\nratingPoor = RadioItem(\"Not good\", \"Something needs attention\", \"poor\")\ncommentField = FormControl(\"What could we improve?\", commentInput, \"Optional — a short note is plenty.\")\ncommentInput = TextArea(\"comment\", \"Tell us what happened or what would make this better...\", 3, null, $comment)\ncontactField = FormControl(\"Email for a reply\", contactInput, \"Optional — only if you would like us to follow up.\")\ncontactInput = Input(\"email\", \"you@example.com\", \"email\", {email: true}, $email)\nformButtons = Buttons([submitButton])\nsubmitButton = Button(\"Send feedback\", Action([@ToAssistant(\"Submit feedback: rating, comment, and optional reply email.\")]), \"primary\", \"normal\", \"large\")","promptTokens":4433,"completionTokens":481,"latencyMs":5546,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/minimal","repeat":1,"output":"root = Stack([title, tableCard], \"column\", \"l\")\ntitle = TextContent(\"Three Largest Countries by Total Area\", \"large-heavy\")\ntableCard = Card([countryTable])\ncountryTable = Table([rankCol, countryCol, areaCol, continentCol])\nrankCol = Col(\"Rank\", [1, 2, 3], \"number\")\ncountryCol = Col(\"Country\", [\"Russia\", \"Canada\", \"China\"])\nareaCol = Col(\"Total area (km²)\", [17098242, 9984670, 9596961], \"number\")\ncontinentCol = Col(\"Primary continent\", [\"Europe / Asia\", \"North America\", \"Asia\"])","promptTokens":4389,"completionTokens":144,"latencyMs":4085,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/minimal","repeat":2,"output":"root = Stack([title, countryTable], \"column\", \"m\")\ntitle = CardHeader(\"Three Largest Countries by Area\", \"Total area figures are approximate\")\ncountryTable = Table([rankCol, countryCol, areaCol])\nrankCol = Col(\"Rank\", [1, 2, 3], \"number\")\ncountryCol = Col(\"Country\", [\"Russia\", \"Canada\", \"China\"])\nareaCol = Col(\"Area (km²)\", [17098242, 9984670, 9596961], \"number\")","promptTokens":4389,"completionTokens":114,"latencyMs":2992,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/minimal","repeat":3,"output":"root = Stack([title, countriesTable], \"column\", \"m\")\ntitle = TextContent(\"Three Largest Countries by Total Area\", \"large-heavy\")\ncountriesTable = Table([rankCol, countryCol, areaCol])\nrankCol = Col(\"Rank\", [1, 2, 3], \"number\")\ncountryCol = Col(\"Country\", [\"Russia\", \"Canada\", \"China\"])\nareaCol = Col(\"Total area (km²)\", [17098242, 9984670, 9596961], \"number\")","promptTokens":4389,"completionTokens":188,"latencyMs":4665,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/minimal","repeat":4,"output":"root = Stack([pageTitle, countryTable], \"column\", \"l\")\npageTitle = Card([header, note], \"clear\", \"column\", \"xs\")\nheader = CardHeader(\"Largest Countries by Total Area\", \"The three largest sovereign countries by total area\")\nnote = TextContent(\"Areas are approximate and include inland water.\", \"small\")\ncountryTable = Table([rankColumn, countryColumn, areaColumn])\nrankColumn = Col(\"Rank\", [1, 2, 3], \"number\")\ncountryColumn = Col(\"Country\", [\"Russia\", \"Canada\", \"China\"])\nareaColumn = Col(\"Total area (km²)\", [17098242, 9984670, 9596961], \"number\")","promptTokens":4389,"completionTokens":153,"latencyMs":3743,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/realistic","repeat":0,"output":"root = Stack([header, resultsTable], \"column\", \"l\")\nheader = Card([title, subtitle], \"column\", \"xs\")\ntitle = CardHeader(\"Q3 Sales Team Results\", \"EMEA and APAC • Mid-size software company\")\nsubtitle = TextContent(\"Quota attainment reflects closed-won revenue through the end of Q3.\", \"small\")\nresultsTable = Table([repColumn, regionColumn, dealsColumn, revenueColumn, quotaColumn])\nrepColumn = Col(\"Rep Name\", [\"Amelia Grant\", \"Daniel Weber\", \"Sofia Marin\", \"Oliver Hayes\", \"Priya Nair\", \"Kenji Tanaka\", \"Mei Lin\", \"Arjun Rao\"])\nregionColumn = Col(\"Region\", [\"EMEA\", \"EMEA\", \"EMEA\", \"EMEA\", \"APAC\", \"APAC\", \"APAC\", \"APAC\"])\ndealsColumn = Col(\"Deals Closed\", [18, 15, 21, 13, 17, 19, 16, 14], \"number\")\nrevenueColumn = Col(\"Revenue\", [\"$428,000\", \"$356,000\", \"$489,000\", \"$301,000\", \"$412,000\", \"$461,000\", \"$374,000\", \"$329,000\"])\nquotaColumn = Col(\"Against Quota\", [attainmentAmelia, attainmentDaniel, attainmentSofia, attainmentOliver, attainmentPriya, attainmentKenji, attainmentMei, attainmentArjun])\nattainmentAmelia = Tag(\"107% • $400K quota\", null, \"sm\", \"success\")\nattainmentDaniel = Tag(\"102% • $350K quota\", null, \"sm\", \"success\")\nattainmentSofia = Tag(\"109% • $450K quota\", null, \"sm\", \"success\")\nattainmentOliver = Tag(\"94% • $320K quota\", null, \"sm\", \"warning\")\nattainmentPriya = Tag(\"103% • $400K quota\", null, \"sm\", \"success\")\nattainmentKenji = Tag(\"108% • $425K quota\", null, \"sm\", \"success\")\nattainmentMei = Tag(\"94% • $400K quota\", null, \"sm\", \"warning\")\nattainmentArjun = Tag(\"94% • $350K quota\", null, \"sm\", \"warning\")","promptTokens":4425,"completionTokens":555,"latencyMs":9482,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/realistic","repeat":2,"output":"root = Stack([report], \"column\", \"l\")\nreport = Card([header, resultsTable], \"column\", \"m\")\nheader = CardHeader(\"Q3 Sales Team Results\", \"Eight account executives across EMEA and APAC • Revenue in USD\")\nresultsTable = Table([repColumn, regionColumn, dealsColumn, revenueColumn, quotaColumn])\nrepColumn = Col(\"Rep Name\", salesData.repName)\nregionColumn = Col(\"Region\", salesData.region)\ndealsColumn = Col(\"Deals Closed\", salesData.dealsClosed, \"number\")\nrevenueColumn = Col(\"Revenue\", salesData.revenue, \"number\")\nquotaColumn = Col(\"Against Quota\", salesData.quotaAttainment)\nsalesData = {\"repName\":[\"Amelia Hart\",\"Jonas Weber\",\"Sofia Marin\",\"Theo Bennett\",\"Mei Lin\",\"Arjun Patel\",\"Hana Kim\",\"Liam Chen\"],\"region\":[\"EMEA\",\"EMEA\",\"EMEA\",\"EMEA\",\"APAC\",\"APAC\",\"APAC\",\"APAC\"],\"dealsClosed\":[18,15,21,13,17,19,16,14],\"revenue\":[684000,571000,762000,498000,638000,721000,603000,529000],\"quotaAttainment\":[\"114%\",\"103%\",\"127%\",\"91%\",\"106%\",\"120%\",\"101%\",\"88%\"]}","promptTokens":4425,"completionTokens":421,"latencyMs":5887,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/adversarial","repeat":2,"output":"root = Stack([header, intro, overviewCard, detailHeader, detailAccordion, rollbackCallout, quickRefCard, nextSteps], \"column\", \"l\")\n\nheader = CardHeader(\"Production Deployment Runbook\", \"8 sequential steps — with rationale, failure modes, and exact commands\")\n\nintro = MarkDownRenderer(\"This runbook assumes a **Git-based project** with CI, a container image, and a staging environment that mirrors production. Replace `myapp`, `registry.example.com`, and branch names with your own values.\", \"sunk\")\n\noverviewCard = Card([overviewHeader, overviewSteps], \"card\")\noverviewHeader = CardHeader(\"Overview\", \"Read top to bottom — order is not optional\")\noverviewSteps = Steps([s1, s2, s3, s4, s5, s6, s7, s8])\n\ns1 = StepsItem(\"1. Freeze and verify the source\", \"Confirm the exact commit you intend to ship. Pull latest, check status is clean, tag the release.\")\ns2 = StepsItem(\"2. Install locked dependencies\", \"Use the lockfile, never a fresh resolve. Guarantees the same tree CI tested.\")\ns3 = StepsItem(\"3. Run tests, lint, and type checks\", \"Cheapest place to catch a regression. Failing fast here costs minutes; failing in prod costs hours.\")\ns4 = StepsItem(\"4. Build a versioned artifact\", \"One immutable, content-addressed build promoted through every environment.\")\ns5 = StepsItem(\"5. Back up data and run migrations\", \"Schema changes are the only truly hard-to-reverse part of a deploy.\")\ns6 = StepsItem(\"6. Deploy to staging and smoke test\", \"Last checkpoint with real infrastructure before customers are exposed.\")\ns7 = StepsItem(\"7. Progressive rollout to production\", \"Limit blast radius: 10% of traffic first, then ramp while watching error rates.\")\ns8 = StepsItem(\"8. Verify, monitor, and announce\", \"A deploy isn't done until metrics are green for a full observation window.\")\n\ndetailHeader = CardHeader(\"Step Detail\", \"Why it matters · what breaks if you skip it · commands\")\n\ndetailAccordion = Accordion([a1, a2, a3, a4, a5, a6, a7, a8])\n\na1 = AccordionItem(\"step1\", \"Step 1 — Freeze and verify the source\", [a1why, a1skip, a1cmd])\na1why = MarkDownRenderer(\"**Why it matters:** A deploy is a claim about a specific commit. If you don't pin it, you can't answer \\\"what is running in production right now?\\\" Tagging creates a permanent, human-readable pointer for incident response and rollback.\\n\\n**If you skip it:** You ship uncommitted local changes or a half-merged branch. Rollback becomes archaeology — you guess at the last good SHA while the site is down. Two engineers deploying within minutes of each other silently overwrite one another.\")\na1skip = TextCallout(\"danger\", \"Failure mode\", \"\\\"It works on my machine\\\" ships to prod because an untracked file was in the build context.\")\na1cmd = CodeBlock(\"bash\", \"git checkout main\\ngit pull --ff-only origin main\\ngit status --porcelain # must print nothing\\ngit log -1 --oneline\\n\\n# Pin the release\\nVERSION=$(date +%Y.%m.%d)-$(git rev-parse --short HEAD)\\ngit tag -a \\\"v$VERSION\\\" -m \\\"Release $VERSION\\\"\\ngit push origin \\\"v$VERSION\\\"\\necho \\\"$VERSION\\\"\")\n\na2 = AccordionItem(\"step2\", \"Step 2 — Install locked dependencies\", [a2why, a2skip, a2cmd])\na2why = MarkDownRenderer(\"**Why it matters:** `npm install` may *resolve* new transitive versions; `npm ci` installs the exact tree in `package-lock.json` and fails if the lockfile and manifest disagree. Reproducibility is the whole point of a lockfile.\\n\\n**If you skip it:** A patch release of a dependency you've never heard of lands in your production bundle at 2am. The bug is untraceable because your local install and the build machine's install differ.\")\na2skip = TextCallout(\"warning\", \"Also do this\", \"Run an audit. A known CVE in a direct dependency is far cheaper to fix before the build than after disclosure.\")\na2cmd = CodeBlock(\"bash\", \"rm -rf node_modules\\nnpm ci # NOT npm install\\nnpm audit --audit-level=high\\n\\n# Python equivalent\\n# pip install --require-hashes -r requirements.lock\")\n\na3 = AccordionItem(\"step3\", \"Step 3 — Tests, lint, and type checks\", [a3why, a3skip, a3cmd])\na3why = MarkDownRenderer(\"**Why it matters:** Each gate catches a different class of defect. Type checks catch shape errors, unit tests catch logic errors, integration tests catch wiring errors, lint catches footguns. Running them locally *and* in CI means a red build never becomes a deploy.\\n\\n**If you skip it:** You trade a 4-minute test run for a customer-facing incident. The regression is usually in the code path you were \\\"sure\\\" you didn't touch — a shared utility, a serializer, an env var default.\")\na3skip = TextCallout(\"info\", \"Rule\", \"Never deploy from a branch with a red or skipped CI run. \\\"I'll fix the test after\\\" is how flaky suites are born.\")\na3cmd = CodeBlock(\"bash\", \"npm run lint\\nnpm run typecheck\\nnpm test -- --run --coverage\\nnpm run test:e2e -- --headless\\n\\n# Confirm CI agrees with your machine\\ngh run list --branch main --limit 1\\ngh run watch\")\n\na4 = AccordionItem(\"step4\", \"Step 4 — Build a versioned, immutable artifact\", [a4why, a4skip, a4cmd])\na4why = MarkDownRenderer(\"**Why it matters:** Build once, promote everywhere. The bytes you smoke-tested on staging must be the identical bytes serving production — same image digest, no rebuild. Tagging by version *and* commit SHA makes the artifact self-documenting.\\n\\n**If you skip it:** Rebuilding per environment means staging and prod are different software. You also lose the ability to roll back instantly, because the previous artifact was never stored anywhere.\")\na4skip = TextCallout(\"danger\", \"Never\", \"Do not deploy the `latest` tag. It is mutable, unversioned, and impossible to roll back to a known state.\")\na4cmd = CodeBlock(\"bash\", \"SHA=$(git rev-parse --short HEAD)\\nIMAGE=registry.example.com/myapp\\n\\ndocker build \\\\\\n --build-arg VERSION=\\\"$VERSION\\\" \\\\\\n --build-arg GIT_SHA=\\\"$SHA\\\" \\\\\\n -t \\\"$IMAGE:$VERSION\\\" \\\\\\n -t \\\"$IMAGE:$SHA\\\" .\\n\\ndocker push \\\"$IMAGE:$VERSION\\\"\\ndocker push \\\"$IMAGE:$SHA\\\"\\n\\n# Record the immutable digest — deploy by digest, not tag\\ndocker inspect --format='{{index .RepoDigests 0}}' \\\"$IMAGE:$VERSION\\\"\")\n\na5 = AccordionItem(\"step5\", \"Step 5 — Back up data, then migrate\", [a5why, a5skip, a5cmd])\na5why = MarkDownRenderer(\"**Why it matters:** Code rollback is a one-line command; data loss is permanent. Take a backup *immediately before* migrating, and write migrations to be **backward compatible** so the old app version can still run against the new schema during rollout.\\n\\n**If you skip the backup:** A migration with a bad `DROP COLUMN` or an unbounded `UPDATE` destroys production data with no recovery path.\\n\\n**If you skip backward compatibility:** During the rollout window, old and new pods run simultaneously against one database. The old code queries a column you just renamed and throws 500s for every request it serves.\")\na5skip = TextCallout(\"warning\", \"Expand / contract\", \"Deploy additive changes (add column, backfill, dual-write) in release N. Remove the old column in release N+1, after all traffic is on the new code.\")\na5cmd = CodeBlock(\"bash\", \"# 1. Verified backup\\npg_dump \\\"$DATABASE_URL\\\" -Fc -f \\\"backup-$VERSION.dump\\\"\\npg_restore --list \\\"backup-$VERSION.dump\\\" | head\\n\\n# 2. Dry run: inspect the SQL before it executes\\nnpm run migrate:status\\nnpm run migrate:plan\\n\\n# 3. Apply with a lock timeout so you can't freeze the table\\nPGOPTIONS='-c lock_timeout=5s -c statement_timeout=60s' \\\\\\n npm run migrate:up\\n\\nnpm run migrate:status # confirm all applied\")\n\na6 = AccordionItem(\"step6\", \"Step 6 — Deploy to staging and smoke test\", [a6why, a6skip, a6cmd])\na6why = MarkDownRenderer(\"**Why it matters:** Staging is the only place you exercise the real artifact against real infrastructure: DNS, TLS, secrets, connection pools, object storage, third-party webhooks. Unit tests cannot cover any of those.\\n\\n**If you skip it:** The first request that touches a missing environment variable, an unrotated secret, or an unreachable internal service is a customer's request. This is the single most common cause of \\\"the code was fine but the deploy failed.\\\"\")\na6skip = TextCallout(\"info\", \"Smoke test scope\", \"Health endpoint, login, one read path, one write path, and one external integration. Keep it under two minutes so nobody is tempted to skip it.\")\na6cmd = CodeBlock(\"bash\", \"kubectl -n staging set image deploy/myapp \\\\\\n app=\\\"$IMAGE:$VERSION\\\"\\nkubectl -n staging rollout status deploy/myapp --timeout=120s\\n\\n# Verify the running version is the one you built\\ncurl -fsS https://staging.example.com/healthz | jq .\\ncurl -fsS https://staging.example.com/version | jq -r .git_sha\\n\\nnpm run smoke -- --base-url=https://staging.example.com\\nkubectl -n staging logs deploy/myapp --tail=100 | grep -i error\")\n\na7 = AccordionItem(\"step7\", \"Step 7 — Progressive rollout to production\", [a7why, a7skip, a7cmd])\na7why = MarkDownRenderer(\"**Why it matters:** Some failures only appear at production scale or with production data. Sending 10% of traffic to the new version means a bad release affects 10% of users for 5 minutes instead of 100% for 30. Surge/maxUnavailable settings keep capacity constant so you never brown out mid-deploy.\\n\\n**If you skip it:** A big-bang deploy replaces every instance at once. If the new version crash-loops, you have zero healthy capacity and a total outage — and the rollback itself takes minutes you don't have.\")\na7skip = TextCallout(\"success\", \"Gate on data, not vibes\", \"Hold at each stage until error rate and p95 latency match the pre-deploy baseline. If either degrades, roll back — do not \\\"wait and see.\\\"\")\na7cmd = CodeBlock(\"bash\", \"# Canary: one replica on the new version\\nkubectl -n prod apply -f k8s/canary.yaml\\nkubectl -n prod set image deploy/myapp-canary app=\\\"$IMAGE:$VERSION\\\"\\nkubectl -n prod rollout status deploy/myapp-canary --timeout=180s\\n\\n# Watch for 5 minutes before ramping\\nkubectl -n prod logs -l app=myapp,track=canary -f --tail=50\\n\\n# Ramp: 10% -> 50% -> 100%\\nfor W in 10 50 100; do\\n kubectl -n prod annotate ingress myapp \\\\\\n nginx.ingress.kubernetes.io/canary-weight=\\\"$W\\\" --overwrite\\n echo \\\"weight=$W — observing\\\"; sleep 300\\ndone\\n\\n# Promote the primary deployment\\nkubectl -n prod set image deploy/myapp app=\\\"$IMAGE:$VERSION\\\"\\nkubectl -n prod rollout status deploy/myapp --timeout=300s\")\n\na8 = AccordionItem(\"step8\", \"Step 8 — Verify, monitor, announce\", [a8why, a8skip, a8cmd])\na8why = MarkDownRenderer(\"**Why it matters:** Latent failures — memory leaks, connection-pool exhaustion, cache stampedes, cron jobs — surface 15 to 60 minutes in, not at minute one. An explicit observation window with a named owner is what turns \\\"deployed\\\" into \\\"done.\\\" Recording the release in your changelog and chat gives the next on-call engineer a timeline to correlate against.\\n\\n**If you skip it:** The deploy is declared successful, everyone leaves, and the alert fires during dinner with no context. Mean time to recovery triples because nobody remembers what changed.\")\na8skip = TextCallout(\"neutral\", \"Observation window\", \"Stay hands-on for 30 minutes. Watch error rate, p95/p99 latency, saturation (CPU, memory, pool), and queue depth.\")\na8cmd = CodeBlock(\"bash\", \"# Confirm what's actually live\\ncurl -fsS https://example.com/version | jq .\\nkubectl -n prod get pods -l app=myapp -o wide\\nkubectl -n prod get events --sort-by=.lastTimestamp | tail -20\\n\\n# Error rate over the last 15 minutes\\ncurl -sG \\\"$PROM/api/v1/query\\\" \\\\\\n --data-urlencode 'query=sum(rate(http_requests_total{status=~\\\"5..\\\"}[15m]))'\\n\\n# Announce\\ngh release create \\\"v$VERSION\\\" --generate-notes\\ncurl -X POST \\\"$SLACK_WEBHOOK\\\" \\\\\\n -d \\\"{\\\\\\\"text\\\\\\\":\\\\\\\"Deployed myapp v$VERSION to production ($SHA)\\\\\\\"}\\\"\")\n\nrollbackCallout = Callout(\"error\", \"If anything looks wrong: roll back first, debug second\", \"Rollback is not failure — it is the cheapest available action. Restore service, then investigate from a calm state with real logs.\")\n\nquickRefCard = Card([quickRefHeader, rollbackCode, gateTable], \"card\")\nquickRefHeader = CardHeader(\"Quick Reference\", \"Rollback commands and the gate that must pass at each step\")\nrollbackCode = CodeBlock(\"bash\", \"# Undo the last rollout (fastest, ~15s)\\nkubectl -n prod rollout undo deploy/myapp\\nkubectl -n prod rollout status deploy/myapp\\n\\n# Or pin the previous known-good artifact\\nkubectl -n prod set image deploy/myapp app=\\\"$IMAGE:$PREV_VERSION\\\"\\n\\n# Kill the canary immediately\\nkubectl -n prod annotate ingress myapp \\\\\\n nginx.ingress.kubernetes.io/canary-weight=0 --overwrite\\n\\n# Data rollback (last resort — expect downtime)\\nnpm run migrate:down -- --step=1\\npg_restore -d \\\"$DATABASE_URL\\\" --clean \\\"backup-$VERSION.dump\\\"\")\n\ngateTable = Table([colStep, colGate, colTime, colRisk])\ncolStep = Col(\"Step\", [\"1 Freeze\", \"2 Deps\", \"3 Tests\", \"4 Build\", \"5 Migrate\", \"6 Staging\", \"7 Rollout\", \"8 Verify\"])\ncolGate = Col(\"Must be true before continuing\", [\"git status clean, tag pushed\", \"npm ci exits 0, no high CVEs\", \"lint + types + tests all green\", \"Image digest recorded\", \"Backup verified, migration reversible\", \"Smoke suite passes, /version matches\", \"Error rate flat at each weight\", \"30 min green, release announced\"])\ncolTime = Col(\"Minutes\", [2, 3, 6, 5, 8, 5, 20, 30], \"number\")\ncolRisk = Col(\"Risk if skipped\", [riskHigh, riskMed, riskCrit, riskMed, riskCrit, riskHigh, riskCrit, riskMed])\nriskCrit = Tag(\"critical\", null, \"sm\", \"danger\")\nriskHigh = Tag(\"high\", null, \"sm\", \"warning\")\nriskMed = Tag(\"medium\", null, \"sm\", \"info\")\n\nnextSteps = Buttons([btnAutomate, btnRollback, btnMigrations, btnObserve])\nbtnAutomate = Button(\"Turn this into a CI/CD pipeline\", Action([@ToAssistant(\"Write a GitHub Actions workflow that automates all 8 deployment steps from this runbook, including the canary ramp and rollback job.\")]), \"primary\")\nbtnRollback = Button(\"Deep dive on rollback strategy\", Action([@ToAssistant(\"Explain rollback strategies in detail: blue-green vs canary vs rolling, and how to handle irreversible database migrations.\")]), \"secondary\")\nbtnMigrations = Button(\"Expand/contract migration guide\", Action([@ToAssistant(\"Show me a concrete expand/contract migration example for renaming a column with zero downtime, across three releases.\")]), \"secondary\")\nbtnObserve = Button(\"What to monitor after deploy\", Action([@ToAssistant(\"Give me a checklist of the specific metrics, alerts, and dashboards to watch during the 30-minute post-deploy observation window.\")]), \"secondary\")","promptTokens":6969,"completionTokens":6326,"latencyMs":79009,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/realistic","repeat":4,"output":"root = Stack([salesTable])\nsalesTable = Table([repColumn, regionColumn, dealsColumn, revenueColumn, quotaColumn])\nrepColumn = Col(\"Rep Name\", [\"Amelia Hart\", \"Luca Bianchi\", \"Sofia Lindström\", \"James Okafor\", \"Priya Nair\", \"Kenji Tanaka\", \"Mei Lin\", \"Arjun Patel\"])\nregionColumn = Col(\"Region\", [\"EMEA\", \"EMEA\", \"EMEA\", \"EMEA\", \"APAC\", \"APAC\", \"APAC\", \"APAC\"])\ndealsColumn = Col(\"Deals Closed\", [14, 12, 11, 13, 15, 10, 12, 14], \"number\")\nrevenueColumn = Col(\"Revenue\", [\"$428,000\", \"$367,000\", \"$341,000\", \"$396,000\", \"$452,000\", \"$318,000\", \"$374,000\", \"$421,000\"])\nquotaColumn = Col(\"Against Quota\", [Tag(\"119% • Exceeded\", null, \"sm\", \"success\"), Tag(\"105% • Exceeded\", null, \"sm\", \"success\"), Tag(\"97% • Near target\", null, \"sm\", \"warning\"), Tag(\"110% • Exceeded\", null, \"sm\", \"success\"), Tag(\"126% • Exceeded\", null, \"sm\", \"success\"), Tag(\"91% • Below target\", null, \"sm\", \"danger\"), Tag(\"102% • Exceeded\", null, \"sm\", \"success\"), Tag(\"114% • Exceeded\", null, \"sm\", \"success\")])","promptTokens":4425,"completionTokens":429,"latencyMs":6206,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/realistic","repeat":3,"output":"root = Stack([salesCard])\nsalesCard = Card([salesHeader, salesTable])\nsalesHeader = CardHeader(\"Q3 Sales Team Results\", \"Eight account executives across EMEA and APAC • Q3 performance against individual revenue quota\")\nsalesTable = Table([repColumn, regionColumn, dealsColumn, revenueColumn, quotaColumn])\nrepColumn = Col(\"Rep name\", [\"Amelia Hart\", \"Luca Moretti\", \"Sophie Laurent\", \"Jonas Weber\", \"Priya Nair\", \"Kenji Tanaka\", \"Mei Lin\", \"Daniel Wong\"])\nregionColumn = Col(\"Region\", [\"EMEA\", \"EMEA\", \"EMEA\", \"EMEA\", \"APAC\", \"APAC\", \"APAC\", \"APAC\"])\ndealsColumn = Col(\"Deals closed\", [18, 15, 21, 13, 17, 19, 16, 14], \"number\")\nrevenueColumn = Col(\"Revenue\", [\"$486,000\", \"$421,000\", \"$574,000\", \"$338,000\", \"$459,000\", \"$512,000\", \"$437,000\", \"$368,000\"])\nquotaColumn = Col(\"Against quota\", [Tag(\"108% • Exceeded\", null, \"sm\", \"success\"), Tag(\"94% • Near target\", null, \"sm\", \"warning\"), Tag(\"115% • Exceeded\", null, \"sm\", \"success\"), Tag(\"85% • Below target\", null, \"sm\", \"danger\"), Tag(\"102% • Exceeded\", null, \"sm\", \"success\"), Tag(\"114% • Exceeded\", null, \"sm\", \"success\"), Tag(\"97% • Near target\", null, \"sm\", \"warning\"), Tag(\"89% • Below target\", null, \"sm\", \"danger\")])","promptTokens":4425,"completionTokens":504,"latencyMs":6915,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/adversarial","repeat":4,"output":"root = Stack([header, overview, stepsOverview, detailHeader, detail, rollbackNote, footerBtns], \"column\", \"l\")\n\nheader = Card([headerText, headerTags])\nheaderText = CardHeader(\"Deployment Runbook\", \"Eight steps from clean checkout to verified production release — with rationale, failure modes, and exact commands\")\nheaderTags = TagBlock([\"CI/CD\", \"Zero-downtime\", \"Node/Docker\", \"Kubernetes\", \"Rollback-ready\"])\n\noverview = Callout(\"info\", \"Read this first\", \"Never skip steps 1, 4, or 8. Everything else can be automated, but those three are what stop a bad build from becoming an outage.\")\n\nstepsOverview = Card([stepsOverviewHeader, stepsList])\nstepsOverviewHeader = CardHeader(\"The pipeline at a glance\", \"Each step gates the next — a red step means stop, not 'retry until green'\")\nstepsList = Steps([s1, s2, s3, s4, s5, s6, s7, s8])\ns1 = StepsItem(\"1. Freeze & verify the source\", \"Pull the exact commit, confirm a clean working tree, and tag it. The artifact must be traceable to one immutable SHA.\")\ns2 = StepsItem(\"2. Install locked dependencies\", \"Use the lockfile, not the manifest. Reproducible installs mean the build machine and prod see identical trees.\")\ns3 = StepsItem(\"3. Lint, typecheck, and test\", \"Cheap checks first, expensive checks second. Fail in seconds, not in production.\")\ns4 = StepsItem(\"4. Build the immutable artifact\", \"One build, promoted through environments. Never rebuild per environment.\")\ns5 = StepsItem(\"5. Run database migrations\", \"Backwards-compatible, forward-only, applied before the new code goes live.\")\ns6 = StepsItem(\"6. Deploy to staging & smoke test\", \"A production-shaped rehearsal against real infrastructure and real config.\")\ns7 = StepsItem(\"7. Progressive rollout to production\", \"Canary a small slice of traffic, watch error rates, then widen.\")\ns8 = StepsItem(\"8. Verify, monitor, and record\", \"Confirm health post-deploy, watch the window, and log what shipped.\")\n\ndetailHeader = CardHeader(\"Step-by-step detail\", \"Expand any step for why it matters, the skip risk, and copy-paste commands\")\n\ndetail = Accordion([d1, d2, d3, d4, d5, d6, d7, d8])\n\nd1 = AccordionItem(\"d1\", \"Step 1 — Freeze & verify the source\", [d1why, d1skip, d1cmd, d1note])\nd1why = TextContent(\"**Why it matters**\\n\\nA deploy is only meaningful if you can say exactly *what* you deployed. Checking out a specific commit and confirming `git status` is clean guarantees the artifact you build matches what reviewers approved. An annotated tag gives you a human-readable handle (`v2.14.0`) permanently bound to an immutable SHA, which is what you'll need at 3am when you're comparing 'before' and 'after'.\")\nd1skip = TextCallout(\"danger\", \"If you skip it\", \"You build from a dirty tree containing uncommitted local edits. The bug you later see in production does not exist in any commit, so it cannot be reproduced, bisected, or reverted. Rollback becomes guesswork.\")\nd1cmd = CodeBlock(\"bash\", \"# Fetch everything, including tags\\ngit fetch --all --tags --prune\\n\\n# Check out the release branch and fast-forward\\ngit checkout main\\ngit pull --ff-only origin main\\n\\n# HARD GATE: working tree must be empty\\ngit status --porcelain\\n# (no output = clean; any output = stop and commit or stash)\\n\\n# Record and tag the exact commit\\nexport RELEASE_SHA=$(git rev-parse --short HEAD)\\necho \\\"Deploying $RELEASE_SHA\\\"\\n\\ngit tag -a \\\"v2.14.0\\\" -m \\\"Release v2.14.0 ($RELEASE_SHA)\\\"\\ngit push origin \\\"v2.14.0\\\"\")\nd1note = TextContent(\"_Tip:_ make `git status --porcelain` a hard failure in CI: `[ -z \\\"$(git status --porcelain)\\\" ] || exit 1`\", \"small\")\n\nd2 = AccordionItem(\"d2\", \"Step 2 — Install locked dependencies\", [d2why, d2skip, d2cmd, d2note])\nd2why = TextContent(\"**Why it matters**\\n\\n`npm install` is allowed to *resolve* new versions that satisfy your semver ranges; `npm ci` is not — it installs the lockfile byte-for-byte and errors if the lockfile and `package.json` disagree. That distinction is the difference between a reproducible build and a lottery. It also deletes `node_modules` first, so stale transitive packages from a previous build can't leak in.\")\nd2skip = TextCallout(\"warning\", \"If you skip it\", \"A patch release of a transitive dependency ships a regression overnight. Your code did not change, your tests passed yesterday, and today production breaks. You'll spend hours diffing your own diff — which is empty. This is the classic 'works on my machine' outage.\")\nd2cmd = CodeBlock(\"bash\", \"# Pin the toolchain first (.nvmrc keeps CI and laptops aligned)\\nnvm install && nvm use\\nnode --version && npm --version\\n\\n# Reproducible install from the lockfile only\\nnpm ci --no-audit --no-fund\\n\\n# Fail the build on known high-severity CVEs\\nnpm audit --audit-level=high\")\nd2note = TextContent(\"_Equivalents:_ `pnpm install --frozen-lockfile` · `yarn install --immutable` · `pip install -r requirements.txt --require-hashes` · `bundle install --deployment`\", \"small\")\n\nd3 = AccordionItem(\"d3\", \"Step 3 — Lint, typecheck, and test\", [d3why, d3skip, d3cmd, d3order, d3note])\nd3why = TextContent(\"**Why it matters**\\n\\nThis is your cheapest possible failure. A typecheck error costs ~20 seconds to find here; the same error found by a customer costs a rollback, an incident review, and trust. Order matters: run the fast, broad checks (lint, types) before the slow, narrow ones (integration, e2e) so the pipeline fails as early as possible.\")\nd3skip = TextCallout(\"danger\", \"If you skip it\", \"Broken code ships. Worse, once a team learns the test gate is optional or flaky, it gets bypassed routinely — and the suite decays into decoration. Skipping tests 'just this once, it's urgent' is how most preventable outages start.\")\nd3cmd = CodeBlock(\"bash\", \"# Fast checks first — seconds\\nnpm run lint # eslint . --max-warnings=0\\nnpm run typecheck # tsc --noEmit\\n\\n# Unit tests with coverage gate\\nnpm test -- --run --coverage --coverage.thresholds.lines=80\\n\\n# Integration + e2e last — minutes\\nnpm run test:integration\\nnpx playwright test --reporter=line\")\nd3order = Table([d3ColCheck, d3ColTime, d3ColCatches])\nd3ColCheck = Col(\"Check\", [\"Lint\", \"Typecheck\", \"Unit\", \"Integration\", \"E2E\"])\nd3ColTime = Col(\"Typical time\", [\"5s\", \"20s\", \"45s\", \"3m\", \"8m\"])\nd3ColCatches = Col(\"Catches\", [\"Style, dead code, footguns\", \"Contract & shape errors\", \"Logic regressions\", \"Wiring, SQL, serialization\", \"Real user journeys\"])\nd3note = TextContent(\"_Rule:_ zero tolerance for flaky tests. Quarantine or delete them — a test you ignore is worse than no test.\", \"small\")\n\nd4 = AccordionItem(\"d4\", \"Step 4 — Build the immutable artifact\", [d4why, d4skip, d4cmd, d4note])\nd4why = TextContent(\"**Why it matters**\\n\\nBuild once, promote everywhere. Tagging the image with the commit SHA (not `latest`) means staging and production run *literally the same bytes*, so a staging pass is real evidence about production. `latest` is a mutable pointer: two machines pulling `latest` an hour apart can run different code with no record of it.\")\nd4skip = TextCallout(\"danger\", \"If you skip it\", \"Rebuilding per environment reintroduces nondeterminism — different base image digest, different build cache, different timestamp. Staging goes green, production goes red, and nothing in your logs explains why. With `latest`, rollback is also impossible: there is no previous tag to point back to.\")\nd4cmd = CodeBlock(\"bash\", \"export REGISTRY=ghcr.io/acme\\nexport IMAGE=$REGISTRY/api\\nexport RELEASE_SHA=$(git rev-parse --short HEAD)\\n\\n# Build with reproducible, traceable tags\\ndocker build \\\\\\n --build-arg GIT_SHA=$RELEASE_SHA \\\\\\n --label org.opencontainers.image.revision=$RELEASE_SHA \\\\\\n -t $IMAGE:$RELEASE_SHA \\\\\\n -t $IMAGE:v2.14.0 \\\\\\n .\\n\\n# Push immutable tags (never push :latest as the deploy target)\\ndocker push $IMAGE:$RELEASE_SHA\\ndocker push $IMAGE:v2.14.0\\n\\n# Capture the digest — this is the true identity of the artifact\\ndocker inspect --format='{{index .RepoDigests 0}}' $IMAGE:$RELEASE_SHA\")\nd4note = TextContent(\"_Bonus:_ scan before promoting — `trivy image --exit-code 1 --severity CRITICAL $IMAGE:$RELEASE_SHA`\", \"small\")\n\nd5 = AccordionItem(\"d5\", \"Step 5 — Run database migrations\", [d5why, d5skip, d5cmd, d5rules, d5note])\nd5why = TextContent(\"**Why it matters**\\n\\nDuring a rolling deploy, old and new code run **simultaneously** against one database. Migrations must therefore be backwards-compatible: add columns as nullable, add tables and indexes before code reads them, and never drop or rename in the same release that stops using them. Always back up first — schema changes are the one deploy step that can destroy data rather than just break requests.\")\nd5skip = TextCallout(\"danger\", \"If you skip it\", \"New code queries a column that does not exist yet and every request 500s the instant the first pod goes live. Or you drop a column the still-running old pods depend on, taking down the version you were about to roll back to. This is the failure mode that turns a bad deploy into data loss.\")\nd5cmd = CodeBlock(\"bash\", \"# 1. Snapshot first — always\\npg_dump \\\"$DATABASE_URL\\\" -Fc -f \\\"backup-$(date +%F-%H%M).dump\\\"\\n\\n# 2. Review exactly what will run\\nnpx prisma migrate status\\nnpx prisma migrate diff \\\\\\n --from-url \\\"$DATABASE_URL\\\" \\\\\\n --to-migrations ./prisma/migrations \\\\\\n --script\\n\\n# 3. Apply forward-only, no interactive prompts\\nnpx prisma migrate deploy\\n\\n# 4. Build big indexes without locking writes\\npsql \\\"$DATABASE_URL\\\" -c \\\\\\n \\\"CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_orders_created_at ON orders (created_at);\\\"\")\nd5rules = TextCallout(\"info\", \"The expand / contract pattern\", \"Release N: add the new nullable column and dual-write. Release N+1: backfill and switch reads. Release N+2: stop writing the old column. Release N+3: drop it. Three boring deploys beat one heroic one.\")\nd5note = TextContent(\"_Equivalents:_ `rails db:migrate` · `alembic upgrade head` · `flyway migrate` · `golang-migrate up`\", \"small\")\n\nd6 = AccordionItem(\"d6\", \"Step 6 — Deploy to staging & smoke test\", [d6why, d6skip, d6cmd, d6note])\nd6why = TextContent(\"**Why it matters**\\n\\nUnit tests validate your code; staging validates your *environment* — secrets, DNS, TLS certs, IAM permissions, connection pools, sidecars, feature flags. These are the things that cannot be tested on a laptop and cause a large share of real incidents. A short smoke test against the deployed URL proves the wiring, not the logic.\")\nd6skip = TextCallout(\"warning\", \"If you skip it\", \"The image is fine but a missing environment variable, an expired secret, or a wrong database hostname takes production down on boot. You discover your misconfiguration from your users instead of from a pipeline step that costs four minutes.\")\nd6cmd = CodeBlock(\"bash\", \"# Deploy the exact artifact from step 4\\nkubectl -n staging set image deploy/api api=$IMAGE:$RELEASE_SHA --record\\nkubectl -n staging rollout status deploy/api --timeout=180s\\n\\n# Confirm the running pods report the expected commit\\nkubectl -n staging get pods -l app=api \\\\\\n -o jsonpath='{.items[*].spec.containers[0].image}'\\n\\n# Smoke test the real deployed surface\\ncurl -fsS https://staging.acme.dev/healthz | jq .\\ncurl -fsS https://staging.acme.dev/readyz\\ncurl -fsS -o /dev/null -w '%{http_code}\\\\n' https://staging.acme.dev/api/v1/orders\\n\\n# Critical-path e2e against staging\\nBASE_URL=https://staging.acme.dev npx playwright test --grep @smoke\")\nd6note = TextContent(\"_Health endpoint contract:_ `/healthz` = process alive. `/readyz` = dependencies reachable (DB, cache, queue). Kubernetes needs both to roll safely.\", \"small\")\n\nd7 = AccordionItem(\"d7\", \"Step 7 — Progressive rollout to production\", [d7why, d7skip, d7cmd, d7table, d7note])\nd7why = TextContent(\"**Why it matters**\\n\\nA canary limits the blast radius. Sending 5% of traffic to the new version exposes it to real production load, real data, and real edge cases while 95% of users stay on the known-good version. Combined with `maxUnavailable: 0` and readiness probes, this gives you a deploy that fails *small* and reverts in seconds instead of failing *total*.\")\nd7skip = TextCallout(\"danger\", \"If you skip it\", \"A big-bang deploy pushes the regression to 100% of users at once. Every request fails, every alert fires simultaneously, and you're now debugging under maximum pressure with no healthy version left serving traffic to compare against.\")\nd7cmd = CodeBlock(\"bash\", \"# Ensure zero-downtime rolling strategy is configured\\nkubectl -n prod patch deploy/api -p '{\\\"spec\\\":{\\\"strategy\\\":{\\\"rollingUpdate\\\":{\\\"maxUnavailable\\\":0,\\\"maxSurge\\\":1}}}}'\\n\\n# 1) Canary: one pod, ~5% of traffic\\nkubectl -n prod set image deploy/api-canary api=$IMAGE:$RELEASE_SHA\\nkubectl -n prod rollout status deploy/api-canary --timeout=180s\\n\\n# 2) Watch the canary for 10 minutes before widening\\nkubectl -n prod logs -l app=api,track=canary --since=10m -f | grep -Ei 'error|exception|panic'\\n\\n# 3) Widen only if error rate and p99 latency are flat\\nkubectl -n prod set image deploy/api api=$IMAGE:$RELEASE_SHA --record\\nkubectl -n prod rollout status deploy/api --timeout=300s\\n\\n# ABORT at any point\\nkubectl -n prod rollout undo deploy/api\")\nd7table = Table([d7ColStage, d7ColTraffic, d7ColHold, d7ColGate])\nd7ColStage = Col(\"Stage\", [\"Canary\", \"Wave 1\", \"Wave 2\", \"Full\"])\nd7ColTraffic = Col(\"Traffic\", [\"5%\", \"25%\", \"50%\", \"100%\"])\nd7ColHold = Col(\"Hold time\", [\"10 min\", \"10 min\", \"5 min\", \"—\"])\nd7ColGate = Col(\"Promote if\", [\"5xx rate flat, no new error signatures\", \"p99 latency within 10% of baseline\", \"No saturation on DB pool or CPU\", \"Dashboards green for 30 min\"])\nd7note = TextContent(\"_Automate the gates:_ Argo Rollouts or Flagger can promote and abort on Prometheus queries so no human has to squint at a graph.\", \"small\")\n\nd8 = AccordionItem(\"d8\", \"Step 8 — Verify, monitor, and record\", [d8why, d8skip, d8cmd, d8chart, d8note])\nd8why = TextContent(\"**Why it matters**\\n\\n'Rollout complete' means Kubernetes is happy, not that your users are. Most regressions surface 5–30 minutes in, when caches cool, cron jobs fire, and traffic patterns shift. Actively watching that window — and writing down what shipped — is what turns a deploy into something you can reason about later. The deploy log is the first thing you'll read during the next incident.\")\nd8skip = TextCallout(\"warning\", \"If you skip it\", \"A slow memory leak or a doubled query count goes unnoticed until it becomes a 2am page, now correlated with nothing because no one recorded that a deploy happened. Time-to-detect balloons from minutes to hours, and MTTR follows.\")\nd8cmd = CodeBlock(\"bash\", \"# Confirm every pod runs the intended SHA\\nkubectl -n prod get pods -l app=api \\\\\\n -o jsonpath='{range .items[*]}{.metadata.name}{\\\"\\\\t\\\"}{.spec.containers[0].image}{\\\"\\\\n\\\"}{end}'\\n\\n# Verify the deployed build identifies itself correctly\\ncurl -fsS https://api.acme.com/version | jq '{sha, version, builtAt}'\\n\\n# Watch the 30-minute window\\nkubectl -n prod get pods -l app=api -w # restarts / CrashLoopBackOff\\nkubectl -n prod top pods -l app=api # memory & CPU drift\\n\\n# Annotate dashboards and notify\\ncurl -fsS -X POST \\\"$GRAFANA_URL/api/annotations\\\" \\\\\\n -H \\\"Authorization: Bearer $GRAFANA_TOKEN\\\" \\\\\\n -H 'Content-Type: application/json' \\\\\\n -d \\\"{\\\\\\\"tags\\\\\\\":[\\\\\\\"deploy\\\\\\\",\\\\\\\"prod\\\\\\\"],\\\\\\\"text\\\\\\\":\\\\\\\"api v2.14.0 ($RELEASE_SHA)\\\\\\\"}\\\"\\n\\n# Publish release notes tied to the tag\\ngh release create v2.14.0 --generate-notes\")\nd8chart = Card([d8chartHeader, d8lineChart])\nd8chartHeader = CardHeader(\"Post-deploy watch window\", \"Error rate per minute — canary at T+0, full rollout at T+15\")\nd8lineChart = LineChart([\"T+0\", \"T+5\", \"T+10\", \"T+15\", \"T+20\", \"T+25\", \"T+30\"], [d8SeriesErr, d8SeriesBaseline], \"natural\", \"Minutes since canary\", \"Errors / min\")\nd8SeriesErr = Series(\"This release\", [3, 4, 3, 6, 5, 4, 3])\nd8SeriesBaseline = Series(\"7-day baseline\", [4, 4, 3, 4, 4, 3, 4])\nd8note = TextContent(\"_Definition of done:_ intended SHA on every pod, error rate within baseline, dashboard annotated, release notes published.\", \"small\")\n\nrollbackNote = Card([rollbackHeader, rollbackText, rollbackCmd, rollbackCallout], \"sunk\")\nrollbackHeader = CardHeader(\"The abort path\", \"Know this before you start — not while it's burning\")\nrollbackText = TextContent(\"Every step above is reversible except an unbacked destructive migration. Roll back **code** first, since it's instant and safe; only consider a schema rollback if the migration itself is the cause, and only from the step 5 snapshot. Decide with a timer, not a debate: if the canary is unhealthy for more than five minutes, revert and investigate on a healthy system.\")\nrollbackCmd = CodeBlock(\"bash\", \"# Instant code rollback to the previous ReplicaSet\\nkubectl -n prod rollout undo deploy/api\\nkubectl -n prod rollout status deploy/api --timeout=180s\\n\\n# Or pin explicitly to the last known-good SHA\\nkubectl -n prod set image deploy/api api=$IMAGE: --record\\n\\n# Inspect history to pick a target revision\\nkubectl -n prod rollout history deploy/api\")\nrollbackCallout = TextCallout(\"info\", \"Rehearse it\", \"A rollback you've never practiced is a hope, not a plan. Run `rollout undo` in staging once per release cycle so the muscle memory exists before you need it.\")\n\nfooterBtns = Buttons([btnCi, btnZero, btnRollback, btnDocs])\nbtnCi = Button(\"Turn this into a GitHub Actions workflow\", Action([@ToAssistant(\"Convert this 8-step deployment runbook into a complete GitHub Actions workflow file with separate jobs for build, staging, and production, including required approvals and OIDC auth to the registry.\")]), \"primary\")\nbtnZero = Button(\"Explain zero-downtime probes\", Action([@ToAssistant(\"Explain in detail how Kubernetes readiness, liveness, and startup probes plus maxUnavailable/maxSurge and preStop hooks combine to achieve true zero-downtime deploys. Include example YAML.\")]), \"secondary\")\nbtnRollback = Button(\"Deep dive on safe migrations\", Action([@ToAssistant(\"Give me a detailed guide to the expand/contract database migration pattern with concrete SQL for adding, backfilling, renaming, and dropping columns safely across multiple releases.\")]), \"secondary\")\nbtnDocs = Button(\"Kubernetes rollout docs\", Action([@OpenUrl(\"https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#rolling-back-a-deployment\")]), \"tertiary\")","promptTokens":6969,"completionTokens":7880,"latencyMs":98890,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/realistic","repeat":1,"output":"root = Stack([resultsCard])\nresultsCard = Card([header, summary, resultsTable], \"column\", \"m\")\nheader = CardHeader(\"Q3 Sales Team Results\", \"Mid-size software company performance across EMEA and APAC\")\nsummary = TextContent(\"Eight account executives closed 176 deals and delivered $4.58M in Q3 revenue.\", \"default\")\nresultsTable = Table([repCol, regionCol, dealsCol, revenueCol, quotaCol])\nrepCol = Col(\"Rep Name\", [\"Amelia Hart\", \"Lukas Weber\", \"Sofia Moretti\", \"James Okafor\", \"Priya Nair\", \"Kenji Tanaka\", \"Mei Lin\", \"Arjun Kapoor\"])\nregionCol = Col(\"Region\", [\"EMEA\", \"EMEA\", \"EMEA\", \"EMEA\", \"APAC\", \"APAC\", \"APAC\", \"APAC\"])\ndealsCol = Col(\"Deals Closed\", [28, 24, 21, 19, 26, 22, 20, 16], \"number\")\nrevenueCol = Col(\"Revenue\", [\"$742,000\", \"$615,000\", \"$538,000\", \"$461,000\", \"$684,000\", \"$571,000\", \"$512,000\", \"$458,000\"])\nquotaCol = Col(\"Against Quota\", [Tag(\"124% — Exceeded\", null, \"sm\", \"success\"), Tag(\"112% — Exceeded\", null, \"sm\", \"success\"), Tag(\"108% — Exceeded\", null, \"sm\", \"success\"), Tag(\"96% — Near target\", null, \"sm\", \"warning\"), Tag(\"118% — Exceeded\", null, \"sm\", \"success\"), Tag(\"104% — Exceeded\", null, \"sm\", \"success\"), Tag(\"98% — Near target\", null, \"sm\", \"warning\"), Tag(\"91% — Below target\", null, \"sm\", \"danger\")])","promptTokens":4425,"completionTokens":502,"latencyMs":35690,"finishReason":"stop"} +{"model":"anthropic/claude-opus-5","format":"openui-v2","scenario":"tasklist/adversarial","repeat":3,"output":"root = Stack([header, overview, stepsOutline, detailSection, gateTable, rollbackCard, footerNote, footerButtons], \"column\", \"l\")\n\nheader = CardHeader(\"Production Deployment Runbook\", \"9 ordered steps — with rationale, failure modes, and copy-paste commands\")\n\noverview = Callout(\"info\", \"Read the ordering first\", \"Steps 1-4 are verification gates, 5-6 are the actual release, 7-9 are confirmation and cleanup. Every step is reversible until step 6.\")\n\nstepsOutline = Card([outlineHeader, outlineSteps], \"sunk\")\noutlineHeader = CardHeader(\"The short version\", \"Expand any step below for the full explanation and commands\")\noutlineSteps = Steps([s1, s2, s3, s4, s5, s6, s7, s8, s9])\n\ns1 = StepsItem(\"1. Freeze and sync the branch\", \"Confirm main is green and your local clone matches the remote exactly.\")\ns2 = StepsItem(\"2. Install from the lockfile\", \"Reproduce the exact dependency tree CI used, not a fresh resolution.\")\ns3 = StepsItem(\"3. Run the full test and lint suite\", \"Type check, unit, integration, lint — locally, before the pipeline.\")\ns4 = StepsItem(\"4. Build the production artifact\", \"Compile once, deploy that same artifact everywhere.\")\ns5 = StepsItem(\"5. Run database migrations\", \"Backwards-compatible schema changes go out before the new code.\")\ns6 = StepsItem(\"6. Deploy behind a health check\", \"Rolling or blue/green release with automatic abort on failure.\")\ns7 = StepsItem(\"7. Smoke test production\", \"Hit the real endpoints on the real host with real DNS.\")\ns8 = StepsItem(\"8. Watch metrics for 15 minutes\", \"Error rate, p95 latency, saturation. Do not walk away.\")\ns9 = StepsItem(\"9. Tag, release notes, cleanup\", \"Make the deploy findable six months from now.\")\n\ndetailSection = Accordion([a1, a2, a3, a4, a5, a6, a7, a8, a9])\n\na1 = AccordionItem(\"step1\", \"Step 1 — Freeze and sync the branch\", [w1, k1, c1])\nw1 = TextContent(\"**Why it matters.** A deploy is a claim about a specific commit. If your working tree has uncommitted changes, or your local `main` is three commits behind the remote, then the thing you tested is not the thing you ship. You also want CI to have already passed on that exact SHA — CI runs on a clean machine and catches the \\\"works on my laptop\\\" class of bug that local runs never will.\")\nk1 = TextContent(\"**If you skip it.** You ship a merge of your stale branch and someone else's new code that no machine has ever built together. Classic outcome: a teammate merged a migration an hour ago, you deploy without it, and every request that touches the new column 500s. Debugging is brutal because the deployed SHA does not exist in anyone's history.\")\nc1 = CodeBlock(\"bash\", \"# Confirm nothing uncommitted or stashed\\ngit status --porcelain\\ngit stash list\\n\\n# Sync exactly with remote main\\ngit checkout main\\ngit fetch origin --prune\\ngit reset --hard origin/main\\n\\n# Record the SHA you are shipping — you will need it for rollback\\nexport RELEASE_SHA=$(git rev-parse HEAD)\\necho \\\"Deploying $RELEASE_SHA\\\"\\n\\n# Verify CI is green for this exact SHA\\ngh run list --commit \\\"$RELEASE_SHA\\\" --limit 5\\ngh pr checks \\\"$RELEASE_SHA\\\" 2>/dev/null || true\")\n\na2 = AccordionItem(\"step2\", \"Step 2 — Install from the lockfile\", [w2, k2, c2])\nw2 = TextContent(\"**Why it matters.** `npm install` is allowed to *resolve* dependencies — it can pick up a newer patch release that matches your semver range and rewrite the lockfile. `npm ci` is not allowed to do that: it deletes `node_modules` and installs the exact tree in `package-lock.json`, failing loudly if the lockfile and `package.json` disagree. Reproducibility is the entire point of a lockfile, and `install` quietly opts you out of it.\")\nk2 = TextContent(\"**If you skip it.** You get a supply-chain roulette wheel. A transitive dependency ships a broken patch version at 14:00, you install at 14:05, and your build now contains code that was never tested by anyone in your org. The failure is often not at build time but at runtime, in production, and it is not reproducible on any other machine because their lockfile-less install picked different versions.\")\nc2 = CodeBlock(\"bash\", \"# Pin the runtime version first\\nnode --version # must match .nvmrc / engines field\\nnvm use\\n\\n# Deterministic install — fails if lockfile is out of sync\\nnpm ci\\n\\n# Audit for known vulnerabilities; block on high severity\\nnpm audit --audit-level=high\\n\\n# Python equivalents\\n# pip install --require-hashes -r requirements.lock\\n# uv sync --frozen\\n# poetry install --sync\")\n\na3 = AccordionItem(\"step3\", \"Step 3 — Run the full test and lint suite\", [w3, k3, c3])\nw3 = TextContent(\"**Why it matters.** This is your last cheap failure. A bug caught here costs you 90 seconds; the same bug caught in production costs you an incident channel, a rollback, a postmortem, and whatever your error budget was worth. Run the *full* suite, not the subset you think is relevant — the tests you did not think were relevant are exactly the ones that catch unintended coupling.\")\nk3 = TextContent(\"**If you skip it.** You are outsourcing correctness to your users. Type errors and lint failures are the mild case. The severe case is an integration test you skipped because \\\"it's just a config change\\\" — and the config change pointed the cache at the wrong Redis cluster. Also note: skipping locally to \\\"let CI catch it\\\" wastes 10 minutes of pipeline time per attempt and pollutes the commit history with `fix lint` commits.\")\nc3 = CodeBlock(\"bash\", \"# Static analysis\\nnpm run lint\\nnpm run typecheck # tsc --noEmit\\n\\n# Full suite, no bail, with coverage floor enforced\\nnpm test -- --run --coverage\\n\\n# Integration tests against ephemeral services\\ndocker compose -f docker-compose.test.yml up -d\\nnpm run test:integration\\ndocker compose -f docker-compose.test.yml down -v\\n\\n# Fail fast on any non-zero exit before continuing\\nif [ $? -ne 0 ]; then echo \\\"ABORT: tests failed\\\"; exit 1; fi\")\n\na4 = AccordionItem(\"step4\", \"Step 4 — Build the production artifact once\", [w4, k4, c4])\nw4 = TextContent(\"**Why it matters.** Build once, promote the same bytes through every environment. If staging builds its own image and production builds another, you have two artifacts and only tested one. Tagging the image with the git SHA (not just `latest`) makes the artifact traceable back to source and makes rollback a one-line tag change instead of an archaeology project.\")\nk4 = TextContent(\"**If you skip it** — or rebuild per environment — you get the \\\"it passed in staging\\\" paradox. A base image floated to a new minor version between the two builds, or an env var was inlined at build time with the staging value. Using `latest` compounds it: you cannot tell which code is running, and rollback has no target because the previous `latest` was overwritten.\")\nc4 = CodeBlock(\"bash\", \"# Build with the SHA baked into the tag and the image labels\\ndocker build \\\\\\n --platform linux/amd64 \\\\\\n --build-arg NODE_ENV=production \\\\\\n --label \\\"org.opencontainers.image.revision=$RELEASE_SHA\\\" \\\\\\n -t registry.example.com/api:$RELEASE_SHA \\\\\\n -t registry.example.com/api:latest .\\n\\n# Verify the artifact actually boots before pushing\\ndocker run --rm -e NODE_ENV=production \\\\\\n registry.example.com/api:$RELEASE_SHA node -e \\\"require('./dist/server.js')\\\"\\n\\n# Push both tags\\ndocker push registry.example.com/api:$RELEASE_SHA\\ndocker push registry.example.com/api:latest\")\n\na5 = AccordionItem(\"step5\", \"Step 5 — Run database migrations (expand, never contract)\", [w5, k5, c5])\nw5 = TextContent(\"**Why it matters.** During a rolling deploy, old and new code run **simultaneously** against **one** database. So migrations must be backwards compatible: add nullable columns, add tables, add indexes concurrently. Never rename or drop in the same release that stops using the column — split that into a later \\\"contract\\\" deploy after the old code is fully gone. Always take a snapshot first, because a bad migration is the one failure mode that rollback cannot fix.\")\nk5 = TextContent(\"**If you skip it** (or do it in the wrong order): if you migrate *after* deploying, new code queries a column that does not exist and every request fails until the migration lands. If you drop a column in the same release, the old pods still selecting it crash-loop mid-rollout. If you build an index non-concurrently on Postgres, you take a write lock on a hot table and effectively cause an outage with a \\\"safe\\\" migration. And without a snapshot, a destructive migration means data loss, not downtime.\")\nc5 = CodeBlock(\"bash\", \"# 1. Snapshot first — non-negotiable\\naws rds create-db-cluster-snapshot \\\\\\n --db-cluster-identifier prod-api \\\\\\n --db-cluster-snapshot-identifier \\\"pre-$RELEASE_SHA\\\"\\naws rds wait db-cluster-snapshot-available \\\\\\n --db-cluster-snapshot-identifier \\\"pre-$RELEASE_SHA\\\"\\n\\n# 2. Review the SQL that will actually execute\\nnpx prisma migrate diff \\\\\\n --from-url \\\"$DATABASE_URL\\\" \\\\\\n --to-schema-datamodel prisma/schema.prisma \\\\\\n --script\\n\\n# 3. Apply — additive changes only\\nnpx prisma migrate deploy\\n# Rails: bundle exec rails db:migrate\\n# Django: python manage.py migrate --noinput\\n\\n# 4. Postgres: build hot indexes without locking writes\\npsql \\\"$DATABASE_URL\\\" -c \\\\\\n \\\"CREATE INDEX CONCURRENTLY IF NOT EXISTS idx_events_user_id ON events(user_id);\\\"\")\n\na6 = AccordionItem(\"step6\", \"Step 6 — Deploy behind a health check\", [w6, k6, c6])\nw6 = TextContent(\"**Why it matters.** A health check plus a readiness gate is what converts a bad deploy from an outage into a non-event. The orchestrator sends traffic to a new pod only after it reports ready, and if the new pods never become ready it stops the rollout with the old version still serving. `maxUnavailable: 0` guarantees you never dip below current capacity. This is the step where automation earns its keep — a human watching logs cannot react in the 200ms it takes to fail thousands of requests.\")\nk6 = TextContent(\"**If you skip it.** Traffic hits containers that are still warming caches or have not opened their DB pool, producing a burst of 502s at the start of every deploy. Worse, without a readiness gate a fully broken image replaces every healthy pod and you get a total outage that only ends when a human notices and manually rolls back. With the gate, the rollout simply stalls and pages you while users see nothing.\")\nc6 = CodeBlock(\"bash\", \"# Roll out the pinned SHA (never :latest in the apply)\\nkubectl set image deployment/api \\\\\\n api=registry.example.com/api:$RELEASE_SHA \\\\\\n --record\\n\\n# Block until healthy, abort automatically on timeout\\nkubectl rollout status deployment/api --timeout=300s || {\\n echo \\\"ROLLOUT FAILED — rolling back\\\"\\n kubectl rollout undo deployment/api\\n exit 1\\n}\\n\\n# Confirm every pod is on the new SHA\\nkubectl get pods -l app=api \\\\\\n -o jsonpath='{range .items[*]}{.metadata.name}{\\\"\\\\t\\\"}{.spec.containers[0].image}{\\\"\\\\n\\\"}{end}'\")\n\na7 = AccordionItem(\"step7\", \"Step 7 — Smoke test production\", [w7, k7, c7])\nw7 = TextContent(\"**Why it matters.** Every prior test ran against a fake. Only production has the real load balancer, the real TLS certificate, the real DNS record, the real secrets, and the real database. Smoke tests validate the *wiring*, not the logic: can the app reach the DB, is the cert valid, does the CDN serve the new asset hashes, does one real login work end to end. Keep it to under a minute of checks on the critical path.\")\nk7 = TextContent(\"**If you skip it.** The deploy reports success and the app is broken in a way no test could catch: an env var missing from the prod secret store so the payment provider is unreachable, a cert that expired this morning, a CORS origin still pointing at staging. Health checks say `200 OK` because `/healthz` does not touch any of that. You find out from a customer, which is the most expensive possible detection channel.\")\nc7 = CodeBlock(\"bash\", \"# Version endpoint must report the SHA you just shipped\\ncurl -fsS https://api.example.com/version | jq -e \\\\\\n --arg sha \\\"$RELEASE_SHA\\\" '.revision == $sha'\\n\\n# Deep health check: DB, cache, and downstream dependencies\\ncurl -fsS https://api.example.com/healthz?deep=1 | jq .\\n\\n# Critical path with a real auth token\\ncurl -fsS -o /dev/null -w \\\"%{http_code} %{time_total}s\\\\n\\\" \\\\\\n -H \\\"Authorization: Bearer $SMOKE_TOKEN\\\" \\\\\\n https://api.example.com/v1/me\\n\\n# TLS expiry sanity check\\necho | openssl s_client -connect api.example.com:443 2>/dev/null \\\\\\n | openssl x509 -noout -dates\")\n\na8 = AccordionItem(\"step8\", \"Step 8 — Watch metrics for 15 minutes\", [w8, k8, c8])\nw8 = TextContent(\"**Why it matters.** A whole class of regressions is invisible at t=0 and obvious at t=10min: memory leaks, connection-pool exhaustion, an N+1 query that only hurts under real concurrency, a cache that was warm for the old code and is now cold. The four signals worth staring at are error rate, p95/p99 latency, saturation (CPU/memory/pool), and throughput. Compare against the pre-deploy baseline, not against zero.\")\nk8 = TextContent(\"**If you skip it.** You declare victory and go to lunch while pods slowly OOM-kill and restart in a loop, or latency creeps from 80ms to 4s as the DB connection pool drains. By the time alerts fire you have lost the causal link to the deploy, and the person on call — who did not deploy — spends an hour rediscovering that a release happened. Detection time is the single biggest lever on incident cost.\")\nc8 = CodeBlock(\"bash\", \"# Live error rate and restarts\\nkubectl get pods -l app=api -w\\nkubectl logs -l app=api --since=15m -f | grep -Ei \\\"error|fatal|panic|timeout\\\"\\n\\n# Restart counts should stay flat at their pre-deploy value\\nkubectl get pods -l app=api \\\\\\n -o jsonpath='{range .items[*]}{.metadata.name}{\\\" restarts=\\\"}{.status.containerStatuses[0].restartCount}{\\\"\\\\n\\\"}{end}'\\n\\n# 5xx rate from Prometheus\\ncurl -sG http://prometheus:9090/api/v1/query \\\\\\n --data-urlencode 'query=sum(rate(http_requests_total{app=\\\"api\\\",status=~\\\"5..\\\"}[5m]))\\n / sum(rate(http_requests_total{app=\\\"api\\\"}[5m]))' | jq '.data.result'\\n\\n# p95 latency\\ncurl -sG http://prometheus:9090/api/v1/query \\\\\\n --data-urlencode 'query=histogram_quantile(0.95,\\n sum by (le) (rate(http_request_duration_seconds_bucket{app=\\\"api\\\"}[5m])))' | jq '.data.result'\")\n\na9 = AccordionItem(\"step9\", \"Step 9 — Tag, write release notes, clean up\", [w9, k9, c9])\nw9 = TextContent(\"**Why it matters.** This is the step everyone skips and everyone later wishes they had not. An annotated git tag turns \\\"which version is in prod?\\\" from a Slack thread into a one-command answer. Release notes give the next on-call engineer the diff between working and broken. Removing feature flags and finished migration scaffolding prevents dead code paths from becoming permanent load-bearing mysteries.\")\nk9 = TextContent(\"**If you skip it.** Six weeks later something breaks and nobody can bisect because there are no tags — only a wall of merge commits. Feature flags accumulate until you have 2^n untested combinations of behaviour and no one dares delete any of them. Old images fill the registry and inflate your bill. None of this hurts today, which is precisely why it compounds.\")\nc9 = CodeBlock(\"bash\", \"# Annotated, pushed tag pointing at the deployed SHA\\ngit tag -a \\\"v$(date +%Y.%m.%d)-$(git rev-parse --short HEAD)\\\" \\\\\\n -m \\\"Production deploy $RELEASE_SHA\\\" \\\"$RELEASE_SHA\\\"\\ngit push origin --tags\\n\\n# Generate release notes from the previous tag\\ngh release create \\\"v$(date +%Y.%m.%d)\\\" --generate-notes --target \\\"$RELEASE_SHA\\\"\\n\\n# Retire flags that are now fully rolled out\\n# (delete the flag AND the now-dead branch of the conditional)\\n\\n# Prune old images, keep the last 10 for rollback\\ncrane ls registry.example.com/api | head -n -10 \\\\\\n | xargs -I{} crane delete registry.example.com/api:{}\")\n\ngateTable = Card([gateHeader, gateTableInner])\ngateHeader = CardHeader(\"Abort gates at a glance\", \"If any gate fails, stop — do not proceed to the next step\")\ngateTableInner = Table([colStep, colGate, colOnFail, colReversible])\ncolStep = Col(\"Step\", [\"1 Sync\", \"2 Install\", \"3 Tests\", \"4 Build\", \"5 Migrate\", \"6 Deploy\", \"7 Smoke\", \"8 Monitor\"])\ncolGate = Col(\"Gate\", [\"git status clean, CI green\", \"npm ci exits 0, no high CVEs\", \"lint + types + tests pass\", \"image boots locally\", \"snapshot exists, additive only\", \"rollout status ready in 300s\", \"version endpoint returns SHA\", \"5xx and p95 within baseline\"])\ncolOnFail = Col(\"On failure\", [\"Commit or reset, wait for CI\", \"Fix lockfile, re-run\", \"Fix code, restart from step 1\", \"Fix Dockerfile, rebuild\", \"Do not deploy; restore snapshot\", \"kubectl rollout undo\", \"kubectl rollout undo\", \"kubectl rollout undo, then investigate\"])\ncolReversible = Col(\"Reversible\", [\"Yes\", \"Yes\", \"Yes\", \"Yes\", \"Only via snapshot\", \"Yes\", \"Yes\", \"Yes\"])\n\nrollbackCard = Card([rbHeader, rbWarn, rbText, rbCode], \"sunk\")\nrbHeader = CardHeader(\"The rollback path\", \"Know this before you start, not while the graphs are red\")\nrbWarn = Callout(\"warning\", \"Code rolls back. Data does not.\", \"Reverting the deployment is instant and safe. Reverting a destructive migration means restoring a snapshot and losing every write since it was taken. This is the entire reason step 5 insists on additive-only changes.\")\nrbText = TextContent(\"Decide the rollback trigger **before** you deploy and write it down: for example, *\\\"5xx rate above 1% for 2 consecutive minutes, or p95 above 800ms.\\\"* A pre-committed threshold removes the in-the-moment temptation to wait one more minute and see if it recovers. Roll back first, diagnose second — the deploy is not the place to debug.\")\nrbCode = CodeBlock(\"bash\", \"# Immediate revert to the previous ReplicaSet\\nkubectl rollout undo deployment/api\\nkubectl rollout status deployment/api --timeout=300s\\n\\n# Or pin explicitly to a known-good SHA\\nkubectl set image deployment/api api=registry.example.com/api:$LAST_GOOD_SHA\\n\\n# Inspect history to pick a revision\\nkubectl rollout history deployment/api\\nkubectl rollout undo deployment/api --to-revision=7\\n\\n# Database restore — last resort, expect data loss\\naws rds restore-db-cluster-from-snapshot \\\\\\n --db-cluster-identifier prod-api-restored \\\\\\n --snapshot-identifier \\\"pre-$RELEASE_SHA\\\"\")\n\nfooterNote = TextCallout(\"info\", \"Automate the gates, not the judgement\", \"Steps 1-7 should live in a pipeline script so they cannot be skipped under pressure. Steps 8 and 9 stay human: a person decides whether the graphs look right, and a person writes the notes the next on-call engineer will read.\")\n\nfooterButtons = Buttons([btn1, btn2, btn3, btn4])\nbtn1 = Button(\"Turn this into a CI pipeline file\", Action([@ToAssistant(\"Write this deployment runbook as a GitHub Actions workflow with each step as a job, including the abort gates and automatic rollback.\")]), \"primary\")\nbtn2 = Button(\"Explain blue/green vs canary\", Action([@ToAssistant(\"Compare blue/green, canary, and rolling deployments in detail — when to use each, and how step 6 changes for each strategy.\")]), \"secondary\")\nbtn3 = Button(\"Deep dive on safe migrations\", Action([@ToAssistant(\"Explain the expand/contract migration pattern in detail with concrete SQL examples for renaming a column across three deploys with zero downtime.\")]), \"secondary\")\nbtn4 = Button(\"Build the monitoring dashboard\", Action([@ToAssistant(\"Give me the Prometheus queries and Grafana panel definitions for the four signals in step 8, with alert rules for the rollback thresholds.\")]), \"tertiary\")","promptTokens":6969,"completionTokens":8118,"latencyMs":107069,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/minimal","repeat":0,"output":"root = Stack([revenueCard])\nrevenueCard = Card([revenueHeader, revenueChart])\nrevenueHeader = CardHeader(\"Monthly Revenue\", \"October 2025 – March 2026\")\nrevenueChart = BarChart([\"Oct\", \"Nov\", \"Dec\", \"Jan\", \"Feb\", \"Mar\"], [revenueSeries], \"grouped\", \"Month\", \"Revenue ($)\")\nrevenueSeries = Series(\"Revenue\", [84200, 91750, 108400, 96300, 112800, 124600])","promptTokens":4392,"completionTokens":194,"latencyMs":3085,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/minimal","repeat":1,"output":"root = Stack([revenueCard])\nrevenueCard = Card([revenueHeader, revenueChart, revenueNote])\nrevenueHeader = CardHeader(\"Monthly Revenue\", \"Last 6 months\")\nrevenueChart = BarChart(months, [revenueSeries], \"grouped\", \"Month\", \"Revenue ($)\")\nrevenueNote = TextContent(\"Revenue has grown steadily, with the strongest result in August.\", \"small\")\nmonths = [\"March\", \"April\", \"May\", \"June\", \"July\", \"August\"]\nrevenueSeries = Series(\"Revenue\", [84200, 91750, 88900, 103400, 112650, 121300])","promptTokens":4392,"completionTokens":143,"latencyMs":2630,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/minimal","repeat":2,"output":"root = Stack([revenueCard])\nrevenueCard = Card([revenueHeader, revenueChart, revenueNote])\nrevenueHeader = CardHeader(\"Monthly Revenue\", \"Last 6 months\")\nrevenueChart = BarChart(months, [revenueSeries], \"grouped\", \"Month\", \"Revenue (USD)\")\nrevenueNote = TextContent(\"Total revenue: $867,000\", \"small\")\nmonths = [\"October\", \"November\", \"December\", \"January\", \"February\", \"March\"]\nrevenueSeries = Series(\"Revenue\", [118000, 132000, 156000, 141000, 154000, 166000])","promptTokens":4392,"completionTokens":140,"latencyMs":2351,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/minimal","repeat":3,"output":"root = Stack([chartCard])\nchartCard = Card([chartHeader, revenueChart])\nchartHeader = CardHeader(\"Monthly Revenue\", \"Last 6 months\")\nrevenueChart = BarChart(months, [revenueSeries], \"grouped\", \"Month\", \"Revenue (USD)\")\nmonths = [\"October\", \"November\", \"December\", \"January\", \"February\", \"March\"]\nrevenueSeries = Series(\"Revenue\", [84200, 91750, 108400, 96400, 112800, 121600])","promptTokens":4392,"completionTokens":115,"latencyMs":2676,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/adversarial","repeat":0,"output":"root = Stack([pageHeader, recommendation, scoreCard, overviewCard, pricingTabs, capabilityTabs, caveats], \"column\", \"l\")\npageHeader = Card([title, subtitle], \"clear\", \"column\", \"xs\")\ntitle = TextContent(\"Top cloud providers compared\", \"large-heavy\")\nsubtitle = TextContent(\"Sorted by overall enterprise capability and global ecosystem: AWS, Microsoft Azure, Google Cloud, Oracle Cloud Infrastructure, and Alibaba Cloud. Prices and service limits vary by region, contract, currency, and usage pattern.\", \"default\")\nrecommendation = Callout(\"success\", \"My pick: AWS for the broadest default choice\", \"AWS is the best all-around selection when you need the widest service catalog, global reach, mature partners, and strong support options. Choose Azure for Microsoft-centric enterprises, Google Cloud for data/AI and Kubernetes-heavy teams, OCI for Oracle estate economics, and Alibaba Cloud for China-focused workloads.\")\nscoreCard = Card([scoreHeader, scoreTable], \"card\")\nscoreHeader = CardHeader(\"Quick decision scorecard\", \"Higher is stronger for a typical global enterprise; relative assessments, not vendor-published benchmarks.\")\nscoreTable = Table([scoreProvider, scoreBestFor, scoreReach, scoreCatalog, scoreDataAi, scoreEnterprise, scoreValue])\nscoreProvider = Col(\"Provider\", [\"1. AWS\", \"2. Microsoft Azure\", \"3. Google Cloud\", \"4. Oracle Cloud Infrastructure\", \"5. Alibaba Cloud\"])\nscoreBestFor = Col(\"Best fit\", [\"Broadest default / startups to enterprise\", \"Microsoft and hybrid enterprise\", \"Analytics, AI, Kubernetes\", \"Oracle databases and predictable networking\", \"China and Asia-first expansion\"])\nscoreReach = Col(\"Global reach\", [\"Excellent\", \"Excellent\", \"Very strong\", \"Strong\", \"Strong in Asia\"])\nscoreCatalog = Col(\"Service breadth\", [\"Excellent\", \"Excellent\", \"Very strong\", \"Moderate\", \"Strong\"])\nscoreDataAi = Col(\"Data & AI\", [\"Very strong\", \"Very strong\", \"Excellent\", \"Strong\", \"Strong\"])\nscoreEnterprise = Col(\"Enterprise integration\", [\"Excellent\", \"Excellent\", \"Very strong\", \"Strong\", \"Strong\"])\nscoreValue = Col(\"Relative value\", [\"Good at scale\", \"Good with existing licenses\", \"Good for data workloads\", \"Excellent for Oracle/networking\", \"Competitive in target regions\"])\noverviewCard = Card([overviewHeader, overviewTable], \"card\")\noverviewHeader = CardHeader(\"Infrastructure, reliability, and core platform comparison\", \"Publicly available positioning; exact availability zones and regional counts change frequently.\")\noverviewTable = Table([providerCol, regionsCol, zonesCol, slaCol, computeCol, containersCol, serverlessCol, storageCol, databaseCol, networkCol])\nproviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nregionsCol = Col(\"Regions / geography\", [\"~33 regions globally\", \"60+ regions; broad enterprise footprint\", \"40+ regions\", \"50+ public cloud regions\", \"30+ regions; strongest in China/APAC\"])\nzonesCol = Col(\"Availability domains\", [\"100+ Availability Zones\", \"Multiple zones in many regions\", \"3+ zones in many regions\", \"Availability Domains + Fault Domains\", \"Multiple zones in major regions\"])\nslaCol = Col(\"Typical multi-zone compute SLA\", [\"Up to 99.99%\", \"Up to 99.99%\", \"Up to 99.99%\", \"Up to 99.99%\", \"Typically up to 99.99%\"])\ncomputeCol = Col(\"VM / bare metal\", [\"EC2; extensive instance families\", \"Azure VMs; strong Windows integration\", \"Compute Engine; custom machine types\", \"Flexible VMs and bare metal\", \"ECS and bare metal\"])\ncontainersCol = Col(\"Managed Kubernetes\", [\"EKS\", \"AKS\", \"GKE\", \"OKE\", \"ACK\"])\nserverlessCol = Col(\"Functions\", [\"Lambda\", \"Azure Functions\", \"Cloud Functions / Cloud Run\", \"OCI Functions\", \"Function Compute\"])\nstorageCol = Col(\"Object storage\", [\"S3\", \"Blob Storage\", \"Cloud Storage\", \"Object Storage\", \"OSS\"])\ndatabaseCol = Col(\"Managed data\", [\"RDS, Aurora, DynamoDB, Redshift\", \"Azure SQL, Cosmos DB, Fabric\", \"Cloud SQL, Spanner, BigQuery\", \"Autonomous DB, Exadata\", \"ApsaraDB, PolarDB, AnalyticDB\"])\nnetworkCol = Col(\"Private connectivity\", [\"Direct Connect\", \"ExpressRoute\", \"Cloud Interconnect\", \"FastConnect\", \"Express Connect\"])\npricingTabs = Tabs([pricingTab, freeTab, egressTab, commitmentTab])\npricingTab = TabItem(\"pricing\", \"Compute pricing\", [pricingIntro, pricingTable])\npricingIntro = TextContent(\"Indicative on-demand Linux VM economics only. Compare equivalent CPU, memory, disk, operating system, availability design, and discounts before procurement.\", \"small\")\npricingTable = Table([priceProvider, priceOnDemand, priceSpot, priceSavings, priceBilling, priceFinops])\npriceProvider = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\npriceOnDemand = Col(\"Typical entry VM\", [\"t3/t4g-class often ~$0.01–$0.03/hr\", \"B-series often ~$0.01–$0.03/hr\", \"E2-class often ~$0.01–$0.03/hr\", \"Flexible shapes can be highly competitive\", \"Burstable ECS often ~$0.01–$0.03/hr\"])\npriceSpot = Col(\"Interruptible capacity\", [\"Spot; discounts can reach ~90%\", \"Spot VMs; often up to ~90%\", \"Spot VMs; often 60–91%\", \"Preemptible capacity available\", \"Preemptible instances available\"])\npriceSavings = Col(\"Commitment models\", [\"Savings Plans, Reserved Instances\", \"Reservations, Savings Plan\", \"Committed Use Discounts\", \"Annual / universal credits\", \"Savings plans, subscriptions\"])\npriceBilling = Col(\"Billing granularity\", [\"Mostly per-second, 60-sec minimum\", \"Mostly per-second; some exceptions\", \"Per-second, 1-min minimum\", \"Per-second for many services\", \"Depends on service / region\"])\npriceFinops = Col(\"Cost-management strength\", [\"Excellent tooling; complexity is high\", \"Strong with Microsoft licensing\", \"Excellent sustained-use mechanics\", \"Simple networking and Oracle bundles\", \"Competitive regional offers\"])\nfreeTab = TabItem(\"free\", \"Free tier & trials\", [freeIntro, freeTable])\nfreeIntro = TextContent(\"Free tiers are promotional and change regularly. Always confirm current eligibility, duration, quotas, and overage behavior before deploying.\", \"small\")\nfreeTable = Table([freeProvider, freeOffer, freeCompute, freeServerless, freeStorage, freeNotes])\nfreeProvider = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nfreeOffer = Col(\"Main offer\", [\"12-month offers plus always-free services\", \"New-account credit plus selected free services\", \"New-customer credit plus always-free quotas\", \"Always Free services and trial credits\", \"Free trial credits and limited free products\"])\nfreeCompute = Col(\"VM allowance\", [\"Small eligible VM quota for 12 months\", [\"Limited B-series-style quota / credit\"], \"Small E2 micro instances always free in select US regions\", \"Always-free Ampere / micro VM capacity subject to limits\", \"Trial-dependent\"])\nfreeServerless = Col(\"Functions\", [\"1M requests/month plus compute quota\", \"1M requests/month consumption grant\", \"2M invocations/month\", \"Monthly invocation / execution allocations\", \"Monthly request and compute quotas\"])\nfreeStorage = Col(\"Object storage\", [\"Small S3 allowance\", \"Small Blob allowance\", \"Small Cloud Storage allowance\", \"Always-free object storage allocation\", \"Trial-dependent\"])\nfreeNotes = Col(\"Watch-outs\", [\"Charges begin after quota or term\", \"Credit expiration and subscription conversion\", \"Project billing account required\", \"Capacity availability can constrain free VMs\", \"Availability varies materially by market\"])\negressTab = TabItem(\"egress\", \"Network & egress\", [egressIntro, egressTable])\negressIntro = TextContent(\"Public internet egress is a major cost driver. Rates are tiered and destination-specific; private links, CDN, same-region architecture, and committed spend can change the result.\", \"small\")\negressTable = Table([egressProvider, egressInternet, egressIntra, egressPrivate, egressCdn, egressTakeaway])\negressProvider = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\negressInternet = Col(\"Internet egress\", [\"Often starts around $0.09/GB in US tiers\", \"Often around $0.08–$0.09/GB in common tiers\", \"Often around $0.08–$0.12/GB depending destination\", \"Notably generous: first 10 TB/month outbound often free\", \"Region and destination dependent\"])\negressIntra = Col(\"Cross-zone / regional\", [\"Cross-AZ traffic commonly billed\", \"Zone and regional traffic may be billed\", \"Same-region rules vary by product\", \"Generally simpler / favorable networking\", \"Varies by region and product\"])\negressPrivate = Col(\"Dedicated connection\", [\"Direct Connect\", \"ExpressRoute\", \"Cloud Interconnect\", \"FastConnect\", \"Express Connect\"])\negressCdn = Col(\"CDN\", [\"CloudFront\", \"Front Door / CDN\", \"Cloud CDN\", \"OCI CDN partners / services\", \"Alibaba Cloud CDN\"])\negressTakeaway = Col(\"Planning note\", [\"Model architecture carefully\", \"Bundle with enterprise agreement where possible\", \"Review destination matrix closely\", \"Best headline option for high outbound volume\", \"Validate China cross-border pathways\"])\ncommitmentTab = TabItem(\"commitments\", \"Discounts & licensing\", [commitmentTable])\ncommitmentTable = Table([commitProvider, commitDiscount, commitLicense, commitMarketplace, commitAdvice])\ncommitProvider = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncommitDiscount = Col(\"Long-term discount\", [\"1- and 3-year commitments; payment options\", \"1- and 3-year reservations / savings\", \"1- and 3-year CUDs; sustained-use discounts\", \"Annual commitments and universal credits\", \"Subscription and savings plans\"])\ncommitLicense = Col(\"Bring-your-own-license\", [\"Broad BYOL support; license mobility nuances\", \"Best alignment with Windows/SQL Server benefits\", \"Supports eligible licenses\", \"Strong Oracle database licensing alignment\", \"Supports common enterprise licenses\"])\ncommitMarketplace = Col(\"Marketplace\", [\"Largest ecosystem\", \"Large enterprise software catalog\", \"Strong data/AI ecosystem\", \"Growing marketplace\", \"Strong regional catalog\"])\ncommitAdvice = Col(\"Procurement strategy\", [\"Use Savings Plans after stable baseline\", \"Price with Azure Hybrid Benefit\", \"Commit steady data/compute workloads\", \"Bundle Oracle estate and high egress\", \"Negotiate for regional volume\"])\ncapabilityTabs = Tabs([securityTab, supportTab, platformTab, serverlessTab])\nsecurityTab = TabItem(\"security\", \"Security & compliance\", [securityIntro, securityTable])\nsecurityIntro = TextContent(\"All five offer core enterprise controls. Certification scope is service-, region-, and configuration-specific; compliance is a shared responsibility.\", \"small\")\nsecurityTable = Table([secProvider, secIdentity, secEncryption, secCompliance, secSovereignty, secStrength])\nsecProvider = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nsecIdentity = Col(\"Identity & access\", [\"IAM, IAM Identity Center\", \"Microsoft Entra ID, Azure RBAC\", \"Cloud IAM, Workforce Identity\", \"OCI IAM\", \"RAM / IDaaS\"])\nsecEncryption = Col(\"Keys / secrets\", [\"KMS, CloudHSM, Secrets Manager\", \"Key Vault, Managed HSM\", \"Cloud KMS, Secret Manager\", \"Vault / HSM\", \"KMS, Secrets Manager\"])\nsecCompliance = Col(\"Common frameworks\", [\"SOC, ISO, PCI, HIPAA-eligible services, FedRAMP\", \"SOC, ISO, PCI, HIPAA, extensive government offerings\", \"SOC, ISO, PCI, HIPAA-eligible services, FedRAMP\", \"SOC, ISO, PCI, HIPAA-eligible services\", \"ISO, SOC, PCI and regional certifications\"])\nsecSovereignty = Col(\"Data residency\", [\"Regional controls and sovereign options\", \"Strong sovereign / industry cloud portfolio\", \"Assured Workloads and sovereign offerings\", \"Dedicated region / sovereign cloud options\", \"Important China regulatory localization capabilities\"])\nsecStrength = Col(\"Security differentiator\", [\"Largest security-service catalog\", \"Deep identity and endpoint integration\", \"Zero-trust design and data security\", \"Isolation and Oracle workload controls\", \"Regional compliance in China/APAC\"])\nsupportTab = TabItem(\"support\", \"Support operations\", [supportTable])\nsupportTable = Table([supportProvider, supportFree, supportPaid, supportResponse, supportPartner, supportTakeaway])\nsupportProvider = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nsupportFree = Col(\"Included\", [\"Docs, forums, basic account support\", \"Basic support and billing assistance\", \"Basic support\", \"Basic support\", \"Basic support\"])\nsupportPaid = Col(\"Paid tiers\", [\"Developer, Business, Enterprise On-Ramp, Enterprise\", \"Developer, Standard, Professional Direct, Unified\", \"Standard, Enhanced, Premium\", \"Advanced support offerings\", \"Developer, Business, Enterprise\"])\nsupportResponse = Col(\"Critical-case target\", [\"Fastest enterprise targets can be ~15 min\", \"Highest tiers can target ~15 min\", \"Premium support can target ~15 min\", \"Enterprise terms vary\", \"Enterprise terms vary\"])\nsupportPartner = Col(\"Partner ecosystem\", [\"Largest global SI/MSP network\", \"Exceptional Microsoft partner channel\", \"Strong specialists and SIs\", \"Strong Oracle-focused partners\", \"Strong APAC / China partners\"])\nsupportTakeaway = Col(\"Operational fit\", [\"Best breadth for diverse stacks\", \"Best for Microsoft operating model\", \"Best for cloud-native/data teams\", \"Best for Oracle workloads\", \"Best for regional execution\"])\nplatformTab = TabItem(\"platform\", \"Engineering platform\", [platformTable])\nplatformTable = Table([platformProvider, platformIaC, platformObservability, platformDevops, platformAi, platformHybrid])\nplatformProvider = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nplatformIaC = Col(\"Infrastructure as code\", [\"CloudFormation, CDK, Terraform\", \"ARM, Bicep, Terraform\", \"Terraform, Config Connector\", \"Resource Manager, Terraform\", \"ROS, Terraform\"])\nplatformObservability = Col(\"Monitoring\", [\"CloudWatch, X-Ray\", \"Azure Monitor, App Insights\", \"Cloud Monitoring, Trace\", \"Monitoring, Logging\", \"CloudMonitor, ARMS\"])\nplatformDevops = Col(\"Delivery\", [\"CodePipeline or partner tools\", \"Azure DevOps, GitHub integration\", \"Cloud Build, Deploy, GitHub integration\", \"DevOps service\", \"DevOps and cloud-native tooling\"])\nplatformAi = Col(\"AI platform\", [\"Bedrock, SageMaker\", \"Azure AI, Azure OpenAI\", \"Vertex AI, Gemini\", \"OCI Generative AI\", \"Model Studio, PAI\"])\nplatformHybrid = Col(\"Hybrid / edge\", [\"Outposts, EKS Anywhere\", \"Azure Arc, Stack HCI\", \"Anthos, Distributed Cloud\", \"Cloud@Customer\", \"Apsara Stack, edge offerings\"])\nserverlessTab = TabItem(\"serverless\", \"Serverless & cold starts\", [serverlessIntro, serverlessTable])\nserverlessIntro = TextContent(\"Cold starts are workload-dependent, not contractual provider specs. Runtime, package size, VPC networking, language, concurrency, provisioned capacity, and idle time dominate outcomes.\", \"small\")\nserverlessTable = Table([coldProvider, coldFunction, coldTypical, coldMitigation, coldContainers, coldVerdict])\ncoldProvider = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncoldFunction = Col(\"Primary function service\", [\"Lambda\", \"Azure Functions\", \"Cloud Functions\", \"OCI Functions\", \"Function Compute\"])\ncoldTypical = Col(\"Practical cold-start range\", [\"Often tens to hundreds of ms; Java/.NET and VPC can be seconds\", \"Often hundreds of ms to seconds; plan matters\", \"Often low hundreds of ms to seconds by generation/runtime\", \"Often hundreds of ms to seconds\", \"Often hundreds of ms to seconds\"])\ncoldMitigation = Col(\"Mitigations\", [\"Provisioned Concurrency, SnapStart for supported Java\", \"Premium plan, always-ready instances\", \"Min instances, Cloud Run instance settings\", \"Provision capacity / optimize image\", \"Provisioned instances / optimize package\"])\ncoldContainers = Col(\"Serverless container option\", [\"Fargate, App Runner, Lambda containers\", \"Container Apps\", \"Cloud Run\", \"Container Instances\", \"Serverless Kubernetes / container options\"])\ncoldVerdict = Col(\"Latency guidance\", [\"Excellent controls; measure per runtime\", \"Use Premium for predictable latency\", \"Cloud Run is especially compelling\", \"Validate on target region\", \"Validate regional runtime support\"])\ncaveats = Card([caveatHeader, caveatText, selectionSteps], \"sunk\")\ncaveatHeader = CardHeader(\"How to make the final selection\", \"Avoid selecting solely on list price.\")\ncaveatText = TextContent(\"Run a 30-day proof of concept using your real traffic, data gravity, recovery objectives, security controls, and support workflow. Score total cost of ownership including people, licenses, network egress, managed-service premiums, and migration risk.\", \"default\")\nselectionSteps = Steps([stepOne, stepTwo, stepThree, stepFour])\nstepOne = StepsItem(\"Anchor on workload and data location\", \"Select the provider with the required regions, residency controls, latency, and private connectivity near your users and data.\")\nstepTwo = StepsItem(\"Model a production architecture\", \"Include multi-zone resilience, backups, observability, NAT, egress, support, and 12- to 36-month commitments—not just a single VM.\")\nstepThree = StepsItem(\"Benchmark critical paths\", \"Measure cold starts, p95/p99 latency, managed database throughput, Kubernetes operations, incident response, and deployment speed with your actual code.\")\nstepFour = StepsItem(\"Choose the operating model\", \"AWS is the safest broad default; Azure wins when Microsoft identity and licensing dominate; Google Cloud leads for modern data/AI; OCI excels for Oracle and outbound-heavy estates; Alibaba Cloud is strategic for China/APAC.\")","promptTokens":4438,"completionTokens":4030,"latencyMs":44897,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/minimal","repeat":4,"output":"root = Stack([revenueCard])\nrevenueCard = Card([revenueHeader, revenueChart])\nrevenueHeader = CardHeader(\"Monthly Revenue\", \"Last 6 months\")\nrevenueChart = BarChart(months, [revenueSeries], \"grouped\", \"Month\", \"Revenue ($)\")\nmonths = [\"December\", \"January\", \"February\", \"March\", \"April\", \"May\"]\nrevenueSeries = Series(\"Revenue\", [84200, 91600, 88700, 103400, 110800, 124500])","promptTokens":4392,"completionTokens":162,"latencyMs":4186,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/realistic","repeat":0,"output":"root = Stack([trafficCard], \"column\", \"l\")\ntrafficCard = Card([trafficHeader, trafficPie, trafficTable, trafficNote], \"column\", \"m\")\ntrafficHeader = CardHeader(\"Website traffic sources — last month\", \"Distribution of all sessions by acquisition channel\")\ntrafficPie = PieChart(trafficLabels, trafficValues, \"donut\")\ntrafficTable = Table([sourceCol, percentCol, sessionsCol])\ntrafficNote = TextContent(\"Organic search was the largest source, accounting for 38% of website traffic.\", \"small\")\ntrafficLabels = [\"Organic search\", \"Paid ads\", \"Social\", \"Direct\", \"Referrals\"]\ntrafficValues = [38, 24, 16, 14, 8]\nsourceCol = Col(\"Traffic source\", trafficLabels)\npercentCol = Col(\"Share of traffic\", [\"38%\", \"24%\", \"16%\", \"14%\", \"8%\"])\nsessionsCol = Col(\"Estimated sessions\", [4560, 2880, 1920, 1680, 960], \"number\")","promptTokens":4413,"completionTokens":264,"latencyMs":4300,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/adversarial","repeat":2,"output":"root = Stack([intro, providerTabs, recommendation], \"column\", \"l\")\nintro = Callout(\"info\", \"Cloud provider comparison\", \"Sorted by overall platform breadth and enterprise adoption. Pricing and SLA figures are representative public-list rates and terms; validate by region, contract, commitment, and workload before buying.\")\nproviderTabs = Tabs([overviewTab, pricingTab, footprintTab, reliabilityTab, complianceTab, supportTab, serverlessTab, networkTab, freeTierTab, fitTab])\noverviewTab = TabItem(\"overview\", \"Executive overview\", [overviewCard, scoreCard, marketTable])\npricingTab = TabItem(\"pricing\", \"Compute & pricing\", [pricingCard, computeTable, storageTable, commitmentNote])\nfootprintTab = TabItem(\"footprint\", \"Regions & network\", [footprintCard, footprintTable, networkTable])\nreliabilityTab = TabItem(\"reliability\", \"Reliability & operations\", [reliabilityCard, slaTable, operationsTable])\ncomplianceTab = TabItem(\"compliance\", \"Security & compliance\", [complianceCard, complianceTable, securityTable])\nsupportTab = TabItem(\"support\", \"Support\", [supportCard, supportTable, supportNote])\nserverlessTab = TabItem(\"serverless\", \"Serverless & containers\", [serverlessCard, serverlessTable, coldStartTable])\nnetworkTab = TabItem(\"network\", \"Egress & connectivity\", [networkCard, egressTable, connectivityTable])\nfreeTierTab = TabItem(\"free\", \"Free tier\", [freeTierCard, freeTierTable, freeTierNote])\nfitTab = TabItem(\"fit\", \"Best fit\", [fitCard, fitTable, decisionCallout])\noverviewCard = Card([CardHeader(\"Ranked comparison\", \"A practical shortlist for global enterprise, cloud-native, data, and cost-sensitive workloads\"), overviewText])\noverviewText = TextContent(\"**1. AWS** — widest service catalog, deepest ecosystem, strongest default for broad requirements. \\n**2. Microsoft Azure** — best fit for Microsoft estates, hybrid infrastructure, and enterprise agreements. \\n**3. Google Cloud** — strongest developer/data/AI posture and excellent global networking. \\n**4. Oracle Cloud Infrastructure** — compelling database economics, high-performance networking, and aggressive enterprise pricing. \\n**5. Alibaba Cloud** — leading option for China and Asia-centric deployment needs.\")\nscoreCard = Card([CardHeader(\"Relative platform breadth\", \"Illustrative 10-point assessment\"), breadthChart])\nbreadthChart = HorizontalBarChart([\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"], [Series(\"Overall breadth\", [10, 9, 8, 7, 7])], \"grouped\", \"Score\", \"Provider\")\nmarketTable = Table([marketProviderCol, marketStrengthCol, marketTradeoffCol, marketBestCol])\nmarketProviderCol = Col(\"Provider\", [\"AWS\", \"Microsoft Azure\", \"Google Cloud\", \"Oracle Cloud Infrastructure\", \"Alibaba Cloud\"])\nmarketStrengthCol = Col(\"Primary strength\", [\"Service breadth and ecosystem\", \"Microsoft and hybrid integration\", \"Data, AI, Kubernetes, network\", \"Oracle workloads and price/performance\", \"China and Asia reach\"])\nmarketTradeoffCol = Col(\"Watch-out\", [\"Complex pricing and governance\", \"Portal/product complexity\", \"Smaller enterprise partner ecosystem\", \"Smaller service ecosystem\", \"Less global enterprise standardization\"])\nmarketBestCol = Col(\"Best default use\", [\"Mixed enterprise portfolio\", \"Windows/AD/M365 estate\", \"Analytics and cloud-native apps\", \"Oracle DB modernization\", \"China market entry\"])\npricingCard = Card([CardHeader(\"Representative on-demand pricing\", \"Approximate US-region public list pricing; Linux general-purpose VM, object storage, and standard managed database pricing vary materially by configuration\")])\ncomputeTable = Table([computeProviderCol, computeVmCol, computeK8sCol, computeSavingsCol, computeBillingCol])\ncomputeProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\ncomputeVmCol = Col(\"Small general VM\", [\"EC2 t3.medium ≈ $0.042/hr\", \"B2s ≈ $0.042/hr\", \"e2-medium ≈ $0.034/hr\", \"E4 Flex 1 OCPU/8GB ≈ $0.043/hr\", \"ecs.t6 ≈ $0.04/hr\"])\ncomputeK8sCol = Col(\"Managed Kubernetes control plane\", [\"EKS ≈ $0.10/hr\", \"AKS standard tier ≈ $0.10/hr\", \"GKE standard ≈ $0.10/hr\", \"OKE basic control plane $0\", \"ACK managed ≈ $0.10/hr\"])\ncomputeSavingsCol = Col(\"Commitment options\", [\"Savings Plans, Reserved Instances, Spot\", \"Reservations, Savings Plan, Spot\", \"Committed Use Discounts, Spot VMs\", \"Universal Credits, committed discounts\", \"Savings Plans, Reserved Instances, Preemptible\"])\ncomputeBillingCol = Col(\"Billing granularity\", [\"Per-second after 60 sec for many services\", \"Per-second for many VMs\", \"Per-second with 1-min minimum\", \"Per-second for many services\", \"Per-second or hourly by service\"])\nstorageTable = Table([storageProviderCol, storageObjectCol, storageArchiveCol, storageBlockCol])\nstorageProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\nstorageObjectCol = Col(\"Hot object storage / GB-month\", [\"S3 Standard ≈ $0.023\", \"Blob Hot LRS ≈ $0.018\", \"Cloud Storage Standard ≈ $0.020\", \"Object Storage ≈ $0.026\", \"OSS Standard ≈ $0.023\"])\nstorageArchiveCol = Col(\"Cold archive / GB-month\", [\"S3 Glacier Deep Archive ≈ $0.00099\", \"Blob Archive ≈ $0.00099\", \"Archive Storage ≈ $0.0012\", \"Archive Storage ≈ $0.0026\", \"OSS Deep Cold Archive ≈ $0.003\"])\nstorageBlockCol = Col(\"Typical balanced block / GB-month\", [\"gp3 ≈ $0.08\", \"Premium SSD varies by tier\", \"Persistent Disk balanced ≈ $0.10\", \"Block Volume lower-cost tiers available\", \"ESS cloud disk varies by type\"])\ncommitmentNote = TextCallout(\"neutral\", \"Pricing discipline matters\", \"All five reward commitments, right-sizing, autoscaling, and spot/preemptible capacity. Compare total workload cost: compute, storage operations, backup, support, egress, NAT, observability, and managed-service premiums.\")\nfootprintCard = Card([CardHeader(\"Global footprint\", \"Region counts change frequently; availability-zone design matters more than raw regional count for resilience\")])\nfootprintTable = Table([footprintProviderCol, footprintRegionsCol, footprintZonesCol, footprintChinaCol, footprintEdgeCol])\nfootprintProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\nfootprintRegionsCol = Col(\"Public cloud regions\", [\"≈ 30+\", \"≈ 60+ regions / broadest geography\", \"≈ 35-40+\", \"≈ 50+\", \"≈ 30+\"])\nfootprintZonesCol = Col(\"Availability domains/zones\", [\"≈ 100+ AZs\", \"Availability Zones in selected regions\", \"≈ 100+ zones\", \"Usually 3 ADs in many regions\", \"Multi-zone regions available\"])\nfootprintChinaCol = Col(\"China availability\", [\"Separate AWS China operators\", \"Azure China via 21Vianet\", \"Limited / partner-oriented availability\", \"Limited China presence\", \"Native market leader\"])\nfootprintEdgeCol = Col(\"Edge/CDN\", [\"CloudFront, Local Zones, Wavelength\", \"Front Door, Edge Zones\", \"Cloud CDN, edge POPs\", \"FastConnect edge and CDN options\", \"Global Accelerator, CDN\"])\nnetworkTable = Table([networkProviderCol, networkBackboneCol, networkHybridCol, networkStrengthCol])\nnetworkProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\nnetworkBackboneCol = Col(\"Global network\", [\"Large private backbone and extensive POP network\", \"Global Microsoft backbone\", \"Highly regarded private backbone\", \"High-performance regional backbone\", \"Strong Asia-Pacific backbone\"])\nnetworkHybridCol = Col(\"Private connectivity\", [\"Direct Connect\", \"ExpressRoute\", \"Cloud Interconnect\", \"FastConnect\", \"Express Connect\"])\nnetworkStrengthCol = Col(\"Notable strength\", [\"Mature multi-account networking\", \"Enterprise WAN and Microsoft integration\", \"Low-latency global app design\", \"High bandwidth and database adjacency\", \"China cross-border and regional reach\"])\nreliabilityCard = Card([CardHeader(\"Service reliability\", \"Published SLAs generally require multi-zone architecture and have exclusions; design for failure rather than relying on credits\")])\nslaTable = Table([slaProviderCol, slaVmCol, slaObjectCol, slaManagedDbCol, slaMultiZoneCol])\nslaProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\nslaVmCol = Col(\"VM SLA, typical\", [\"EC2 up to 99.99% multi-AZ\", \"VMs up to 99.99% availability zones\", \"Compute Engine up to 99.99%\", \"Compute up to 99.99%\", \"ECS commonly up to 99.975-99.99%\"])\nslaObjectCol = Col(\"Object storage durability\", [\"11 nines typical design target\", \"11 nines durability target\", \"11 nines annual durability\", \"11 nines durability target\", \"High durability tiers available\"])\nslaManagedDbCol = Col(\"Managed DB HA\", [\"RDS Multi-AZ / Aurora multi-AZ\", \"Azure SQL zone redundancy\", \"Cloud SQL HA / Spanner\", \"Autonomous DB and RAC options\", \"ApsaraDB HA editions\"])\nslaMultiZoneCol = Col(\"Resilience pattern\", [\"Multi-AZ, multi-region\", \"Availability Zones, paired regions\", \"Regional managed services, multi-region\", \"Fault domains and availability domains\", \"Zone-aware architecture\"])\noperationsTable = Table([opsProviderCol, opsObservabilityCol, opsIaCCol, opsLandingCol])\nopsProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\nopsObservabilityCol = Col(\"Native observability\", [\"CloudWatch, X-Ray\", \"Azure Monitor, Application Insights\", \"Cloud Monitoring, Trace\", \"OCI Monitoring, Logging\", \"CloudMonitor, ARMS\"])\nopsIaCCol = Col(\"Infrastructure as code\", [\"CloudFormation, CDK, Terraform\", \"ARM, Bicep, Terraform\", \"Deployment Manager, Terraform\", \"Resource Manager, Terraform\", \"ROS, Terraform\"])\nopsLandingCol = Col(\"Governance foundation\", [\"Control Tower, Organizations\", \"Landing Zones, Management Groups\", \"Landing Zone, Resource Manager\", \"Compartments, IAM domains\", \"Resource Directory\"])\ncomplianceCard = Card([CardHeader(\"Compliance and security\", \"All providers have extensive audit portfolios. Exact availability depends on service, region, and customer configuration.\")])\ncomplianceTable = Table([complianceProviderCol, complianceFrameworksCol, complianceSovereigntyCol, complianceEncryptionCol])\ncomplianceProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\ncomplianceFrameworksCol = Col(\"Common attestations\", [\"ISO 27001, SOC, PCI DSS, HIPAA eligible\", \"ISO, SOC, PCI, HIPAA eligible, FedRAMP\", \"ISO, SOC, PCI, HIPAA eligible, FedRAMP\", \"ISO, SOC, PCI, HIPAA eligible\", \"ISO, SOC, PCI and regional programs\"])\ncomplianceSovereigntyCol = Col(\"Sovereign/government options\", [\"GovCloud and regional controls\", \"Strong government and sovereign cloud portfolio\", \"Assured Workloads and sovereign controls\", \"Government and dedicated-region offerings\", \"China regulatory ecosystem\"])\ncomplianceEncryptionCol = Col(\"Key management\", [\"KMS, CloudHSM, external key options\", \"Key Vault, Managed HSM\", \"Cloud KMS, EKM\", \"Vault, HSM\", \"KMS, dedicated HSM options\"])\nsecurityTable = Table([securityProviderCol, securityIdentityCol, securityPostureCol, securityStrengthCol])\nsecurityProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\nsecurityIdentityCol = Col(\"Identity\", [\"IAM, IAM Identity Center\", \"Microsoft Entra ID, RBAC\", \"Cloud IAM, Workforce Identity\", \"OCI IAM\", \"RAM and CloudSSO\"])\nsecurityPostureCol = Col(\"Security posture\", [\"Security Hub, GuardDuty, Inspector\", \"Defender for Cloud, Sentinel\", \"Security Command Center, Chronicle\", \"Cloud Guard, Security Zones\", \"Security Center\"])\nsecurityStrengthCol = Col(\"Best security fit\", [\"Granular cloud-native controls\", \"Microsoft identity/SOC integration\", \"Zero-trust and data security tooling\", \"Oracle database security\", \"China compliance operations\"])\nsupportCard = Card([CardHeader(\"Support tiers\", \"Named support prices vary with spend. Premium plans frequently combine minimum monthly fees with a percentage of usage.\")])\nsupportTable = Table([supportProviderCol, supportEntryCol, supportPremiumCol, supportEnterpriseCol, supportChannelCol])\nsupportProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\nsupportEntryCol = Col(\"Entry support\", [\"Basic included; Developer from ≈ $29/mo\", \"Basic included; Developer from ≈ $29/mo\", \"Basic included; Standard from ≈ $29/mo\", \"Basic included; paid support options\", \"Basic included; paid plans available\"])\nsupportPremiumCol = Col(\"Mid-tier\", [\"Business from ≈ $100/mo\", \"Standard/Professional Direct\", \"Enhanced support tiers\", \"Advanced support\", \"Business support\"])\nsupportEnterpriseCol = Col(\"Top tier\", [\"Enterprise On-Ramp / Enterprise\", \"Unified support\", \"Premium Support\", \"Oracle Premier Support\", \"Enterprise support\"])\nsupportChannelCol = Col(\"Typical top-tier features\", [\"24x7, TAM options, proactive guidance\", \"24x7, account management, Microsoft escalation\", \"24x7, TAM, operational reviews\", \"24x7, technical account support\", \"24x7, enterprise response\"])\nsupportNote = TextCallout(\"warning\", \"Procurement note\", \"Support quality is strongly affected by your account team, partner, support plan, service criticality, and architecture maturity—not just the plan name.\")\nserverlessCard = Card([CardHeader(\"Serverless, containers, and cold starts\", \"Cold-start ranges are workload-dependent estimates for warm runtime initialization, excluding downstream dependencies and provisioning delays.\")])\nserverlessTable = Table([serverlessProviderCol, functionsCol, containersCol, kubernetesCol, scaleToZeroCol])\nserverlessProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\nfunctionsCol = Col(\"Functions\", [\"Lambda\", \"Azure Functions\", \"Cloud Functions / Cloud Run functions\", \"OCI Functions\", \"Function Compute\"])\ncontainersCol = Col(\"Serverless containers\", [\"Fargate, App Runner\", \"Container Apps\", \"Cloud Run\", \"Container Instances\", \"Elastic Container Instance\"])\nkubernetesCol = Col(\"Managed Kubernetes\", [\"EKS\", \"AKS\", \"GKE\", \"OKE\", \"ACK\"])\nscaleToZeroCol = Col(\"Scale-to-zero option\", [\"Lambda/App Runner patterns\", \"Functions and Container Apps\", \"Cloud Run native\", \"Functions/containers\", \"Function Compute/ECI\"])\ncoldStartTable = Table([coldProviderCol, coldTypicalCol, coldMitigationCol, coldNotesCol])\ncoldProviderCol = Col(\"Provider\", [\"AWS Lambda\", \"Azure Functions\", \"Google Cloud Run\", \"OCI Functions\", \"Alibaba Function Compute\"])\ncoldTypicalCol = Col(\"Typical cold start\", [\"~100 ms to several seconds\", \"~0.5 to several seconds\", \"~100 ms to a few seconds\", \"~0.5 to several seconds\", \"~100 ms to several seconds\"])\ncoldMitigationCol = Col(\"Mitigation\", [\"Provisioned Concurrency, SnapStart for Java\", \"Premium plan, always-ready instances\", \"Minimum instances, smaller images\", \"Provisioned concurrency patterns\", \"Provisioned instances, lean packages\"])\ncoldNotesCol = Col(\"Main driver\", [\"Runtime, package size, VPC setup\", \"Plan type, runtime, dependencies\", \"Container image and language\", \"Runtime and dependency initialization\", \"Runtime, image/package size\"])\nnetworkCard = Card([CardHeader(\"Data egress and connectivity economics\", \"Internet egress is commonly free for small monthly allowances, then charged by tier and destination. Rates differ sharply by region and route.\")])\negressTable = Table([egressProviderCol, egressInternetCol, egressSameRegionCol, egressInboundCol, egressCostNoteCol])\negressProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\negressInternetCol = Col(\"Public internet egress\", [\"Often starts ≈ $0.09/GB\", \"Often starts ≈ $0.087/GB\", \"Often starts ≈ $0.12/GB\", \"First 10 TB/mo often free; then ≈ $0.0085/GB\", \"Often ≈ $0.08-0.12/GB\"])\negressSameRegionCol = Col(\"Same-region traffic\", [\"Depends on AZ/service path\", \"Depends on zone/service path\", \"Often free within zone; service-specific\", \"Often favorable within region\", \"Service and zone dependent\"])\negressInboundCol = Col(\"Internet ingress\", [\"Generally free\", \"Generally free\", \"Generally free\", \"Generally free\", \"Generally free\"])\negressCostNoteCol = Col(\"Cost lever\", [\"CloudFront, private links, architecture\", \"Front Door, ExpressRoute\", \"CDN, interconnect, caching\", \"Excellent outbound allowance\", \"CDN and regional routing\"])\nconnectivityTable = Table([connectProviderCol, connectPrivateCol, connectVpnCol, connectTransitCol])\nconnectProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\nconnectPrivateCol = Col(\"Dedicated private link\", [\"Direct Connect\", \"ExpressRoute\", \"Cloud Interconnect\", \"FastConnect\", \"Express Connect\"])\nconnectVpnCol = Col(\"Managed VPN\", [\"Site-to-Site VPN\", \"VPN Gateway\", \"Cloud VPN\", \"IPSec VPN\", \"VPN Gateway\"])\nconnectTransitCol = Col(\"Hub networking\", [\"Transit Gateway\", \"Virtual WAN\", \"Network Connectivity Center\", \"DRG\", \"Cloud Enterprise Network\"])\nfreeTierCard = Card([CardHeader(\"Free tiers and trials\", \"Free offers change regularly and may require a payment method. Treat these as evaluation tools, not production capacity.\")])\nfreeTierTable = Table([freeProviderCol, freeTrialCol, freeAlwaysCol, freeComputeCol, freeCautionCol])\nfreeProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\"])\nfreeTrialCol = Col(\"New-account trial\", [\"12-month offers plus credits/services\", \"≈ $200 credit for 30 days\", \"≈ $300 credit for 90 days\", \"≈ $300 credit for 30 days\", \"Trial credits vary by country\"])\nfreeAlwaysCol = Col(\"Always-free services\", [\"Selected Lambda, S3, DynamoDB and more\", \"Selected Functions, App Service and more\", \"Cloud Run, BigQuery, storage allowances\", \"Notable Always Free compute/storage\", \"Selected free quotas\"])\nfreeComputeCol = Col(\"Notable free compute\", [\"Limited EC2 eligibility offers\", \"Limited VM trial offers\", \"Small Compute Engine e2-micro in eligible US region\", \"Up to 2 AMD micro VMs or Arm allocation\", \"Trial/free quota varies\"])\nfreeCautionCol = Col(\"Expiration risk\", [\"12-month free offers expire\", \"Credit and 30-day trial expire\", \"Credits expire; usage beyond quota bills\", \"Capacity availability can be constrained\", \"Country and service eligibility varies\"])\nfreeTierNote = TextCallout(\"info\", \"FinOps hygiene\", \"Set budgets, billing alerts, quota limits, and automatic cleanup on day one. Free tiers can still incur egress, public IP, logging, managed database, or overage charges.\")\nfitCard = Card([CardHeader(\"Decision matrix\", \"Choose based on your dominant constraint, then validate with a 30-day representative proof of concept\")])\nfitTable = Table([fitNeedCol, fitPickCol, fitWhyCol, fitAlternativeCol])\nfitNeedCol = Col(\"If your priority is...\", [\"Broadest cloud platform\", \"Microsoft-heavy enterprise\", \"Analytics, ML, Kubernetes\", \"Oracle database estate or egress economics\", \"China and Asia operations\", \"Lowest operational complexity\"])\nfitPickCol = Col(\"First choice\", [\"AWS\", \"Azure\", \"Google Cloud\", \"Oracle Cloud\", \"Alibaba Cloud\", \"Google Cloud or Azure\"])\nfitWhyCol = Col(\"Why\", [\"Largest catalog, marketplace, partner ecosystem\", \"Entra, Windows Server, SQL Server, M365 and hybrid alignment\", \"BigQuery, Vertex AI, GKE and network quality\", \"Autonomous Database, Exadata options, generous outbound policy\", \"Local ecosystem, regulatory fit, regional presence\", \"Strong managed services and developer experience\"])\nfitAlternativeCol = Col(\"Also evaluate\", [\"Azure or Google Cloud\", \"AWS\", \"AWS\", \"AWS or Azure\", \"AWS China / Azure China where applicable\", \"AWS with a strong platform team\"])\nrecommendation = Callout(\"success\", \"My pick: AWS for the broad default\", \"I would pick AWS when choosing one provider without a stated constraint: it has the most complete service portfolio, mature multi-account governance, global resilience options, ecosystem depth, and hiring/partner availability. I would deliberately choose Azure for a Microsoft-centered organization, Google Cloud for a data/AI-first platform, Oracle Cloud for Oracle-heavy economics and high egress workloads, and Alibaba Cloud for China-first operations.\")","promptTokens":4438,"completionTokens":4822,"latencyMs":49176,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/realistic","repeat":2,"output":"root = Stack([header, trafficCard, breakdownCard], \"column\", \"l\")\nheader = Card([title, subtitle], \"clear\", \"column\", \"xs\")\ntitle = TextContent(\"Website Traffic Sources\", \"large-heavy\")\nsubtitle = TextContent(\"Last month • Share of total sessions\", \"default\")\ntrafficCard = Card([trafficHeader, pieChart, totalNote], \"card\", \"column\", \"m\")\ntrafficHeader = CardHeader(\"Traffic acquisition mix\", \"Distribution of all website sessions\")\npieChart = PieChart([\"Organic Search\", \"Paid Ads\", \"Social\", \"Direct\", \"Referrals\"], [42, 24, 15, 12, 7], \"donut\")\ntotalNote = TextCallout(\"info\", \"100% of traffic accounted for\", \"Organic search led acquisition with 42%, followed by paid ads at 24%.\")\nbreakdownCard = Card([breakdownHeader, trafficTable], \"card\", \"column\", \"m\")\nbreakdownHeader = CardHeader(\"Source breakdown\", \"Percentage of total website traffic\")\ntrafficTable = Table([sourceCol, shareCol])\nsourceCol = Col(\"Traffic source\", [\"Organic Search\", \"Paid Ads\", \"Social\", \"Direct\", \"Referrals\"])\nshareCol = Col(\"Share of traffic\", [\"42%\", \"24%\", \"15%\", \"12%\", \"7%\"])","promptTokens":4413,"completionTokens":331,"latencyMs":6124,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/adversarial","repeat":1,"output":"root = Stack([header, scope, recommendation, comparisonTabs, sources], \"column\", \"l\")\nheader = Card([title, subtitle], \"column\", \"s\")\ntitle = TextContent(\"Top cloud providers: buyer’s comparison\", \"large-heavy\")\nsubtitle = TextContent(\"AWS, Microsoft Azure, Google Cloud, Oracle Cloud Infrastructure, and IBM Cloud — sorted by overall enterprise breadth and ecosystem.\")\nscope = Callout(\"info\", \"How to read this\", \"This is a planning snapshot, not a quote. Exact prices, regions, availability commitments, free allowances, and egress rules vary by service, contract, billing currency, destination, and date. “Cold starts” are typical serverless ranges and depend heavily on runtime, package size, VPC networking, and provisioned capacity.\")\nrecommendation = Card([recommendationHeader, recommendationText, recommendationTags], \"column\", \"s\")\nrecommendationHeader = CardHeader(\"My pick: AWS\", \"Best default when you need the widest service catalog, deepest global footprint, mature partner ecosystem, and the strongest hiring pool.\")\nrecommendationText = TextContent(\"Choose **AWS** for a broad, multi-year platform strategy unless your organization is already deeply committed to Microsoft identity, licensing, and data tooling—in which case **Azure** is usually the better economic and operational choice. Pick **Google Cloud** for data/AI-centric teams that value BigQuery, Kubernetes leadership, and simpler analytics workflows. OCI is compelling for Oracle-heavy estates and aggressive price-performance deals; IBM Cloud is most relevant for hybrid regulated environments and IBM platform alignment.\")\nrecommendationTags = TagBlock([\"Best overall: AWS\", \"Best Microsoft estate: Azure\", \"Best data & AI workflow: Google Cloud\", \"Best Oracle economics: OCI\", \"Best hybrid IBM estate: IBM Cloud\"])\ncomparisonTabs = Tabs([overviewTab, economicsTab, reliabilityTab, securityTab, platformTab, supportTab, decisionTab])\noverviewTab = TabItem(\"overview\", \"Executive comparison\", [overviewCard, overviewTable])\noverviewCard = Card([overviewHeader, overviewNote], \"column\", \"s\")\noverviewHeader = CardHeader(\"At-a-glance\", \"A practical starting point for platform selection\")\noverviewNote = TextContent(\"Rank reflects global breadth, service maturity, ecosystem depth, and enterprise adoption—not universal price or technical superiority.\")\noverviewTable = Table([rankCol, providerCol, bestForCol, footprintCol, strengthsCol, tradeoffsCol])\nrankCol = Col(\"Rank\", [\"1\", \"2\", \"3\", \"4\", \"5\"], \"number\")\nproviderCol = Col(\"Provider\", [\"Amazon Web Services\", \"Microsoft Azure\", \"Google Cloud\", \"Oracle Cloud Infrastructure\", \"IBM Cloud\"])\nbestForCol = Col(\"Best fit\", [\"Broad enterprise/cloud-native default\", \"Microsoft-centric enterprises\", \"Analytics, ML, Kubernetes teams\", \"Oracle workloads and price-sensitive compute\", \"Hybrid regulated IBM environments\"])\nfootprintCol = Col(\"Global footprint\", [\"Largest: ~30+ regions, ~100+ AZs\", \"~60+ regions; broad enterprise reach\", \"~40+ regions; global backbone\", \"~50+ commercial/public regions\", \"Smaller public footprint; strong hybrid options\"])\nstrengthsCol = Col(\"Standout strengths\", [\"Service breadth, ecosystem, operational maturity\", \"AD/M365/Windows integration, hybrid management\", \"BigQuery, GKE, AI/data products\", \"Oracle DB affinity, networking and pricing\", \"Red Hat OpenShift, mainframe/hybrid integration\"])\ntradeoffsCol = Col(\"Primary watch-outs\", [\"Complex pricing and service sprawl\", [\"Portal/service complexity; capacity can vary\"], \"Smaller catalog/partner pool than AWS/Azure\", \"Narrower ecosystem outside Oracle\", \"Smaller cloud-native ecosystem and footprint\"])\neconomicsTab = TabItem(\"economics\", \"Pricing & free tier\", [economicsCard, economicsTable, costNote])\neconomicsCard = Card([economicsHeader, economicsText], \"column\", \"s\")\neconomicsHeader = CardHeader(\"Economic model\", \"Use committed-use discounts and workload-level benchmarking—not list prices alone\")\neconomicsText = TextContent(\"All five use pay-as-you-go billing plus committed-use or subscription discounts. AWS and Azure offer the deepest discount constructs and marketplace ecosystems; Google’s sustained/committed use is often straightforward for steady workloads; OCI frequently competes aggressively on compute, database, and network economics.\")\neconomicsTable = Table([econProviderCol, computeCol, commitmentCol, freeTierCol, egressCol, storageCol, pricingClarityCol])\neconProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"IBM Cloud\"])\ncomputeCol = Col(\"On-demand compute\", [\"Very broad instance catalog; spot savings can be substantial\", \"VM breadth plus Hybrid Benefit for eligible licenses\", \"Strong custom machine types; spot VMs\", \"Often aggressive list and negotiated pricing\", \"Competitive selected VM/bare-metal offerings\"])\ncommitmentCol = Col(\"Commitment discounts\", [\"Savings Plans / Reserved Instances, typically 1–3 years\", \"Reserved Instances / Savings Plan, 1–3 years\", \"Committed Use Discounts, commonly 1–3 years\", \"Universal Credits / commitments; contract-led\", \"Term commitments and enterprise agreements\"])\nfreeTierCol = Col(\"Free tier\", [\"12-month introductory offers plus always-free quotas on selected services\", \"Account credit/trial plus selected always-free services\", [\"$300 trial credit for new accounts plus always-free products\"], \"Notable always-free compute/storage allocations plus trial credit\", \"Lite plans and trial credits; service-specific\"])\negressCol = Col(\"Internet egress\", [\"Usually charged by GB after allowances; region/service dependent\", \"Usually charged by GB; zone/region paths can also matter\", \"Often competitive; charged by destination/service after allowances\", \"Frequently generous outbound allowance; verify service/region\", \"Charged by GB and service path; verify contract\"])\nstorageCol = Col(\"Object storage classes\", [\"S3 Standard through archive tiers; retrieval/API costs matter\", \"Hot/Cool/Cold/Archive tiers; transaction/retrieval costs\", \"Standard/Nearline/Coldline/Archive; lifecycle-friendly\", \"Standard/Infrequent/Archive choices\", \"Standard and archive-oriented offerings\"])\npricingClarityCol = Col(\"Budget risk\", [\"High: many meters, transfer and managed-service details\", \"High: licensing, regions, and service meters\", \"Medium: simpler data analytics economics, still nuanced\", \"Medium: simpler network story, contract dependence\", \"Medium-high: catalog and contract variation\"])\ncostNote = TextCallout(\"warning\", \"Cost-control checklist\", \"Model compute, storage operations, backup, observability ingestion, managed NAT, cross-zone/cross-region traffic, support, software licenses, and disaster-recovery idle capacity. Obtain a written egress schedule for your exact architecture.\")\nreliabilityTab = TabItem(\"reliability\", \"Regions, uptime & DR\", [reliabilityCard, reliabilityTable, uptimeNote])\nreliabilityCard = Card([reliabilityHeader, reliabilityText], \"column\", \"s\")\nreliabilityHeader = CardHeader(\"Reliability posture\", \"Design matters more than a provider-wide uptime number\")\nreliabilityText = TextContent(\"Cloud providers publish service-specific SLAs, commonly around **99.9% to 99.99%+** depending on topology and product. Higher commitments generally require multi-zone, multi-region, or redundant-instance architecture. Credits are usually the remedy—not compensation for business loss.\")\nreliabilityTable = Table([reliabilityProviderCol, regionsCol, zoneModelCol, slaCol, drCol, networkCol])\nreliabilityProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"IBM Cloud\"])\nregionsCol = Col(\"Regions / locations\", [\"~30+ regions globally\", \"~60+ announced/available regions\", \"~40+ regions\", \"~50+ public/commercial regions\", \"Smaller multi-region footprint\"])\nzoneModelCol = Col(\"Fault domains\", [\"Regions with multiple Availability Zones\", \"Regions with Availability Zones / paired-region concepts\", \"Regions with multiple zones\", \"Regions with Availability Domains and Fault Domains\", \"Multi-zone regions plus satellite/hybrid options\"])\nslaCol = Col(\"Typical service SLA range\", [\"~99.9–99.99%+ by service and architecture\", \"~99.9–99.99%+ by service and architecture\", \"~99.9–99.99%+ by service and architecture\", \"~99.9–99.99%+ by service and architecture\", \"~99.9–99.99%+ by service and architecture\"])\ndrCol = Col(\"DR tooling\", [\"Elastic DR, Backup, cross-region replication, multi-AZ patterns\", \"Azure Site Recovery, Backup, paired regions\", \"Backup and DR, replication, global load balancing\", \"Full Stack DR, replication, database DR\", \"Backup, replication, OpenShift/hybrid patterns\"])\nnetworkCol = Col(\"Private connectivity\", [\"Direct Connect, Transit Gateway, PrivateLink\", \"ExpressRoute, Virtual WAN, Private Link\", \"Cloud Interconnect, Cloud VPN, Private Service Connect\", \"FastConnect, DRG, private endpoints\", \"Direct Link, Transit Gateway, Satellite\"])\nuptimeNote = TextCallout(\"info\", \"Architecture recommendation\", \"For critical services, use at least multi-zone deployment, automated backups, tested restore procedures, infrastructure-as-code, health-based failover, and explicit RTO/RPO targets. Add multi-region only when the business impact justifies its cost and operational complexity.\")\nsecurityTab = TabItem(\"security\", \"Security & compliance\", [securityCard, securityTable, securityNote])\nsecurityCard = Card([securityHeader, securityText], \"column\", \"s\")\nsecurityHeader = CardHeader(\"Security, governance & sovereignty\", \"All are capable for highly regulated use cases with correct configuration\")\nsecurityText = TextContent(\"The shared-responsibility model applies everywhere: the provider secures underlying infrastructure; you secure identity, data, configuration, application code, and many workload controls. Validate a required certification at the exact service, region, and contractual scope.\")\nsecurityTable = Table([securityProviderCol, identityCol, encryptionCol, complianceCol, sovereigntyCol, securityOpsCol])\nsecurityProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"IBM Cloud\"])\nidentityCol = Col(\"Identity\", [\"IAM, IAM Identity Center, external federation\", \"Microsoft Entra ID, RBAC, PIM\", \"Cloud IAM, Workforce Identity Federation\", \"OCI IAM, federation\", \"IAM Access Groups, App ID integration\"])\nencryptionCol = Col(\"Key management\", [\"KMS, CloudHSM, external key patterns\", \"Key Vault, Managed HSM, customer-managed keys\", \"Cloud KMS, Cloud HSM, external key manager options\", \"Vault and HSM services\", \"Key Protect, Hyper Protect Crypto Services\"])\ncomplianceCol = Col(\"Common coverage\", [\"Broad: ISO, SOC, PCI, HIPAA-eligible services, FedRAMP programs, more\", \"Broad: ISO, SOC, PCI, HIPAA-eligible services, government programs\", \"Broad: ISO, SOC, PCI, HIPAA-eligible services, government programs\", \"Broad enterprise/regional certifications; verify scope\", \"Strong regulated-industry and financial-services focus; verify scope\"])\nsovereigntyCol = Col(\"Sovereign options\", [\"Dedicated/sovereign and government offerings in selected markets\", \"Strong sovereign cloud and government portfolio\", \"Assured Workloads and sovereign controls in selected markets\", \"Dedicated region and sovereign cloud offerings\", \"Sovereign/hybrid and industry-specific offerings\"])\nsecurityOpsCol = Col(\"Native security operations\", [\"CloudTrail, Config, GuardDuty, Security Hub\", \"Defender for Cloud, Sentinel, Policy\", \"Security Command Center, Chronicle ecosystem\", \"Cloud Guard, Security Zones\", \"Security and Compliance Center, QRadar ecosystem\"])\nsecurityNote = TextCallout(\"warning\", \"Compliance caveat\", \"A provider certification does not automatically make your workload compliant. Require evidence for data residency, encryption-key control, audit logging, subprocessors, retention, incident notification, BAA/DPA terms, and service-level scope.\")\nplatformTab = TabItem(\"platform\", \"Compute, data & developer platform\", [platformCard, platformTable, coldStartNote])\nplatformCard = Card([platformHeader, platformText], \"column\", \"s\")\nplatformHeader = CardHeader(\"Platform capabilities\", \"All support VMs, containers, Kubernetes, serverless, managed databases, AI, and hybrid patterns\")\nplatformText = TextContent(\"AWS leads catalog breadth; Azure leads integration with Microsoft enterprise tooling; Google Cloud is particularly strong in analytics and Kubernetes; OCI is optimized around Oracle platforms; IBM is strongest where OpenShift and hybrid integration are strategic.\")\nplatformTable = Table([platformProviderCol, k8sCol, serverlessCol, coldStartCol, dataCol, aiCol, hybridCol])\nplatformProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"IBM Cloud\"])\nk8sCol = Col(\"Managed Kubernetes\", [\"EKS\", \"AKS\", \"GKE\", \"OKE\", \"Red Hat OpenShift / Kubernetes Service\"])\nserverlessCol = Col(\"Functions / containers\", [\"Lambda, Fargate, App Runner\", \"Azure Functions, Container Apps\", \"Cloud Functions, Cloud Run\", \"Functions, Container Instances\", \"Code Engine, Cloud Functions\"])\ncoldStartCol = Col(\"Typical function cold start\", [\"Often ~100 ms to several seconds; Java/.NET/VPC can be longer\", \"Often sub-second to several seconds; plan/runtime dependent\", \"Often ~100 ms to a few seconds; Cloud Run behavior differs\", \"Often hundreds of ms to several seconds\", \"Often hundreds of ms to several seconds\"])\ndataCol = Col(\"Data strengths\", [\"Broad databases, Redshift, EMR, Athena\", \"SQL ecosystem, Fabric/Synapse, Cosmos DB\", \"BigQuery, Bigtable, Dataflow\", \"Autonomous Database, Exadata services\", \"Db2, watsonx.data, data fabric\"])\naiCol = Col(\"AI platform\", [\"Bedrock, SageMaker, broad partner models\", \"Azure AI, OpenAI integration, Fabric ecosystem\", \"Vertex AI, Gemini ecosystem, TPU\", \"OCI Generative AI and database integrations\", \"watsonx and governed enterprise AI\"])\nhybridCol = Col(\"Hybrid / edge\", [\"Outposts, Local Zones, Wavelength\", \"Azure Arc, Stack HCI, edge portfolio\", \"Anthos, Distributed Cloud\", \"Cloud@Customer, Dedicated Region\", \"OpenShift, Satellite, mainframe integration\"])\ncoldStartNote = TextCallout(\"info\", \"Cold-start reality\", \"Do not choose a cloud on generic cold-start claims. Benchmark your exact runtime, artifact size, dependencies, network attachment, concurrency, and warm-up strategy. For latency-sensitive endpoints, use provisioned/minimum instances or always-on containers where available.\")\nsupportTab = TabItem(\"support\", \"Support & operations\", [supportCard, supportTable])\nsupportCard = Card([supportHeader, supportText], \"column\", \"s\")\nsupportHeader = CardHeader(\"Support tiers and operating model\", \"Premium support is usually priced as a minimum monthly charge, a percentage of spend, or a negotiated enterprise agreement\")\nsupportText = TextContent(\"Named tiers and pricing change regularly. Every provider offers community/basic help, paid technical support, and enterprise-grade support with faster response targets, proactive guidance, and account management. Confirm severity-1 response commitments in your contract.\")\nsupportTable = Table([supportProviderCol, basicCol, paidCol, enterpriseCol, toolingCol, marketplaceCol])\nsupportProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"IBM Cloud\"])\nbasicCol = Col(\"Entry support\", [\"Basic account support\", \"Basic support\", \"Basic role-based support\", \"Basic included support\", \"Lite / basic support\"])\npaidCol = Col(\"Technical tiers\", [\"Developer, Business, Enterprise On-Ramp\", \"Developer, Standard, Professional Direct\", \"Standard, Enhanced\", \"Advanced support options\", \"Advanced / Premium support options\"])\nenterpriseCol = Col(\"Enterprise option\", [\"Enterprise Support with TAM-style engagement options\", \"Unified / enterprise support arrangements\", \"Premium Support and account services\", \"Premier-style enterprise engagement\", \"Premium support and customer success\"])\ntoolingCol = Col(\"FinOps & operations\", [\"Cost Explorer, Budgets, CloudWatch, Organizations\", \"Cost Management, Advisor, Monitor, Policy\", \"Billing reports, Recommender, Cloud Operations\", \"Cost Analysis, Advisor, Observability\", \"Usage monitoring, Activity Tracker, Turbonomic ecosystem\"])\nmarketplaceCol = Col(\"Ecosystem\", [\"Largest marketplace and partner network\", \"Very large Microsoft/SI ecosystem\", \"Strong data/AI and cloud-native ecosystem\", \"Oracle-led marketplace and partners\", \"IBM/Red Hat-led ecosystem\"])\ndecisionTab = TabItem(\"decision\", \"Selection guide\", [decisionCard, decisionSteps, decisionMatrix])\ndecisionCard = Card([decisionHeader, decisionText], \"column\", \"s\")\ndecisionHeader = CardHeader(\"Fast decision guide\", \"Start from organizational constraints, then validate with a proof of concept\")\ndecisionText = TextContent(\"Avoid a single-provider mandate based only on a benchmark or headline discount. Score providers against your workload portfolio, required geographies, commercial terms, operating skills, and exit strategy.\")\ndecisionSteps = Steps([stepOne, stepTwo, stepThree, stepFour])\nstepOne = StepsItem(\"Set non-negotiables\", \"List required regions, data residency, certifications, identity integration, RTO/RPO, latency, and contractual requirements.\")\nstepTwo = StepsItem(\"Build a representative cost model\", \"Include 12–36 month growth, commitments, support, egress, logs, backups, licenses, and DR—not only compute.\")\nstepThree = StepsItem(\"Run a 2–4 week proof of concept\", \"Measure deployment speed, operational burden, p95 latency, cold starts, throughput, observability, and support responsiveness using your real workload.\")\nstepFour = StepsItem(\"Negotiate and design for portability\", \"Seek price protection, egress terms, support response commitments, credits, and a documented data-export/exit plan.\")\ndecisionMatrix = Table([scenarioCol, choiceCol, rationaleCol])\nscenarioCol = Col(\"If your priority is…\", [\"Maximum service breadth and talent availability\", \"Microsoft licensing, Entra ID, Windows/SQL Server integration\", \"BigQuery, modern analytics, ML, Kubernetes UX\", \"Oracle Database, Exadata, Oracle app estate\", \"Red Hat OpenShift, mainframe-adjacent hybrid integration\"])\nchoiceCol = Col(\"Start with\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"IBM Cloud\"])\nrationaleCol = Col(\"Why\", [\"Largest overall ecosystem and mature services\", \"Lowest integration friction in Microsoft-centric estates\", \"Strong managed data and cloud-native developer experience\", \"Commercial and technical alignment with Oracle workloads\", \"Strong fit for IBM/Red Hat hybrid operating models\"])\nsources = Buttons([pricingButton, slaButton, architectureButton], \"row\")\npricingButton = Button(\"Open provider pricing pages\", Action([@OpenUrl(\"https://aws.amazon.com/pricing/\")]), \"secondary\")\nslaButton = Button(\"Review AWS SLA examples\", Action([@OpenUrl(\"https://aws.amazon.com/legal/service-level-agreements/\")]), \"tertiary\")\narchitectureButton = Button(\"Explore cloud architecture guidance\", Action([@ToAssistant(\"Help me build a weighted cloud-provider scorecard for my workloads, regions, compliance requirements, and expected monthly spend.\")]), \"primary\")","promptTokens":4438,"completionTokens":4216,"latencyMs":50795,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/realistic","repeat":1,"output":"root = Stack([dashboard], \"column\", \"l\")\ndashboard = Card([header, summary, trafficChart, breakdown], \"column\", \"l\")\nheader = CardHeader(\"Website traffic sources\", \"Last month\")\nsummary = TextContent(\"A total of 48,260 visits were recorded last month. Organic search remained the largest acquisition channel.\")\ntrafficChart = PieChart([\"Organic search\", \"Paid ads\", \"Social\", \"Direct\", \"Referrals\"], [42, 24, 16, 11, 7], \"donut\")\nbreakdown = Table([sourceCol, percentageCol, visitsCol])\nsourceCol = Col(\"Traffic source\", [\"Organic search\", \"Paid ads\", \"Social\", \"Direct\", \"Referrals\"])\npercentageCol = Col(\"Share of traffic\", [\"42%\", \"24%\", \"16%\", \"11%\", \"7%\"])\nvisitsCol = Col(\"Visits\", [20269, 11582, 7722, 5309, 3378], \"number\")","promptTokens":4413,"completionTokens":243,"latencyMs":10453,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/adversarial","repeat":1,"output":"root = Stack([header, chartCard, dataCard], \"column\", \"l\")\nheader = Card([title, subtitle], \"clear\", \"column\", \"xs\")\ntitle = TextContent(\"Monthly trend\", \"large-heavy\")\nsubtitle = TextContent(\"January–July performance\")\nchartCard = Card([chartHeader, trendChart], \"card\", \"column\", \"m\")\nchartHeader = CardHeader(\"Monthly values\", \"Values rose from 4,200 in January to a peak of 7,100 in June.\")\ntrendChart = LineChart(months, [valueSeries], \"natural\", \"Month\", \"Value\")\nvalueSeries = Series(\"Value\", values)\ndataCard = Card([dataHeader, rawTable], \"card\", \"column\", \"m\")\ndataHeader = CardHeader(\"Raw numbers\", \"Monthly values for verification\")\nrawTable = Table([monthColumn, valueColumn])\nmonthColumn = Col(\"Month\", months)\nvalueColumn = Col(\"Value\", values, \"number\")\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nvalues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]","promptTokens":4433,"completionTokens":261,"latencyMs":3298,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/adversarial","repeat":0,"output":"root = Stack([header, chartCard, tableCard], \"column\", \"l\")\nheader = Card([title, summary], \"clear\", \"column\", \"xs\")\ntitle = TextContent(\"Monthly performance\", \"large-heavy\")\nsummary = TextContent(\"January through July values, with a peak of 7,100 in June.\", \"default\")\nchartCard = Card([chartHeader, trendChart], \"card\", \"column\", \"m\")\nchartHeader = CardHeader(\"Monthly trend\", \"Values by month\")\ntrendChart = LineChart(months, [performanceSeries], \"natural\", \"Month\", \"Value\")\ntableCard = Card([tableHeader, rawTable], \"card\", \"column\", \"m\")\ntableHeader = CardHeader(\"Raw numbers\", \"Source values for verification\")\nrawTable = Table([monthColumn, valueColumn])\nmonthColumn = Col(\"Month\", months)\nvalueColumn = Col(\"Value\", values, \"number\")\nperformanceSeries = Series(\"Value\", values)\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nvalues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]","promptTokens":4433,"completionTokens":259,"latencyMs":3917,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/realistic","repeat":3,"output":"root = Stack([trafficCard])\ntrafficCard = Card([trafficHeader, trafficChart, trafficSummary])\ntrafficHeader = CardHeader(\"Website traffic sources\", \"Last month • 48,260 sessions\")\ntrafficChart = PieChart([\"Organic search\", \"Paid ads\", \"Social\", \"Direct\", \"Referrals\"], [42, 23, 15, 12, 8], \"donut\")\ntrafficSummary = Table([sourceCol, percentCol, sessionsCol])\nsourceCol = Col(\"Source\", [\"Organic search\", \"Paid ads\", \"Social\", \"Direct\", \"Referrals\"])\npercentCol = Col(\"Share\", [\"42%\", \"23%\", \"15%\", \"12%\", \"8%\"])\nsessionsCol = Col(\"Sessions\", [20269, 11100, 7239, 5791, 3861], \"number\")","promptTokens":4413,"completionTokens":233,"latencyMs":5930,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/realistic","repeat":4,"output":"root = Stack([dashboard], \"column\", \"l\")\ndashboard = Card([header, summary, trafficChart, breakdownTable], \"column\", \"l\")\nheader = CardHeader(\"Website traffic sources\", \"Last month\")\nsummary = TextContent(\"Traffic acquisition mix across all website sessions. Percentages total 100%.\", \"default\")\ntrafficChart = PieChart(sourceLabels, sourcePercentages, \"donut\")\nbreakdownTable = Table([sourceColumn, percentageColumn, sessionsColumn])\nsourceColumn = Col(\"Traffic source\", sourceLabels)\npercentageColumn = Col(\"Share\", [\"42%\", \"24%\", \"16%\", \"12%\", \"6%\"])\nsessionsColumn = Col(\"Estimated sessions\", [5040, 2880, 1920, 1440, 720], \"number\")\nsourceLabels = [\"Organic search\", \"Paid ads\", \"Social\", \"Direct\", \"Referrals\"]\nsourcePercentages = [42, 24, 16, 12, 6]","promptTokens":4413,"completionTokens":207,"latencyMs":6352,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/minimal","repeat":0,"output":"root = Stack([warning])\nwarning = Callout(\"warning\", \"Trial expires soon\", \"Your trial expires in 3 days. Upgrade or choose a plan to keep uninterrupted access.\")","promptTokens":4389,"completionTokens":41,"latencyMs":1563,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/adversarial","repeat":4,"output":"root = Stack([header, chartCard, dataCard], \"column\", \"l\")\nheader = Card([title, subtitle], \"clear\", \"column\", \"xs\")\ntitle = TextContent(\"Monthly trend\", \"large-heavy\")\nsubtitle = TextContent(\"January–July performance\")\nchartCard = Card([chartHeader, trendChart], \"card\", \"column\", \"m\")\nchartHeader = CardHeader(\"Monthly values\", \"Trend across seven months\")\ntrendChart = LineChart(months, [monthlySeries], \"natural\", \"Month\", \"Value\")\nmonthlySeries = Series(\"Value\", values)\ndataCard = Card([dataHeader, rawTable], \"card\", \"column\", \"m\")\ndataHeader = CardHeader(\"Raw numbers\", \"Values used in the chart\")\nrawTable = Table([monthColumn, valueColumn])\nmonthColumn = Col(\"Month\", months)\nvalueColumn = Col(\"Value\", values, \"number\")\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nvalues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]","promptTokens":4433,"completionTokens":247,"latencyMs":3097,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/adversarial","repeat":3,"output":"root = Stack([header, trendCard, rawCard], \"column\", \"l\")\nheader = CardHeader(\"Monthly performance\", \"January through July\")\ntrendCard = Card([trendTitle, trendChart], \"column\", \"m\")\ntrendTitle = TextContent(\"Trend over time\", \"large-heavy\")\ntrendChart = LineChart([\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"], [Series(\"Value\", [4200, 3800, 5100, 4900, 6300, 7100, 6800])], \"natural\", \"Month\", \"Value\")\nrawCard = Card([rawTitle, rawTable], \"column\", \"m\")\nrawTitle = TextContent(\"Raw numbers\", \"large-heavy\")\nrawTable = Table([Col(\"Month\", [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]), Col(\"Value\", [4200, 3800, 5100, 4900, 6300, 7100, 6800], \"number\")])","promptTokens":4433,"completionTokens":282,"latencyMs":5220,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/minimal","repeat":1,"output":"root = Stack([trialWarning])\ntrialWarning = Callout(\"warning\", \"Trial expires soon\", \"Your trial expires in 3 days. Upgrade or update your plan to avoid interruption.\")","promptTokens":4389,"completionTokens":42,"latencyMs":1969,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"chart/adversarial","repeat":2,"output":"root = Stack([header, summary, trendCard, dataCard], \"column\", \"l\")\nheader = Card([title, subtitle], \"clear\", \"column\", \"xs\")\ntitle = TextContent(\"Monthly trend\", \"large-heavy\")\nsubtitle = TextContent(\"January–July performance\")\nsummary = Stack([peakKpi, totalKpi, averageKpi], \"row\", \"m\", \"stretch\", \"start\", true)\npeakKpi = Card([peakLabel, peakValue, peakDetail], \"sunk\", \"column\", \"xs\")\npeakLabel = TextContent(\"Peak month\", \"small\")\npeakValue = TextContent(\"Jun · 7,100\", \"large-heavy\")\npeakDetail = TextContent(\"Highest recorded value\", \"small\")\ntotalKpi = Card([totalLabel, totalValue, totalDetail], \"sunk\", \"column\", \"xs\")\ntotalLabel = TextContent(\"Total\", \"small\")\ntotalValue = TextContent(\"38,200\", \"large-heavy\")\ntotalDetail = TextContent(\"Across 7 months\", \"small\")\naverageKpi = Card([averageLabel, averageValue, averageDetail], \"sunk\", \"column\", \"xs\")\naverageLabel = TextContent(\"Monthly average\", \"small\")\naverageValue = TextContent(\"5,457\", \"large-heavy\")\naverageDetail = TextContent(\"Rounded to nearest whole number\", \"small\")\ntrendCard = Card([trendHeader, trendChart], \"card\", \"column\", \"m\")\ntrendHeader = CardHeader(\"Monthly values\", \"Steady rise through June, followed by a small July dip\")\ntrendChart = LineChart(months, [monthlySeries], \"natural\", \"Month\", \"Value\")\nmonthlySeries = Series(\"Value\", values)\ndataCard = Card([dataHeader, dataTable], \"card\", \"column\", \"m\")\ndataHeader = CardHeader(\"Raw numbers\", \"Values used in the chart\")\ndataTable = Table([monthCol, valueCol])\nmonthCol = Col(\"Month\", months)\nvalueCol = Col(\"Value\", values, \"number\")\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nvalues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]","promptTokens":4433,"completionTokens":484,"latencyMs":6843,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/minimal","repeat":2,"output":"root = Stack([warning])\nwarning = Callout(\"warning\", \"Trial expires soon\", \"Your trial expires in 3 days. Upgrade or choose a plan to avoid interruption.\")","promptTokens":4389,"completionTokens":40,"latencyMs":2797,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/minimal","repeat":4,"output":"root = Stack([warning])\nwarning = Callout(\"warning\", \"Trial expires soon\", \"Your trial expires in 3 days. Upgrade or update your plan to avoid interruption.\")","promptTokens":4389,"completionTokens":40,"latencyMs":2209,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/minimal","repeat":3,"output":"root = Stack([warning])\nwarning = Callout(\"warning\", \"Trial ending soon\", \"Your trial expires in 3 days. Upgrade or choose a plan to avoid interruption.\")","promptTokens":4389,"completionTokens":40,"latencyMs":2422,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/realistic","repeat":1,"output":"root = Stack([noticeCard], \"column\", \"m\")\nnoticeCard = Card([header, plannedNotice, impact, timing, reassurance], \"column\", \"m\")\nheader = CardHeader(\"Planned maintenance this Sunday\", \"Scheduled service window: 02:00–06:00 UTC\")\nplannedNotice = Callout(\"info\", \"Planned maintenance — not an outage\", \"We will perform scheduled maintenance this Sunday between 02:00 and 06:00 UTC.\")\nimpact = TextCallout(\"warning\", \"Temporary service impact\", \"Payments and exports will be unavailable during the maintenance window. All other services will remain available.\")\ntiming = TextContent(\"Please complete any time-sensitive payment or export activity before 02:00 UTC, or plan to resume after 06:00 UTC.\", \"default\")\nreassurance = TextCallout(\"success\", \"What continues to work\", \"Everything except payments and exports will continue operating normally throughout the planned maintenance.\")","promptTokens":4421,"completionTokens":201,"latencyMs":2460,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/realistic","repeat":0,"output":"root = Stack([noticeCard])\nnoticeCard = Card([header, plannedNotice, impactTitle, impactList, reassurance, timing], \"column\", \"m\")\nheader = CardHeader(\"Planned maintenance notice\", \"Scheduled for Sunday, 02:00–06:00 UTC\")\nplannedNotice = Callout(\"info\", \"Planned maintenance — not an outage\", \"We will perform scheduled maintenance this Sunday between 02:00 and 06:00 UTC.\")\nimpactTitle = TextContent(\"During this maintenance window:\", \"small-heavy\")\nimpactList = MarkDownRenderer(\"- **Payments** will be temporarily unavailable.\\n- **Exports** will be temporarily unavailable.\\n- All other product features will continue to work normally.\", \"clear\")\nreassurance = Callout(\"success\", \"What to expect\", \"Services will return automatically once maintenance is complete. No action is needed from you.\")\ntiming = TextCallout(\"neutral\", \"Maintenance window\", \"Sunday, 02:00–06:00 UTC\")","promptTokens":4421,"completionTokens":211,"latencyMs":3997,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/adversarial","repeat":4,"output":"root = Stack([pageTitle, scopeNote, decisionTabs, recommendation], \"column\", \"l\")\npageTitle = Card([titleText, subtitleText], \"clear\", \"column\", \"s\")\ntitleText = TextContent(\"Top cloud providers: practical enterprise comparison\", \"large-heavy\")\nsubtitleText = TextContent(\"Sorted by broadest global capability and enterprise maturity: AWS, Microsoft Azure, Google Cloud, Oracle Cloud Infrastructure, and IBM Cloud.\", \"default\")\nscopeNote = Callout(\"info\", \"Pricing note\", \"Cloud prices vary by region, commitment, operating system, traffic path, support plan, and negotiated enterprise discount. Figures below are directional comparison points, not quotes.\")\ndecisionTabs = Tabs([overviewTab, costTab, platformTab, governanceTab, recommendationTab])\noverviewTab = TabItem(\"overview\", \"Executive comparison\", [overviewIntro, overviewTable, strengthsCard])\noverviewIntro = TextContent(\"For most organizations, the decision is driven by existing ecosystem, regional availability, workload type, commercial leverage, and operating skill—not a single list price.\", \"default\")\noverviewTable = Table([providerCol, bestForCol, globalCol, maturityCol, pricingCol, tradeoffCol])\nproviderCol = Col(\"Provider\", [\"AWS\", \"Microsoft Azure\", \"Google Cloud\", \"Oracle Cloud Infrastructure\", \"IBM Cloud\"])\nbestForCol = Col(\"Best fit\", [\"Broadest services and startup-to-enterprise scale\", \"Microsoft-centric enterprises and hybrid estates\", \"Data, AI/ML, Kubernetes, and developer platforms\", \"Oracle database/application estates and predictable networking\", \"Regulated hybrid environments and IBM software users\"])\nglobalCol = Col(\"Global footprint\", [\"Very large; broad regions, AZs, edge locations\", \"Very large; broad regions and availability zones\", \"Large and growing; strong private backbone\", \"Smaller than hyperscale leaders but broad enterprise reach\", \"Smaller public footprint; strong hybrid and industry focus\"])\nmaturityCol = Col(\"Service breadth\", [\"Market-leading breadth and depth\", \"Near-peer breadth; excellent enterprise integration\", \"Deep data/AI platform; narrower long-tail catalog\", \"Strong core IaaS, database, and networking\", \"Focused portfolio with hybrid emphasis\"])\npricingCol = Col(\"Typical cost posture\", [\"Flexible but complex; savings plans and reservations matter\", \"Competitive with enterprise agreements and hybrid benefit\", \"Often attractive sustained-use and committed-use economics\", \"Often aggressive for Oracle workloads and network-heavy designs\", \"Usually quote-driven; assess workload-specific pricing\"])\ntradeoffCol = Col(\"Main tradeoff\", [\"Operational complexity and pricing sprawl\", \"Portal/service complexity and occasional regional feature variation\", \"Less breadth for niche enterprise infrastructure services\", \"Smaller ecosystem and fewer managed-service choices\", \"Less cloud-native breadth and smaller developer ecosystem\"])\nstrengthsCard = Card([strengthsHeader, strengthsText], \"sunk\")\nstrengthsHeader = CardHeader(\"Fast selection guide\", \"Choose based on your dominant constraint\")\nstrengthsText = MarkDownRenderer(\"- **AWS:** safest default when you need maximum service choice, geographic breadth, partner ecosystem, and hiring market.\\n- **Azure:** strongest default for Microsoft identity, Windows Server, SQL Server, Microsoft 365, and hybrid governance.\\n- **Google Cloud:** strongest default for analytics, AI/ML, Kubernetes, data engineering, and clean cloud-native developer workflows.\\n- **OCI:** strongest default when Oracle database, Exadata, Fusion, or predictable network economics dominate.\\n- **IBM Cloud:** strongest default for regulated hybrid modernization, Red Hat OpenShift, and IBM software integration.\")\ncostTab = TabItem(\"cost\", \"Pricing, free tier & network\", [costIntro, costTable, costNotes])\ncostIntro = TextContent(\"The most useful cost comparison is not instance price alone: include commitments, storage requests, managed-service markup, support, observability, and outbound traffic.\", \"default\")\ncostTable = Table([costProviderCol, computeCol, commitmentCol, freeTierCol, egressCol, storageCol, billingCol])\ncostProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"IBM Cloud\"])\ncomputeCol = Col(\"Compute pricing model\", [\"On-demand, Savings Plans, Reserved Instances, Spot\", \"Pay-as-you-go, Reservations, Savings Plan, Spot\", \"On-demand, committed use, sustained-use discounts, Spot VMs\", \"On-demand, flexible shapes, commitments, preemptible capacity\", \"On-demand, reserved capacity, bare metal options\"])\ncommitmentCol = Col(\"Discount behavior\", [\"Strong with 1–3 year commitments; many options\", [\"Strong with enterprise agreements, reservations, Azure Hybrid Benefit\"], \"Sustained-use can discount eligible usage automatically; commitments deepen discounts\", \"Often highly competitive under enterprise agreements, especially Oracle estates\", \"Commonly contract and account dependent\"])\nfreeTierCol = Col(\"Free tier\", [\"12-month introductory offers plus always-free services; limits vary\", \"12-month popular services plus always-free services; credit offers vary\", \"Introductory credits for new users plus always-free quotas\", \"Notable always-free compute and database allowances, subject to capacity and policy\", \"Lite plans and limited free service tiers\"])\negressCol = Col(\"Internet egress\", [\"Usually charged by GB; some free allowance and service-specific exceptions\", \"Usually charged by GB; zone, region, and service path matter\", \"Usually charged by GB; private backbone and product path matter\", \"Frequently positioned with generous outbound allowance and lower published egress for eligible use\", \"Usually charged; obtain a workload-specific quote\"])\nstorageCol = Col(\"Storage cost drivers\", [\"S3 class, retrieval, requests, replication, lifecycle\", \"Blob tier, operations, redundancy, retrieval\", \"Cloud Storage class, retrieval, operations, replication\", \"Object Storage tiering, requests, replication, egress policy\", \"Object Storage class, requests, retrieval, replication\"])\nbillingCol = Col(\"Billing usability\", [\"Powerful but complex; Cost Explorer, CUR, budgets\", \"Strong enterprise billing; Cost Management integration\", \"Clear billing exports and FinOps tooling; simpler discount model\", \"Often simpler core pricing; validate every managed-service rate\", \"Enterprise account engagement often needed\"])\ncostNotes = Card([costNotesHeader, costNotesText], \"sunk\")\ncostNotesHeader = CardHeader(\"Cost controls to require\", \"Use these regardless of provider\")\ncostNotesText = MarkDownRenderer(\"1. Tag or label every billable resource and enforce ownership.\\n2. Set budgets, anomaly detection, and egress alerts on day one.\\n3. Compare **three-year total cost**, including support, backups, logging, NAT, load balancing, security tooling, and labor.\\n4. Benchmark your real workload before migrating: CPU shape, storage IOPS, database licensing, and network paths can reverse a headline price comparison.\\n5. Negotiate commitments only after rightsizing and utilization baselines are stable.\")\nplatformTab = TabItem(\"platform\", \"Infrastructure & developer platform\", [platformTable, runtimeCard, reliabilityCard])\nplatformTable = Table([platformProviderCol, computePlatformCol, containersCol, serverlessCol, dataAiCol, hybridCol])\nplatformProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"IBM Cloud\"])\ncomputePlatformCol = Col(\"Compute\", [\"EC2, Graviton ARM, Nitro, bare metal, GPUs\", \"VMs, Azure Stack HCI, confidential computing, GPUs\", \"Compute Engine, Tau VMs, GPUs, TPU access\", \"Flexible VM shapes, bare metal, GPUs, Exadata\", \"Virtual servers, bare metal, specialized regulated offerings\"])\ncontainersCol = Col(\"Containers & Kubernetes\", [\"ECS, EKS, Fargate\", \"AKS, Container Apps, ACI\", \"GKE, Autopilot, Cloud Run\", \"OKE, Container Instances\", \"Red Hat OpenShift, Kubernetes Service\"])\nserverlessCol = Col(\"Serverless\", [\"Lambda, Fargate, Step Functions, EventBridge\", \"Functions, Container Apps, Logic Apps, Durable Functions\", \"Cloud Run, Cloud Functions, Workflows, Eventarc\", \"Functions, API Gateway, Events\", \"Code Engine, Functions, Event Notifications\"])\ndataAiCol = Col(\"Data & AI\", [\"Redshift, Athena, EMR, SageMaker, Bedrock\", \"Fabric, Synapse, Databricks, Azure ML, AI services\", \"BigQuery, Dataflow, Vertex AI, Gemini ecosystem\", \"Autonomous Database, MySQL HeatWave, OCI AI\", \"watsonx, Db2, Cloud Pak for Data\"])\nhybridCol = Col(\"Hybrid / multicloud\", [\"Outposts, EKS Anywhere, Direct Connect\", \"Azure Arc, Stack HCI, ExpressRoute\", \"Anthos, Google Distributed Cloud, Interconnect\", \"Cloud@Customer, FastConnect, multicloud database options\", \"OpenShift, Satellite, Cloud Paks\"])\nruntimeCard = Card([runtimeHeader, runtimeText], \"card\")\nruntimeHeader = CardHeader(\"Cold starts and runtime latency\", \"There is no universal provider winner\")\nruntimeText = MarkDownRenderer(\"Cold start time depends more on runtime, package size, VPC/network initialization, concurrency configuration, CPU allocation, image caching, and dependency calls than provider brand.\\n\\n- **AWS Lambda:** typically fast for warm invocations; Java/.NET and VPC-attached functions can need mitigation. Provisioned Concurrency reduces startup variance.\\n- **Azure Functions:** performance varies by plan; Premium and dedicated options reduce scale-to-zero startup impact.\\n- **Google Cloud Run / Functions:** containers offer good control; minimum instances reduce cold-start exposure. Keep images small and startup paths lean.\\n- **OCI Functions / IBM Code Engine:** validate with representative traffic; smaller ecosystem means fewer public benchmark baselines.\\n\\nFor latency-critical paths, use minimum/provisioned capacity, regional affinity, small artifacts, connection reuse, and a load test at expected idle intervals.\")\nreliabilityCard = Card([reliabilityHeader, reliabilityText], \"sunk\")\nreliabilityHeader = CardHeader(\"Availability, uptime & disaster recovery\", \"Design beats headline SLA\")\nreliabilityText = MarkDownRenderer(\"All five providers publish service-specific SLAs, commonly around **99.9% to 99.99%+** depending on service and architecture. Credits are usually the remedy; they do not cover business loss.\\n\\nRequire: multi-zone deployment, tested backups, cross-region recovery where justified, immutable infrastructure, dependency mapping, and quarterly restore/failover exercises. Compare the SLA for the exact SKU—VM, managed database, Kubernetes control plane, object storage, load balancer—not a provider-wide uptime claim.\")\ngovernanceTab = TabItem(\"governance\", \"Regions, security, compliance & support\", [governanceTable, complianceCard, supportCard])\ngovernanceTable = Table([govProviderCol, regionsCol, identityCol, securityCol, complianceCol, supportCol])\ngovProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"IBM Cloud\"])\nregionsCol = Col(\"Regions / availability design\", [\"Large global region and multi-AZ catalog; dedicated local/edge options\", \"Large global catalog; strong sovereign and government offerings\", \"Large global region catalog; global private network is a key differentiator\", \"Broad commercial regions with strong enterprise and sovereign positioning in select markets\", \"More limited public-region footprint; industry and sovereign offerings vary\"])\nidentityCol = Col(\"Identity & access\", [\"IAM, IAM Identity Center, Organizations\", \"Microsoft Entra ID, RBAC, management groups\", \"Cloud IAM, organization policies, workforce identity federation\", \"OCI IAM, compartments, policies\", \"IAM Access Groups, enterprise identity integrations\"])\nsecurityCol = Col(\"Core security tools\", [\"KMS, GuardDuty, Security Hub, Inspector, Macie\", \"Key Vault, Defender for Cloud, Sentinel, Purview\", \"Cloud KMS, Security Command Center, Chronicle, DLP\", \"Vault, Cloud Guard, Security Zones\", \"Key Protect, Security and Compliance Center, QRadar integrations\"])\ncomplianceCol = Col(\"Compliance posture\", [\"Extensive certifications and regulated-industry programs\", \"Extensive certifications; especially strong Microsoft compliance portfolio\", \"Extensive certifications; strong data-security and sovereignty options\", \"Extensive enterprise compliance; verify exact region/service coverage\", \"Strong regulated-industry focus; verify service and location scope\"])\nsupportCol = Col(\"Support tiers\", [\"Basic plus paid developer/business/enterprise-style plans\", \"Basic plus developer, standard, professional direct, unified options\", \"Basic plus role-based and premium support offerings\", \"Basic plus paid support plans and enterprise account coverage\", \"Standard and advanced enterprise support options\"])\ncomplianceCard = Card([complianceHeader, complianceText], \"card\")\ncomplianceHeader = CardHeader(\"Compliance checklist\", \"Do not accept a logo wall as proof\")\ncomplianceText = MarkDownRenderer(\"For each provider and target region, verify the exact service supports your requirements for **SOC 2, ISO 27001, PCI DSS, HIPAA/HITRUST, FedRAMP, GDPR, data residency, encryption key control, audit logs, retention, and contractual DPA terms**. Also assess subcontractors, support access, cross-border metadata, incident notification terms, and evidence export.\")\nsupportCard = Card([supportHeader, supportText], \"sunk\")\nsupportHeader = CardHeader(\"Support reality\", \"Premium support is part of production cost\")\nsupportText = MarkDownRenderer(\"Entry-level support is generally documentation and community-oriented. Production workloads should budget for a paid support tier with 24×7 severity-one response, named technical contacts where needed, architecture reviews, escalation paths, and service-health communications. Response targets and account coverage depend on plan and contract; confirm them in writing.\")\nrecommendationTab = TabItem(\"recommendation\", \"My recommendation\", [recommendationLead, recommendationTable, nextSteps])\nrecommendationLead = Callout(\"success\", \"My default pick: AWS\", \"I would choose AWS for a greenfield organization that needs the broadest set of credible options over time: global infrastructure, service breadth, mature security controls, deep partner ecosystem, strong hiring pool, and flexible paths from simple workloads to advanced data, AI, edge, and regulated deployments.\")\nrecommendationTable = Table([scenarioCol, pickCol, rationaleCol])\nscenarioCol = Col(\"If your situation is...\", [\"Microsoft-first enterprise\", \"Data warehouse, AI/ML, Kubernetes-first product\", \"Oracle-heavy enterprise\", \"Highly regulated hybrid modernization\", \"General-purpose greenfield / maximum optionality\"])\npickCol = Col(\"I would start with\", [\"Azure\", \"Google Cloud\", \"OCI\", \"IBM Cloud or Azure/AWS with Red Hat\", \"AWS\"])\nrationaleCol = Col(\"Why\", [\"Entra ID, Windows/SQL licensing, Microsoft agreements, and Azure Arc can materially simplify operations.\", \"BigQuery, Vertex AI, GKE, Cloud Run, and data engineering are unusually cohesive.\", \"Oracle licensing, database performance, commercial terms, and interconnect patterns may dominate total cost.\", \"OpenShift and IBM ecosystem strengths can fit existing governance and platform constraints; compare alternatives carefully.\", \"AWS minimizes future service constraints and has the strongest all-around ecosystem, though it requires strong FinOps and platform governance.\"])\nnextSteps = Card([nextHeader, nextText, nextButtons], \"card\")\nnextHeader = CardHeader(\"A sensible evaluation plan\", \"Run a focused proof of value before committing\")\nnextText = MarkDownRenderer(\"Shortlist two providers. Deploy the same representative application in two regions, including identity, network, CI/CD, observability, backup, managed database, autoscaling, and a failure test. Measure monthly run rate, developer velocity, p95 latency, cold-start behavior, recovery time, policy friction, and support responsiveness. Then negotiate based on measured demand.\")\nawsButton = Button(\"Explore AWS\", Action([@OpenUrl(\"https://aws.amazon.com/\")]), \"secondary\")\nazureButton = Button(\"Explore Azure\", Action([@OpenUrl(\"https://azure.microsoft.com/\")]), \"secondary\")\ngcpButton = Button(\"Explore Google Cloud\", Action([@OpenUrl(\"https://cloud.google.com/\")]), \"secondary\")\nnextButtons = Buttons([awsButton, azureButton, gcpButton], \"row\")\nrecommendation = Card([recommendationHeader, recommendationText], \"sunk\")\nrecommendationHeader = CardHeader(\"Bottom line\", \"Pick the platform that reduces your biggest operating constraint\")\nrecommendationText = TextContent(\"AWS is my broad default. Azure is often the better economic and operational choice for Microsoft estates. Google Cloud is often the better technical choice for data and AI-centric teams. OCI can be the best commercial and performance choice for Oracle-centered workloads. IBM Cloud is a specialized hybrid choice rather than my general-purpose default.\", \"large\")","promptTokens":4438,"completionTokens":3597,"latencyMs":44258,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/adversarial","repeat":0,"output":"root = Stack([alertCard, detailsCard], \"column\", \"l\")\nalertCard = Card([failureTitle, failureMessage, failureTag], \"column\", \"l\", \"center\", \"center\")\nfailureTitle = TextContent(\"🚨 FAILED 🚨\", \"large-heavy\")\nfailureMessage = TextContent(\"THE OPERATION FAILED. IMMEDIATE ATTENTION REQUIRED.\", \"large-heavy\")\nfailureTag = Tag(\"FAILURE\", \"🚨\", \"lg\", \"danger\")\ndetailsCard = Card([detailsHeader, detailsText], \"column\", \"s\")\ndetailsHeader = CardHeader(\"Action required\", \"The requested operation could not be completed.\")\ndetailsText = TextContent(\"Please review the error, correct the issue, and try again.\", \"large\")","promptTokens":4403,"completionTokens":158,"latencyMs":2710,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/realistic","repeat":4,"output":"root = Stack([announcement, details], \"column\", \"l\")\nannouncement = Callout(\"info\", \"Planned maintenance this Sunday\", \"Scheduled maintenance will take place Sunday from 02:00 to 06:00 UTC. This is planned maintenance, not an unexpected outage.\")\ndetails = Card([header, availability, impact, reassurance], \"column\", \"m\")\nheader = CardHeader(\"Service availability during maintenance\", \"Sunday, 02:00–06:00 UTC\")\navailability = TextContent(\"Most of the platform will remain available throughout the maintenance window.\", \"default\")\nimpact = TextCallout(\"warning\", \"Temporarily unavailable\", \"Payments and exports will be unavailable during the scheduled maintenance window.\")\nreassurance = TextCallout(\"success\", \"Everything else continues working\", \"All other features and services will continue to operate normally.\")","promptTokens":4421,"completionTokens":209,"latencyMs":3718,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/adversarial","repeat":2,"output":"root = Stack([alertCard], \"column\", \"l\")\nalertCard = Card([failureTitle, failureMessage, failureDetails], \"column\", \"l\", \"center\", \"center\")\nfailureTitle = TextContent(\"🚨 FAILED 🚨\", \"large-heavy\")\nfailureMessage = TextCallout(\"danger\", \"THE OPERATION FAILED\", \"It did not complete successfully. Please review the error details and try again.\")\nfailureDetails = Callout(\"error\", \"Failure confirmed\", \"Action required: resolve the issue before continuing.\")","promptTokens":4403,"completionTokens":112,"latencyMs":2805,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/adversarial","repeat":1,"output":"root = Stack([alertCard, details], \"column\", \"l\", \"stretch\", \"center\")\nalertCard = Card([errorBanner, failureTitle, failureMessage], \"column\", \"l\", \"center\", \"center\")\nerrorBanner = Callout(\"error\", \"FAILED\", \"THE OPERATION DID NOT COMPLETE.\")\nfailureTitle = TextContent(\"❌ FAILED ❌\", \"large-heavy\")\nfailureMessage = TextContent(\"This action failed. Please review the error and try again.\", \"large\")\ndetails = TextCallout(\"danger\", \"Action required\", \"The request was unsuccessful. Nothing was saved or completed.\")","promptTokens":4403,"completionTokens":168,"latencyMs":4199,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/adversarial","repeat":3,"output":"root = Stack([alertCard], \"column\", \"l\")\nalertCard = Card([failureCallout, failureTitle, failureMessage, failureActions], \"card\", \"l\", \"center\")\nfailureCallout = Callout(\"error\", \"FAILED\", \"The operation failed and requires attention.\")\nfailureTitle = TextContent(\"❌ FAILED ❌\", \"large-heavy\")\nfailureMessage = TextContent(\"The requested operation could not be completed. Please review the error and try again.\", \"large\")\nfailureActions = Buttons([retryButton], \"row\")\nretryButton = Button(\"Try Again\", Action([@ToAssistant(\"Retry the failed operation.\")]), \"primary\", \"destructive\", \"large\")","promptTokens":4403,"completionTokens":146,"latencyMs":5134,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"data-table/adversarial","repeat":3,"output":"root = Stack([title, scope, recommendation, tabs], \"column\", \"l\")\ntitle = Card([CardHeader(\"Top 5 cloud providers\", \"Enterprise comparison — sorted by breadth, global reach, ecosystem, and operational maturity\"), TextContent(\"Ranked overall for a typical global organization: **AWS → Microsoft Azure → Google Cloud → Oracle Cloud Infrastructure → Alibaba Cloud**.\", \"large\")])\nscope = Callout(\"info\", \"How to read this\", \"Cloud prices, regions, availability commitments, and serverless cold starts vary by service, geography, contract, architecture, and date. Figures below are representative public-list-price ranges in USD, not quotes. Validate against current pricing pages and your negotiated enterprise agreement.\")\nrecommendation = Card([CardHeader(\"My pick: AWS\", \"Best default when requirements are broad or still evolving\"), TextContent(\"I would choose **AWS** for the strongest all-around combination of service breadth, mature global footprint, partner/talent availability, infrastructure options, and operational tooling. Choose **Azure** instead when Microsoft identity, Windows, SQL Server, and Microsoft 365 are strategic. Choose **Google Cloud** for data/AI-centric workloads and Kubernetes-first teams. OCI is compelling for Oracle estates and aggressive price-performance economics; Alibaba is the natural choice for China-centered operations.\")])\ntabs = Tabs([overviewTab, costTab, reliabilityTab, platformTab, supportTab, decisionTab])\noverviewTab = TabItem(\"overview\", \"At a glance\", [overviewTable, overviewNotes])\noverviewTable = Table([providerCol, rankCol, regionsCol, zonesCol, strengthCol, fitCol])\nproviderCol = Col(\"Provider\", [\"1. AWS\", \"2. Microsoft Azure\", \"3. Google Cloud\", \"4. Oracle Cloud (OCI)\", \"5. Alibaba Cloud\"])\nrankCol = Col(\"Overall fit\", [\"Best broad default\", \"Best Microsoft estate\", \"Best data / AI / Kubernetes\", \"Best Oracle estate value\", \"Best China / APAC access\"])\nregionsCol = Col(\"Public regions\", [\"~34\", \"~60+\", \"~40\", \"~50\", \"~30+\"])\nzonesCol = Col(\"Availability zones\", [\"~108\", \"300+ datacenter locations; zone count varies\", \"~121\", \"~100+ availability domains / fault domains\", \"~90+ zones\"])\nstrengthCol = Col(\"Defining strength\", [\"Largest service catalog and ecosystem\", \"Microsoft integration and hybrid tooling\", \"Analytics, ML, global network, GKE\", \"Oracle DB affinity and transparent pricing\", \"China ecosystem and regional reach\"])\nfitCol = Col(\"Choose when\", [\"You need maximum optionality\", \"Entra ID, Windows, M365 dominate\", \"BigQuery, Vertex AI, containers matter\", \"Oracle licenses / Exadata are central\", \"China compliance and local market access matter\"])\noverviewNotes = TextCallout(\"neutral\", \"Footprint nuance\", \"Region counts are not directly comparable: sovereign, government, partner, and China regions may have separate availability, billing, or operating entities. Use the exact service-by-region matrix—not marketing region totals—for architecture decisions.\")\ncostTab = TabItem(\"cost\", \"Pricing & network\", [computePricing, storagePricing, egressPricing, freeTier, costNotes])\ncomputePricing = Card([CardHeader(\"Representative on-demand compute pricing\", \"Approximate Linux general-purpose VM, US region, 2 vCPU / 8 GB equivalent\"), computeTable])\ncomputeTable = Table([computeProviderCol, hourlyCol, monthlyCol, discountCol, billingCol])\ncomputeProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nhourlyCol = Col(\"Typical hourly\", [\"~$0.09–$0.11\", \"~$0.10–$0.13\", \"~$0.07–$0.11\", \"~$0.05–$0.10\", \"~$0.08–$0.12\"])\nmonthlyCol = Col(\"730 hours\", [\"~$70–$80\", \"~$75–$95\", \"~$50–$80\", \"~$40–$75\", \"~$60–$90\"])\ndiscountCol = Col(\"Commitment options\", [\"Savings Plans, Reserved Instances, Spot\", \"Reservations, Savings Plan, Spot\", \"Committed Use Discounts, Spot VMs\", \"Flex/annual commitments, preemptible\", \"Savings plans, reserved, spot\"])\nbillingCol = Col(\"Billing granularity\", [\"Per second for many services; 60-sec minimum common\", \"Per second/minute varies by service\", \"Per second; 1-min minimum common\", \"Per second/minute varies\", \"Per second/minute varies\"])\nstoragePricing = Card([CardHeader(\"Object storage economics\", \"Representative standard object storage in common US regions\"), storageTable])\nstorageTable = Table([storageProviderCol, standardCol, archiveCol, requestCol, storageNotesCol])\nstorageProviderCol = Col(\"Provider\", [\"AWS S3\", \"Azure Blob\", \"Google Cloud Storage\", \"OCI Object Storage\", \"Alibaba OSS\"])\nstandardCol = Col(\"Standard storage / GB-month\", [\"~$0.023\", \"~$0.018–$0.020\", \"~$0.020\", \"~$0.025\", \"~$0.020–$0.023\"])\narchiveCol = Col(\"Archive / GB-month\", [\"~$0.001–$0.004\", \"~$0.001–$0.004\", \"~$0.001–$0.004\", \"~$0.002–$0.004\", \"~$0.002–$0.005\"])\nrequestCol = Col(\"Request charges\", [\"Yes; class-dependent\", \"Yes; operation-dependent\", \"Yes; class-dependent\", \"Yes; generally simple\", \"Yes; operation-dependent\"])\nstorageNotesCol = Col(\"Notable point\", [\"Largest feature set / lifecycle controls\", \"Strong ADLS integration\", \"Multi-region and dual-region options\", \"Often favorable outbound allowance\", \"Strong APAC / China coverage\"])\negressPricing = Card([CardHeader(\"Internet egress\", \"Representative public egress after any free allowance; destination and volume tiers change rates\"), egressTable])\negressTable = Table([egressProviderCol, freeEgressCol, baseEgressCol, interRegionCol, networkNoteCol])\negressProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nfreeEgressCol = Col(\"Typical allowance\", [\"100 GB/month aggregate internet egress\", \"100 GB/month\", \"Often 100 GB/month; some programs differ\", \"10 TB/month free outbound\", \"Small free allowance varies\"])\nbaseEgressCol = Col(\"Typical first-tier internet egress\", [\"~$0.09/GB\", \"~$0.08–$0.09/GB\", \"~$0.08–$0.12/GB\", \"Then ~ $0.0085/GB in many cases\", \"~$0.08–$0.12/GB\"])\ninterRegionCol = Col(\"Cross-region data\", [\"Usually charged both patterns/services\", \"Usually charged\", \"Usually charged; network design matters\", \"Usually charged\", \"Usually charged\"])\nnetworkNoteCol = Col(\"Planning takeaway\", [\"Use CloudFront / private paths\", \"Use Front Door / peering\", \"Leverage premium network and CDN\", \"Major differentiator for egress-heavy apps\", \"Check mainland China routing separately\"])\nfreeTier = Card([CardHeader(\"Free tier and credits\", \"Programs change frequently; treat as onboarding aids, not production capacity\"), freeTable])\nfreeTable = Table([freeProviderCol, offerCol, durationCol, freeNoteCol])\nfreeProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nofferCol = Col(\"New-account offer\", [\"Free tier plus promotional credits in eligible programs\", \"Commonly ~$200 credit\", \"Commonly ~$300 credit\", \"Always Free services plus trial credit\", \"Trial credits / free tier vary by market\"])\ndurationCol = Col(\"Typical credit window\", [\"Often 12 months for selected services\", \"~30 days\", \"~90 days\", \"~30 days; Always Free remains\", \"Usually short trial window\"])\nfreeNoteCol = Col(\"Useful always-free examples\", [\"Lambda, limited EC2/S3/RDS quotas\", \"Selected app, DB, and AI services\", \"Small compute, storage, serverless quotas\", \"Ampere VMs, block/object storage quotas\", \"Limited compute, storage, serverless quotas\"])\ncostNotes = Callout(\"warning\", \"Cost verdict\", \"OCI has the clearest advantage for high outbound-data workloads because of its unusually large egress allowance. For most estates, compute list price is less important than discounts, managed-service consumption, data movement, support, licensing, and governance. Run a workload-specific total-cost model.\")\nreliabilityTab = TabItem(\"reliability\", \"Reliability & compliance\", [slaTableCard, complianceCard, securityCard])\nslaTableCard = Card([CardHeader(\"Availability design and published SLA patterns\", \"Exact SLA depends on the specific service and deployment topology\"), slaTable])\nslaTable = Table([slaProviderCol, regionalDesignCol, vmSlaCol, managedServiceCol, drGuidanceCol])\nslaProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nregionalDesignCol = Col(\"Failure-domain model\", [\"Regions with 3+ AZs in many major regions\", \"Regions with availability zones and paired regions\", \"Regions with 3+ zones in many locations\", \"Availability domains plus fault domains\", \"Regions with multiple zones\"])\nvmSlaCol = Col(\"Typical multi-zone VM SLA\", [\"Up to 99.99% for correctly deployed multi-AZ\", [\"Up to 99.99% with availability zones\", \"varies by VM/service\"], [\"Up to 99.99% across zones\", \"varies by product\"], \"Often up to 99.99% with resilient design\", \"Often up to 99.99% with multi-zone design\"])\nmanagedServiceCol = Col(\"Managed service SLA\", [\"Often 99.9–99.99%+\", \"Often 99.9–99.99%+\", \"Often 99.9–99.99%+\", \"Often 99.9–99.99%+\", \"Often 99.9–99.99%\"])\ndrGuidanceCol = Col(\"Practical DR approach\", [\"Multi-AZ + multi-region for critical tier\", \"Zone-redundant + paired-region recovery\", \"Multi-zone + global load balancing\", \"AD/fault domain + second region\", \"Multi-zone + cross-region replication\"])\ncomplianceCard = Card([CardHeader(\"Compliance, sovereignty, and governance\", \"All five have extensive attestations; certification coverage is service- and region-specific\"), complianceTable])\ncomplianceTable = Table([complianceProviderCol, baselineCol, regulatedCol, sovereigntyCol, governanceCol])\ncomplianceProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nbaselineCol = Col(\"Common baseline\", [\"ISO 27001, SOC, PCI DSS, HIPAA eligible services\", \"ISO, SOC, PCI, HIPAA eligible services\", \"ISO, SOC, PCI, HIPAA eligible services\", \"ISO, SOC, PCI, HIPAA eligible services\", \"ISO, SOC, PCI and regional standards\"])\nregulatedCol = Col(\"Government / regulated strength\", [\"Strong GovCloud and regulated workload portfolio\", \"Strong government and enterprise compliance portfolio\", \"Strong assurance portfolio; verify regional availability\", \"Strong public-sector and Oracle-regulated fit\", \"Strong China-specific compliance capabilities\"])\nsovereigntyCol = Col(\"Data residency options\", [\"Commercial, gov, and sovereign offerings\", \"Broad sovereign / national cloud options\", \"Regional controls and sovereign offerings\", \"Dedicated / sovereign region options\", \"Best fit for mainland China requirements\"])\ngovernanceCol = Col(\"Core tooling\", [\"Organizations, Control Tower, Config\", \"Management Groups, Policy, Defender\", \"Resource Manager, Org Policy, SCC\", \"Compartments, IAM, Cloud Guard\", \"Resource Management, Security Center\"])\nsecurityCard = TextCallout(\"info\", \"Shared responsibility remains decisive\", \"Every provider offers encryption, customer-managed keys, identity policy, audit logs, private networking, WAF, DDoS services, and confidential-computing options. Your architecture, IAM hygiene, logging, patching, backup tests, and incident response determine the real security outcome.\")\nplatformTab = TabItem(\"platform\", \"Platform & performance\", [platformTableCard, serverlessCard, aiCard])\nplatformTableCard = Card([CardHeader(\"Core platform capabilities\", \"Relative strengths rather than exhaustive service inventory\"), platformTable])\nplatformTable = Table([platformProviderCol, containersCol, dataCol, hybridCol, databaseCol, ecosystemCol])\nplatformProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\ncontainersCol = Col(\"Containers\", [\"EKS, ECS, Fargate\", \"AKS, Container Apps\", \"GKE is often best-in-class\", \"OKE\", \"ACK\"])\ndataCol = Col(\"Analytics\", [\"Redshift, Athena, EMR, Glue\", \"Fabric, Synapse, Databricks integration\", \"BigQuery, Dataflow, Dataproc\", \"Autonomous Data Warehouse, analytics\", \"MaxCompute, DataWorks\"])\nhybridCol = Col(\"Hybrid / edge\", [\"Outposts, Local Zones, Wavelength\", \"Azure Arc, Stack HCI, edge\", \"Anthos, Distributed Cloud\", \"Cloud@Customer\", \"Apsara Stack, edge offerings\"])\ndatabaseCol = Col(\"Managed data\", [\"Very broad relational, NoSQL, cache options\", \"SQL, Cosmos DB, PostgreSQL, MySQL\", \"Cloud SQL, Spanner, Bigtable, AlloyDB\", \"Autonomous DB, Exadata, MySQL HeatWave\", \"PolarDB, ApsaraDB family\"])\necosystemCol = Col(\"Talent / partners\", [\"Largest cloud-specific market\", \"Largest enterprise Microsoft channel\", \"Strong cloud-native/data community\", \"Deep Oracle partner ecosystem\", \"Strong APAC and China partner base\"])\nserverlessCard = Card([CardHeader(\"Serverless cold-start expectations\", \"Indicative warm-up ranges; runtime, package size, VPC/networking, concurrency, and provisioned capacity matter more than vendor brand\"), serverlessTable])\nserverlessTable = Table([serverlessProviderCol, serviceCol, typicalColdCol, mitigationCol])\nserverlessProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nserviceCol = Col(\"Primary function service\", [\"Lambda\", \"Azure Functions\", \"Cloud Functions / Cloud Run\", \"Functions\", \"Function Compute\"])\ntypicalColdCol = Col(\"Common cold start range\", [\"~100 ms to several seconds; JVM can be longer\", \"~1–10+ sec on Consumption; plan-dependent\", \"~100 ms to several sec; Cloud Run depends on image/runtime\", \"~sub-second to several sec\", \"~100 ms to several sec\"])\nmitigationCol = Col(\"Best mitigation\", [\"Provisioned Concurrency, SnapStart where eligible\", \"Premium/Flex plans, always-ready instances\", \"Minimum instances, smaller images\", \"Provisioned capacity / lighter dependencies\", \"Reserved instances / smaller packages\"])\naiCard = TextCallout(\"success\", \"AI platform view\", \"Google Cloud is especially compelling for BigQuery + Vertex AI and Kubernetes-centric ML workflows. Azure is compelling for Microsoft-integrated AI and enterprise adoption. AWS offers the broadest menu through Bedrock, SageMaker, and partner models. OCI is attractive for Oracle data and GPU economics; Alibaba is strong for regional language and China-market ecosystems.\")\nsupportTab = TabItem(\"support\", \"Support & operations\", [supportTableCard, operationsCard])\nsupportTableCard = Card([CardHeader(\"Support tiers and operational experience\", \"Commercial terms differ by contract, spend, and geography\"), supportTable])\nsupportTable = Table([supportProviderCol, entryCol, businessCol, premiumCol, responseCol, toolingCol])\nsupportProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nentryCol = Col(\"Entry support\", [\"Basic included; Developer paid\", \"Basic included; Developer paid\", \"Basic included; Standard paid\", \"Basic included\", \"Basic included / paid options\"])\nbusinessCol = Col(\"Business tier\", [\"Business Support\", \"Professional Direct / Unified options\", \"Enhanced Support\", \"Advanced Support\", \"Business support offerings\"])\npremiumCol = Col(\"Premium tier\", [\"Enterprise Support\", \"Unified Enterprise\", \"Premium Support\", \"Premier Support\", \"Enterprise support offerings\"])\nresponseCol = Col(\"Critical-case response\", [\"Often target around 15 min at high tiers\", [\"Often target around 1 hour or faster\", \"contract-dependent\"], [\"Often target around 15 min at premium\", \"contract-dependent\"], \"Contract-dependent premium response\", \"Contract and region-dependent\"])\ntoolingCol = Col(\"Operations tooling\", [\"CloudWatch, X-Ray, CloudTrail, Health\", \"Azure Monitor, Log Analytics, Advisor\", \"Cloud Operations suite, Error Reporting\", \"Monitoring, Logging, APM, Cloud Advisor\", \"CloudMonitor, ActionTrail, ARMS\"])\noperationsCard = Card([CardHeader(\"Operational trade-offs\"), operationsTable])\noperationsTable = Table([opsProviderCol, learningCol, lockinCol, procurementCol, maturityCol])\nopsProviderCol = Col(\"Provider\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nlearningCol = Col(\"Learning curve\", [\"High breadth; many choices\", \"Easier for Microsoft admins; broad platform\", \"Clean developer experience; strong data concepts\", \"Straightforward for Oracle teams\", \"Higher friction outside core APAC markets\"])\nlockinCol = Col(\"Lock-in risk\", [\"High if deep managed-service adoption\", \"High with Microsoft stack integration\", \"Moderate-high with BigQuery / managed data\", \"High around Oracle database services\", \"High for China-specific integrations\"])\nprocurementCol = Col(\"Enterprise buying\", [\"Very mature marketplace and partners\", \"Excellent EA / Microsoft procurement fit\", \"Strong but smaller enterprise channel\", \"Strong Oracle account alignment\", \"Strong in China / APAC\"])\nmaturityCol = Col(\"Operational maturity\", [\"Excellent; huge community\", \"Excellent; enterprise standard\", \"Excellent; cloud-native standard\", \"Strong; smaller ecosystem\", \"Strong regionally\"])\ndecisionTab = TabItem(\"decision\", \"Decision guide\", [decisionSteps, finalMatrix])\ndecisionSteps = Steps([stepOne, stepTwo, stepThree, stepFour])\nstepOne = StepsItem(\"Start with non-negotiables\", \"Filter by required countries, data residency, sovereign-cloud needs, latency, hardware, licensing, and mandated certifications. This can eliminate providers before a feature comparison begins.\")\nstepTwo = StepsItem(\"Model the workload, not a generic VM\", \"Price compute, managed databases, storage classes, backup, observability, support, committed-use discounts, and especially ingress/egress. Include 24–36 month growth and a migration exit scenario.\")\nstepThree = StepsItem(\"Run a production-shaped proof of concept\", \"Test latency, failure recovery, IAM integration, CI/CD, observability, restore time, quota workflows, and real serverless initialization using your runtime and package—not synthetic hello-world tests.\")\nstepFour = StepsItem(\"Optimize for operating model\", \"The best cloud is the one your team can secure, govern, deploy, and recover reliably. Existing identity, data platform, skills, support relationships, and compliance evidence often outweigh small list-price differences.\")\nfinalMatrix = Card([CardHeader(\"Shortlist recommendation matrix\"), finalTable, TextContent(\"**Default selection:** AWS. **Microsoft-first enterprise:** Azure. **Data/AI and Kubernetes-first organization:** Google Cloud. **Oracle-heavy estate or egress-intensive architecture:** OCI. **China-first operating footprint:** Alibaba Cloud.\", \"large-heavy\")])\nfinalTable = Table([scenarioCol, bestCol, whyCol])\nscenarioCol = Col(\"Primary scenario\", [\"Maximum service breadth / global expansion\", \"Microsoft licensing and enterprise identity\", \"Modern analytics and ML platform\", \"Oracle database modernization\", \"Mainland China delivery\"])\nbestCol = Col(\"Best fit\", [\"AWS\", \"Azure\", \"Google Cloud\", \"OCI\", \"Alibaba Cloud\"])\nwhyCol = Col(\"Reason\", [\"Unmatched catalog, ecosystem, and mature global operations\", \"Natural fit with Entra ID, Windows, SQL Server, M365, and Azure Arc\", \"BigQuery, Vertex AI, GKE, and strong global network\", \"Database affinity, Exadata path, pricing clarity, 10 TB free egress\", \"Local ecosystem, compliance context, and regional market access\"])","promptTokens":4438,"completionTokens":4479,"latencyMs":53926,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/adversarial","repeat":4,"output":"root = Stack([alertCard], \"column\", \"xl\", \"center\", \"center\")\nalertCard = Card([errorBanner, errorTitle, errorMessage, retryButton], \"column\", \"l\", \"center\", \"center\")\nerrorBanner = TextCallout(\"danger\", \"⚠️ FAILED ⚠️\", \"THE OPERATION DID NOT COMPLETE.\")\nerrorTitle = TextContent(\"❌ FAILED ❌\", \"large-heavy\")\nerrorMessage = TextContent(\"Something went wrong. Please review the error and try again.\", \"large\")\nretryButton = Button(\"Try Again\", Action([@ToAssistant(\"Please help me retry the failed operation.\")]), \"primary\", \"destructive\", \"large\")","promptTokens":4403,"completionTokens":186,"latencyMs":3624,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/minimal","repeat":0,"output":"root = Stack([invoiceCard])\ninvoiceCard = Card([invoiceHeader, invoiceText, invoiceButtons])\ninvoiceHeader = CardHeader(\"Invoice\", \"Your latest invoice is ready to download.\")\ninvoiceText = TextContent(\"Download a copy for your records.\")\ninvoiceButtons = Buttons([downloadButton])\ndownloadButton = Button(\"Download invoice\", Action([@OpenUrl(\"https://example.com/invoices/latest.pdf\")]), \"primary\")","promptTokens":4388,"completionTokens":90,"latencyMs":3753,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/minimal","repeat":1,"output":"root = Stack([invoiceCard])\ninvoiceCard = Card([invoiceHeader, invoiceText, invoiceButtons])\ninvoiceHeader = CardHeader(\"Invoice\", \"Your latest billing document is ready.\")\ninvoiceText = TextContent(\"Download a copy of your invoice for your records.\")\ninvoiceButtons = Buttons([downloadInvoice])\ndownloadInvoice = Button(\"Download invoice\", Action([@OpenUrl(\"https://example.com/invoices/latest.pdf\")]), \"primary\")","promptTokens":4388,"completionTokens":92,"latencyMs":2171,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/minimal","repeat":2,"output":"root = Stack([invoiceCard])\ninvoiceCard = Card([invoiceHeader, invoiceText, invoiceButtons])\ninvoiceHeader = CardHeader(\"Invoice\", \"Your invoice is ready to download\")\ninvoiceText = TextContent(\"Download a PDF copy for your records.\")\ninvoiceButtons = Buttons([downloadButton])\ndownloadButton = Button(\"Download invoice\", Action([@OpenUrl(\"https://example.com/invoices/latest.pdf\")]), \"primary\")","promptTokens":4388,"completionTokens":90,"latencyMs":2012,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/minimal","repeat":3,"output":"root = Stack([invoiceCard])\ninvoiceCard = Card([header, description, actions])\nheader = CardHeader(\"Invoice\", \"Your latest invoice is ready to download.\")\ndescription = TextContent(\"Download a PDF copy for your records.\")\nactions = Buttons([downloadButton])\ndownloadButton = Button(\"Download invoice\", Action([@OpenUrl(\"https://example.com/invoices/latest.pdf\")]), \"primary\")","promptTokens":4388,"completionTokens":85,"latencyMs":2350,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/realistic","repeat":0,"output":"root = Stack([subscriptionFooter], \"column\", \"m\")\nsubscriptionFooter = Card([footerHeader, footerText, footerActions], \"column\", \"m\")\nfooterHeader = CardHeader(\"Manage your subscription\", \"Choose what you would like to do next.\")\nfooterText = TextContent(\"Upgrade to annual billing to save on your plan, or cancel your current subscription.\", \"default\")\nfooterActions = Buttons([upgradeButton, cancelButton], \"row\")\nupgradeButton = Button(\"Upgrade to annual plan\", Action([@ToAssistant(\"I want to upgrade to the annual plan.\")]), \"primary\", \"normal\")\ncancelButton = Button(\"Cancel subscription\", Action([@ToAssistant(\"I want to cancel my subscription.\")]), \"secondary\", \"destructive\")","promptTokens":4417,"completionTokens":157,"latencyMs":2159,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/realistic","repeat":2,"output":"root = Stack([noticeCard, detailsCard, actions], \"column\", \"l\")\nnoticeCard = Card([noticeHeader, plannedCallout, summaryText], \"column\", \"m\")\nnoticeHeader = CardHeader(\"Planned maintenance\", \"Sunday, 02:00–06:00 UTC\")\nplannedCallout = Callout(\"info\", \"Scheduled service maintenance\", \"This is planned maintenance, not an unexpected outage. Payments and exports will be temporarily unavailable during this window.\")\nsummaryText = TextContent(\"All other product features will remain available and continue operating normally. We recommend completing any time-sensitive payment or export activity before the maintenance window begins.\", \"default\")\ndetailsCard = Card([detailsHeader, detailsTable, customerMessage], \"column\", \"m\")\ndetailsHeader = CardHeader(\"Maintenance details\", \"Customer-facing service impact\")\ndetailsTable = Table([serviceCol, availabilityCol])\nserviceCol = Col(\"Service\", [\"Payments\", \"Exports\", \"All other features\"])\navailabilityCol = Col(\"Availability during maintenance\", [\"Temporarily unavailable\", \"Temporarily unavailable\", \"Available as normal\"])\ncustomerMessage = TextCallout(\"neutral\", \"Suggested customer notice\", \"We will perform planned maintenance on Sunday from 02:00 to 06:00 UTC. During this time, payments and exports will be temporarily unavailable. All other features will remain available. This is scheduled maintenance, not an outage.\")\nactions = Buttons([copyButton, statusButton])\ncopyButton = Button(\"Prepare customer announcement\", Action([@ToAssistant(\"Prepare a polished customer announcement for the planned Sunday maintenance window, 02:00–06:00 UTC, noting that payments and exports will be unavailable while all other features remain available.\")]), \"primary\")\nstatusButton = Button(\"Draft status page update\", Action([@ToAssistant(\"Draft a concise status page update for planned maintenance this Sunday from 02:00–06:00 UTC. Payments and exports will be unavailable; everything else remains operational. Make clear this is planned maintenance, not an outage.\")]), \"secondary\")","promptTokens":4421,"completionTokens":422,"latencyMs":15722,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/minimal","repeat":4,"output":"root = Stack([invoiceCard])\ninvoiceCard = Card([invoiceHeader, invoiceText, invoiceButtons])\ninvoiceHeader = CardHeader(\"Invoice ready\", \"Your invoice is available to download.\")\ninvoiceText = TextContent(\"Download a PDF copy for your records.\")\ninvoiceButtons = Buttons([downloadButton])\ndownloadButton = Button(\"Download invoice\", Action([@OpenUrl(\"https://example.com/invoices/download.pdf\")]), \"primary\")","promptTokens":4388,"completionTokens":91,"latencyMs":4168,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/realistic","repeat":2,"output":"root = Stack([subscriptionSummary, nextSteps])\nsubscriptionSummary = Card([subscriptionHeader, subscriptionText])\nsubscriptionHeader = CardHeader(\"Subscription\", \"Manage your current plan\")\nsubscriptionText = TextContent(\"Your subscription is active. Choose what you would like to do next.\", \"default\")\nnextSteps = Card([nextHeader, nextText, actionButtons], \"column\", \"m\")\nnextHeader = CardHeader(\"Next steps\", \"You can change or end your subscription\")\nnextText = TextContent(\"Upgrade to annual billing to save on your plan, or cancel your subscription. Cancellation is permanent and will end access at the close of your current billing period.\", \"small\")\nactionButtons = Buttons([upgradeButton, cancelButton])\nupgradeButton = Button(\"Upgrade to annual plan\", upgradeAction, \"primary\")\ncancelButton = Button(\"Cancel subscription\", cancelAction, \"secondary\", \"destructive\")\nupgradeAction = Action([@ToAssistant(\"I want to upgrade to the annual plan.\")])\ncancelAction = Action([@ToAssistant(\"I want to cancel my subscription.\")])","promptTokens":4417,"completionTokens":221,"latencyMs":3048,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/adversarial","repeat":0,"output":"root = Stack([button])\nbutton = Button(\"Button\")","promptTokens":4404,"completionTokens":16,"latencyMs":2158,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/realistic","repeat":3,"output":"root = Stack([subscriptionFooter], \"column\", \"m\")\nsubscriptionFooter = Card([footerTitle, footerText, footerActions], \"column\", \"m\")\nfooterTitle = CardHeader(\"Manage your subscription\", \"Choose what you’d like to do next.\")\nfooterText = TextContent(\"Upgrade to annual billing to save on your plan, or cancel if you no longer need your subscription.\", \"default\")\nfooterActions = Buttons([upgradeButton, cancelButton], \"row\")\nupgradeButton = Button(\"Upgrade to annual plan\", Action([@ToAssistant(\"Upgrade my subscription to the annual plan\")]), \"primary\")\ncancelButton = Button(\"Cancel subscription\", Action([@ToAssistant(\"Cancel my subscription\")]), \"secondary\", \"destructive\")","promptTokens":4417,"completionTokens":154,"latencyMs":3634,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/realistic","repeat":4,"output":"root = Stack([subscriptionEnd])\nsubscriptionEnd = Card([endHeader, endText, endActions], \"column\", \"m\")\nendHeader = CardHeader(\"Subscription options\", \"Choose what you’d like to do next.\")\nendText = TextContent(\"Upgrade to annual billing to save on your plan, or cancel your subscription. Cancellation is permanent and will end renewal.\")\nendActions = Buttons([upgradeButton, cancelButton], \"row\")\nupgradeButton = Button(\"Upgrade to annual plan\", Action([@ToAssistant(\"I want to upgrade to the annual plan.\")]), \"primary\")\ncancelButton = Button(\"Cancel subscription\", Action([@ToAssistant(\"I want to cancel my subscription.\")]), \"secondary\", \"destructive\")","promptTokens":4417,"completionTokens":151,"latencyMs":3500,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"callout/realistic","repeat":3,"output":"root = Stack([noticeCard])\nnoticeCard = Card([header, plannedNotice, impactCallout, details, customerMessage], \"column\", \"m\")\nheader = CardHeader(\"Planned maintenance notice\", \"Sunday, 02:00–06:00 UTC\")\nplannedNotice = Callout(\"info\", \"Scheduled maintenance\", \"We will perform planned maintenance this Sunday from 02:00 to 06:00 UTC. This is scheduled work, not an unexpected outage.\")\nimpactCallout = Callout(\"warning\", \"Services affected\", \"Payments and exports will be unavailable during the maintenance window. All other services will remain available.\")\ndetails = TextContent(\"**Maintenance window:** Sunday, 02:00–06:00 UTC\\n\\nCustomers may be unable to process payments or create/download exports during this time. Please plan any time-sensitive payment or export activity outside the window.\")\ncustomerMessage = TextCallout(\"neutral\", \"Suggested customer message\", \"We’re performing planned maintenance this Sunday from 02:00–06:00 UTC. Payments and exports will be temporarily unavailable during this scheduled window. All other features will continue to operate normally. Thank you for your understanding.\")","promptTokens":4421,"completionTokens":243,"latencyMs":19187,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/realistic","repeat":1,"output":"root = Stack([subscriptionFooter], \"column\", \"m\")\nsubscriptionFooter = Card([footerTitle, footerDescription, footerActions], \"column\", \"m\")\nfooterTitle = CardHeader(\"Subscription options\", \"Choose what you would like to do next.\")\nfooterDescription = TextContent(\"Upgrade to annual billing to save on your plan, or cancel your subscription. Cancellation is permanent and will end access at the close of your current billing period.\", \"default\")\nfooterActions = Buttons([upgradeButton, cancelButton], \"row\")\nupgradeButton = Button(\"Upgrade to annual plan\", Action([@ToAssistant(\"I want to upgrade to the annual plan.\")]), \"primary\", \"normal\", \"medium\")\ncancelButton = Button(\"Cancel subscription\", Action([@ToAssistant(\"I want to cancel my subscription.\")]), \"secondary\", \"destructive\", \"medium\")","promptTokens":4417,"completionTokens":177,"latencyMs":6238,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/adversarial","repeat":1,"output":"root = Stack([button])\nbutton = Button(\"Button\")","promptTokens":4404,"completionTokens":16,"latencyMs":2377,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/adversarial","repeat":2,"output":"root = Stack([button])\nbutton = Button(\"Button\")","promptTokens":4404,"completionTokens":16,"latencyMs":1233,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/adversarial","repeat":3,"output":"root = Stack([button])\nbutton = Button(\"Button\")","promptTokens":4404,"completionTokens":16,"latencyMs":1672,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"button/adversarial","repeat":4,"output":"root = Stack([button])\nbutton = Button(\"Click me\")","promptTokens":4404,"completionTokens":17,"latencyMs":1755,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/minimal","repeat":0,"output":"root = Stack([header, progress, checklist, tip], \"column\", \"l\")\nheader = Card([title, subtitle], \"clear\", \"column\", \"s\")\ntitle = TextContent(\"New Laptop Setup Checklist\", \"large-heavy\")\nsubtitle = TextContent(\"Work through these essentials to get secure, productive, and backed up.\", \"default\")\nprogress = Callout(\"info\", \"Start with security\", \"Complete account, update, and backup steps before installing lots of apps.\")\nchecklist = Accordion([securitySection, updatesSection, setupSection, appsSection, backupSection, cleanupSection])\nsecuritySection = AccordionItem(\"security\", \"1. Secure your accounts\", [securityList])\nsecurityList = MarkDownRenderer(\"- [ ] Sign in with your primary account\\n- [ ] Set a strong device password or PIN\\n- [ ] Enable fingerprint or face unlock, if available\\n- [ ] Turn on full-disk encryption\\n- [ ] Enable two-factor authentication for your main accounts\\n- [ ] Review privacy and location-sharing settings\")\nupdatesSection = AccordionItem(\"updates\", \"2. Update the system\", [updatesList])\nupdatesList = MarkDownRenderer(\"- [ ] Install all operating system updates\\n- [ ] Update drivers, firmware, and BIOS/UEFI if offered\\n- [ ] Restart after updates\\n- [ ] Enable automatic updates\\n- [ ] Check that built-in security protection is active\")\nsetupSection = AccordionItem(\"settings\", \"3. Configure essentials\", [settingsList])\nsettingsList = MarkDownRenderer(\"- [ ] Set your time zone, language, and keyboard layout\\n- [ ] Connect to trusted Wi-Fi networks\\n- [ ] Pair Bluetooth accessories\\n- [ ] Configure display scaling, resolution, and night mode\\n- [ ] Set power, sleep, and battery preferences\\n- [ ] Add a printer if needed\")\nappsSection = AccordionItem(\"apps\", \"4. Install core apps\", [appsList])\nappsList = MarkDownRenderer(\"- [ ] Install your preferred web browser\\n- [ ] Install a password manager\\n- [ ] Install office, notes, and PDF tools\\n- [ ] Install video meeting and messaging apps\\n- [ ] Install developer, creative, or work-specific tools\\n- [ ] Pin frequently used apps to the dock or taskbar\")\nbackupSection = AccordionItem(\"backup\", \"5. Set up sync and backups\", [backupList])\nbackupList = MarkDownRenderer(\"- [ ] Sign in to cloud storage\\n- [ ] Enable automatic document and photo sync\\n- [ ] Configure scheduled backups to an external drive or cloud service\\n- [ ] Create a recovery drive or recovery media\\n- [ ] Verify you can restore a test file\\n- [ ] Store recovery keys somewhere safe\")\ncleanupSection = AccordionItem(\"cleanup\", \"6. Final checks\", [cleanupList])\ncleanupList = MarkDownRenderer(\"- [ ] Remove unwanted preinstalled apps\\n- [ ] Review startup apps and disable unnecessary ones\\n- [ ] Organize folders for documents, downloads, and projects\\n- [ ] Test camera, microphone, speakers, and charging\\n- [ ] Record the serial number and warranty information\\n- [ ] Restart once more and confirm everything works\")\ntip = TextCallout(\"success\", \"You're ready\", \"Once the essentials are complete, install only the tools you actually need to keep the laptop fast and uncluttered.\")","promptTokens":4388,"completionTokens":691,"latencyMs":7828,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/minimal","repeat":4,"output":"root = Stack([title, intro, checklist], \"column\", \"l\")\ntitle = TextContent(\"New Laptop Setup Checklist\", \"large-heavy\")\nintro = TextContent(\"Work through these essentials to get your laptop secure, updated, and ready for daily use.\")\nchecklist = Accordion([securitySection, updatesSection, appsSection, dataSection, personalizationSection, maintenanceSection])\nsecuritySection = AccordionItem(\"security\", \"1. Secure the device\", [securityText, securityTags])\nsecurityText = TextContent(\"□ Create a strong account password\\n□ Enable fingerprint or face sign-in\\n□ Turn on full-disk encryption\\n□ Enable automatic screen lock\\n□ Set up device tracking / Find My Device\\n□ Install and configure a password manager\\n□ Turn on two-factor authentication for your main accounts\")\nsecurityTags = TagBlock([\"Password\", \"Encryption\", \"2FA\", \"Device tracking\"])\nupdatesSection = AccordionItem(\"updates\", \"2. Update the system\", [updatesText, updatesTags])\nupdatesText = TextContent(\"□ Install operating system updates\\n□ Update drivers and firmware\\n□ Update the web browser\\n□ Enable automatic updates\\n□ Check battery and power settings\")\nupdatesTags = TagBlock([\"OS\", \"Drivers\", \"Firmware\", \"Battery\"])\nappsSection = AccordionItem(\"apps\", \"3. Install essential apps\", [appsText, appsTags])\nappsText = TextContent(\"□ Install your preferred browser\\n□ Install office or productivity tools\\n□ Install video calling software\\n□ Install cloud storage client\\n□ Install antivirus or endpoint protection if required\\n□ Install developer, creative, or work-specific tools\\n□ Remove unwanted preinstalled software\")\nappsTags = TagBlock([\"Browser\", \"Productivity\", \"Cloud storage\", \"Work tools\"])\ndataSection = AccordionItem(\"data\", \"4. Set up files and backups\", [dataText, dataTags])\ndataText = TextContent(\"□ Sign in to cloud storage\\n□ Sync important documents and photos\\n□ Transfer files from your old computer\\n□ Set up automatic backups\\n□ Create a recovery drive or recovery key\\n□ Test that a backup can be accessed\")\ndataTags = TagBlock([\"Backup\", \"Cloud sync\", \"Recovery\"])\npersonalizationSection = AccordionItem(\"personal\", \"5. Personalize your workspace\", [personalText, personalTags])\npersonalText = TextContent(\"□ Configure Wi-Fi and Bluetooth devices\\n□ Connect printer, monitor, and dock\\n□ Set display scaling, brightness, and night mode\\n□ Configure keyboard layout and shortcuts\\n□ Set default apps for mail, browser, and files\\n□ Organize desktop, taskbar, and app launcher\")\npersonalTags = TagBlock([\"Display\", \"Keyboard\", \"Accessories\", \"Defaults\"])\nmaintenanceSection = AccordionItem(\"maintenance\", \"6. Prepare for ongoing maintenance\", [maintenanceText, maintenanceTags])\nmaintenanceText = TextContent(\"□ Review storage capacity\\n□ Enable battery health or charging limits if available\\n□ Set a monthly update reminder\\n□ Keep recovery codes in a safe place\\n□ Record the serial number and purchase details\\n□ Register warranty coverage\\n□ Create a simple folder structure for documents\")\nmaintenanceTags = TagBlock([\"Storage\", \"Warranty\", \"Recovery codes\", \"Routine\"])","promptTokens":4388,"completionTokens":695,"latencyMs":8299,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/minimal","repeat":2,"output":"root = Stack([header, essentials, security, setup, maintenance], \"column\", \"l\")\nheader = Card([title, intro], \"clear\", \"column\", \"s\")\ntitle = TextContent(\"New Laptop Setup Checklist\", \"large-heavy\")\nintro = TextContent(\"Work through these essentials to get your laptop secure, updated, and ready to use.\")\nessentials = Accordion([updatesItem, accountsItem, appsItem, filesItem])\nupdatesItem = AccordionItem(\"updates\", \"1. Update the system\", [updatesList])\naccountsItem = AccordionItem(\"accounts\", \"2. Set up accounts and access\", [accountsList])\nappsItem = AccordionItem(\"apps\", \"3. Install essential software\", [appsList])\nfilesItem = AccordionItem(\"files\", \"4. Move and organize your files\", [filesList])\nupdatesList = MarkDownRenderer(\"- [ ] Connect to a trusted Wi-Fi network\\n- [ ] Install operating system updates\\n- [ ] Update drivers and firmware\\n- [ ] Restart if prompted\\n- [ ] Enable automatic updates\")\naccountsList = MarkDownRenderer(\"- [ ] Sign in with your main account\\n- [ ] Set a strong device password or PIN\\n- [ ] Configure fingerprint or face sign-in if available\\n- [ ] Sign in to your password manager\\n- [ ] Add your email, calendar, and cloud storage accounts\")\nappsList = MarkDownRenderer(\"- [ ] Install a modern web browser\\n- [ ] Install office or productivity apps\\n- [ ] Install video conferencing software\\n- [ ] Install your password manager\\n- [ ] Install work, school, or development tools\\n- [ ] Remove unwanted preinstalled software\")\nfilesList = MarkDownRenderer(\"- [ ] Sign in to cloud backup or sync storage\\n- [ ] Transfer documents, photos, and project files\\n- [ ] Create folders for work, personal, and downloads\\n- [ ] Set default save locations\\n- [ ] Confirm important files have synced successfully\")\nsecurity = Card([securityHeader, securityList, securityNote], \"sunk\", \"column\", \"m\")\nsecurityHeader = CardHeader(\"Security and backup\", \"Protect the laptop and make recovery easy.\")\nsecurityList = MarkDownRenderer(\"- [ ] Turn on full-disk encryption\\n- [ ] Enable the built-in firewall\\n- [ ] Confirm antivirus or endpoint protection is active\\n- [ ] Turn on device tracking / find-my-device\\n- [ ] Enable automatic cloud backup\\n- [ ] Create a recovery drive or save recovery keys securely\\n- [ ] Review privacy, location, camera, and microphone permissions\")\nsecurityNote = Callout(\"info\", \"Recovery tip\", \"Store recovery keys and backup codes in your password manager or another secure location.\")\nsetup = Card([setupHeader, setupList], \"card\", \"column\", \"m\")\nsetupHeader = CardHeader(\"Personalize your workspace\", \"A few adjustments can make daily use more comfortable.\")\nsetupList = MarkDownRenderer(\"- [ ] Adjust display scaling, brightness, and night mode\\n- [ ] Configure keyboard layout, shortcuts, and trackpad settings\\n- [ ] Connect printers, external displays, headphones, and other accessories\\n- [ ] Set your preferred browser, search engine, and default apps\\n- [ ] Configure notifications and focus/do-not-disturb settings\\n- [ ] Test webcam, microphone, speakers, and Bluetooth\")\nmaintenance = Card([maintenanceHeader, maintenanceList, finishButtons], \"card\", \"column\", \"m\")\nmaintenanceHeader = CardHeader(\"Final checks\", \"Finish with a quick readiness test.\")\nmaintenanceList = MarkDownRenderer(\"- [ ] Restart the laptop once more\\n- [ ] Test Wi-Fi, charging, sleep, and battery status\\n- [ ] Verify backups are running\\n- [ ] Bookmark important sites and support pages\\n- [ ] Keep the original charger, receipt, and warranty details in a safe place\")\nfinishButtons = Buttons([securityHelp, backupHelp])\nsecurityHelp = Button(\"Explain laptop security\", Action([@ToAssistant(\"Explain the most important laptop security settings and how to configure them.\")]), \"secondary\")\nbackupHelp = Button(\"Help me choose a backup plan\", Action([@ToAssistant(\"Help me choose a practical backup plan for my new laptop.\")]), \"primary\")","promptTokens":4388,"completionTokens":871,"latencyMs":9609,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/minimal","repeat":1,"output":"root = Stack([header, essentials, security, setup, maintenance], \"column\", \"l\")\nheader = Card([title, intro], \"clear\", \"column\", \"s\")\ntitle = TextContent(\"New Laptop Setup Checklist\", \"large-heavy\")\nintro = TextContent(\"Work through these essentials to get your laptop secure, updated, and ready for everyday use.\")\nessentials = Card([essentialsHeader, essentialsList], \"card\", \"column\", \"m\")\nessentialsHeader = CardHeader(\"1. First-day essentials\", \"Get the system running smoothly.\")\nessentialsList = Steps([essentialStep1, essentialStep2, essentialStep3, essentialStep4])\nessentialStep1 = StepsItem(\"Complete initial setup\", \"Choose your language and region, connect to Wi-Fi, sign in with your account, and set the correct time zone.\")\nessentialStep2 = StepsItem(\"Install system updates\", \"Run the operating system update tool repeatedly until no important updates remain. Restart when prompted.\")\nessentialStep3 = StepsItem(\"Update drivers and firmware\", \"Use the laptop manufacturer’s support app or website to install current graphics, Wi-Fi, audio, BIOS, and firmware updates.\")\nessentialStep4 = StepsItem(\"Remove unwanted preinstalled apps\", \"Uninstall trial software, duplicate utilities, and promotional apps you do not plan to use.\")\nsecurity = Card([securityHeader, securityList], \"card\", \"column\", \"m\")\nsecurityHeader = CardHeader(\"2. Secure the device\", \"Protect accounts, data, and physical access.\")\nsecurityList = Steps([securityStep1, securityStep2, securityStep3, securityStep4, securityStep5])\nsecurityStep1 = StepsItem(\"Use a strong sign-in method\", \"Set a strong password and enable fingerprint, face recognition, or a PIN where available.\")\nsecurityStep2 = StepsItem(\"Turn on full-disk encryption\", \"Enable BitLocker, FileVault, or your operating system’s equivalent so files remain protected if the laptop is lost.\")\nsecurityStep3 = StepsItem(\"Enable automatic backups\", \"Back up important folders to a trusted cloud service, an external drive, or both.\")\nsecurityStep4 = StepsItem(\"Set up multi-factor authentication\", \"Enable MFA for your main email, cloud storage, password manager, and financial accounts.\")\nsecurityStep5 = StepsItem(\"Review privacy and recovery settings\", \"Adjust permissions for location, camera, microphone, and diagnostics. Create a recovery drive or confirm account recovery details.\")\nsetup = Card([setupHeader, setupList], \"card\", \"column\", \"m\")\nsetupHeader = CardHeader(\"3. Install and organize\", \"Add the tools you need without clutter.\")\nsetupList = Steps([setupStep1, setupStep2, setupStep3, setupStep4, setupStep5])\nsetupStep1 = StepsItem(\"Install a password manager\", \"Store unique passwords securely and save recovery codes for important accounts.\")\nsetupStep2 = StepsItem(\"Choose a web browser and sync it\", \"Sign in to sync bookmarks, passwords, extensions, and open tabs across devices.\")\nsetupStep3 = StepsItem(\"Install core applications\", \"Add office tools, video calling, messaging, cloud storage, PDF tools, media apps, and any work or school software.\")\nsetupStep4 = StepsItem(\"Configure email, calendar, and cloud storage\", \"Connect your main accounts and verify that files sync correctly.\")\nsetupStep5 = StepsItem(\"Organize files and shortcuts\", \"Create folders for documents, photos, projects, and downloads. Pin frequently used apps and websites.\")\nmaintenance = Card([maintenanceHeader, maintenanceList, reminder], \"card\", \"column\", \"m\")\nmaintenanceHeader = CardHeader(\"4. Test and maintain\", \"Make sure everything works before you rely on it.\")\nmaintenanceList = Steps([maintenanceStep1, maintenanceStep2, maintenanceStep3, maintenanceStep4])\nmaintenanceStep1 = StepsItem(\"Test hardware\", \"Check the webcam, microphone, speakers, keyboard, trackpad, ports, charging, Wi-Fi, Bluetooth, and external display support.\")\nmaintenanceStep2 = StepsItem(\"Set power and display preferences\", \"Adjust sleep behavior, battery-saving mode, brightness, scaling, and night-light settings.\")\nmaintenanceStep3 = StepsItem(\"Install security protection\", \"Confirm the built-in firewall and malware protection are enabled. Avoid installing multiple competing antivirus tools.\")\nmaintenanceStep4 = StepsItem(\"Create a maintenance routine\", \"Keep automatic updates enabled, review backups monthly, clear unneeded downloads, and restart occasionally.\")\nreminder = Callout(\"success\", \"Ready to go\", \"Once backups, encryption, updates, and account security are configured, your new laptop is ready for daily use.\")","promptTokens":4388,"completionTokens":959,"latencyMs":10754,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/realistic","repeat":0,"output":"root = Stack([onboardingCard])\nonboardingCard = Card([header, intro, checklistForm], \"column\", \"l\")\nheader = CardHeader(\"First-Day IT Onboarding\", \"Work through each item and tick it off when complete.\")\nintro = TextContent(\"Welcome! These eight steps will get your account, device, and collaboration tools ready for day one.\")\nchecklistForm = Form(\"first-day-it-onboarding\", finishButtons, [accountStep, passwordManagerStep, vpnStep, twoFactorStep, encryptionStep, updatesStep, channelsStep, supportStep])\naccountStep = FormControl(\"1. Activate your company account\", CheckBoxGroup(\"accountActivation\", [accountItem], null, $accountActivation), \"Use the activation link from IT and confirm you can sign in.\")\naccountItem = CheckBoxItem(\"My company account is activated and I can sign in.\", \"Confirm access to your email and employee portal.\", \"activated\")\npasswordManagerStep = FormControl(\"2. Set up the password manager\", CheckBoxGroup(\"passwordManager\", [passwordManagerItem], null, $passwordManager), \"Install the approved password manager and save your recovery information securely.\")\npasswordManagerItem = CheckBoxItem(\"I installed the password manager and created my vault.\", \"Use it to generate a unique password for each work service.\", \"passwordManagerReady\")\nvpnStep = FormControl(\"3. Install and test the VPN\", CheckBoxGroup(\"vpn\", [vpnItem], null, $vpn), \"Connect once from your home or another non-office network.\")\nvpnItem = CheckBoxItem(\"I installed the VPN and completed a successful test connection.\", \"You will need VPN access for internal systems when working remotely.\", \"vpnReady\")\ntwoFactorStep = FormControl(\"4. Enroll in two-factor authentication\", CheckBoxGroup(\"twoFactor\", [twoFactorItem], null, $twoFactor), \"Use the approved authenticator app; avoid SMS when an app option is available.\")\ntwoFactorItem = CheckBoxItem(\"I enrolled my authenticator app and saved backup codes.\", \"Test a sign-in prompt before moving on.\", \"twoFactorReady\")\nencryptionStep = FormControl(\"5. Verify laptop encryption\", CheckBoxGroup(\"encryption\", [encryptionItem], null, $encryption), \"Check that full-disk encryption is enabled and escrowed with IT.\")\nencryptionItem = CheckBoxItem(\"I verified that device encryption is on.\", \"Contact IT immediately if encryption is disabled or shows an error.\", \"encryptionReady\")\nupdatesStep = FormControl(\"6. Run system updates\", CheckBoxGroup(\"updates\", [updatesItem], null, $updates), \"Install pending operating system and security updates, then restart if requested.\")\nupdatesItem = CheckBoxItem(\"My laptop is updated and restarted if needed.\", \"Keeping updates current protects company data.\", \"updatesReady\")\nchannelsStep = FormControl(\"7. Join your team channels\", CheckBoxGroup(\"channels\", [channelsItem], null, $channels), \"Ask your manager which chat channels, project spaces, and mailing lists are required.\")\nchannelsItem = CheckBoxItem(\"I was added to my team, department, and project channels.\", \"Check that you can view announcements and participate in team conversations.\", \"channelsReady\")\nsupportStep = FormControl(\"8. Save the IT support contact\", CheckBoxGroup(\"support\", [supportItem], null, $support), \"Keep the help desk portal and emergency support number handy.\")\nsupportItem = CheckBoxItem(\"I bookmarked the IT help desk and know how to request support.\", \"Use the help desk for access requests, hardware issues, and software questions.\", \"supportReady\")\nfinishButtons = Buttons([completeButton])\ncompleteButton = Button(\"I've completed my IT onboarding\", Action([@ToAssistant(\"I've completed my first-day IT onboarding checklist. What should I do next?\")]), \"primary\")","promptTokens":4425,"completionTokens":919,"latencyMs":11155,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/minimal","repeat":3,"output":"root = Stack([header, checklist])\nheader = Card([title, intro], \"column\", \"s\")\ntitle = CardHeader(\"New Laptop Setup Checklist\", \"Work through these essentials before you start using your device.\")\nintro = TextContent(\"Check off each item as you go. Prioritize security, updates, backups, and the software you need.\", \"default\")\nchecklist = Accordion([hardwareSection, securitySection, accountsSection, softwareSection, backupSection, finishingSection])\nhardwareSection = AccordionItem(\"hardware\", \"1. Inspect and update the basics\", [hardwareForm])\nsecuritySection = AccordionItem(\"security\", \"2. Secure the laptop\", [securityForm])\naccountsSection = AccordionItem(\"accounts\", \"3. Connect accounts and services\", [accountsForm])\nsoftwareSection = AccordionItem(\"software\", \"4. Install essential software\", [softwareForm])\nbackupSection = AccordionItem(\"backup\", \"5. Set up backup and recovery\", [backupForm])\nfinishingSection = AccordionItem(\"finish\", \"6. Personalize and finish\", [finishingForm])\nhardwareForm = Form(\"hardwareChecklist\", hardwareButtons, [hardwareChecks])\nsecurityForm = Form(\"securityChecklist\", securityButtons, [securityChecks])\naccountsForm = Form(\"accountsChecklist\", accountsButtons, [accountsChecks])\nsoftwareForm = Form(\"softwareChecklist\", softwareButtons, [softwareChecks])\nbackupForm = Form(\"backupChecklist\", backupButtons, [backupChecks])\nfinishingForm = Form(\"finishingChecklist\", finishingButtons, [finishingChecks])\nhardwareButtons = Buttons([hardwareDone])\nsecurityButtons = Buttons([securityDone])\naccountsButtons = Buttons([accountsDone])\nsoftwareButtons = Buttons([softwareDone])\nbackupButtons = Buttons([backupDone])\nfinishingButtons = Buttons([finishingDone])\nhardwareDone = Button(\"Save progress\")\nsecurityDone = Button(\"Save progress\")\naccountsDone = Button(\"Save progress\")\nsoftwareDone = Button(\"Save progress\")\nbackupDone = Button(\"Save progress\")\nfinishingDone = Button(\"Save progress\")\nhardwareChecks = FormControl(\"Hardware and system\", CheckBoxGroup(\"hardware\", [hardwareItem1, hardwareItem2, hardwareItem3, hardwareItem4]))\nsecurityChecks = FormControl(\"Security essentials\", CheckBoxGroup(\"security\", [securityItem1, securityItem2, securityItem3, securityItem4]))\naccountsChecks = FormControl(\"Accounts and connectivity\", CheckBoxGroup(\"accounts\", [accountsItem1, accountsItem2, accountsItem3, accountsItem4]))\nsoftwareChecks = FormControl(\"Core applications\", CheckBoxGroup(\"software\", [softwareItem1, softwareItem2, softwareItem3, softwareItem4]))\nbackupChecks = FormControl(\"Backup and recovery\", CheckBoxGroup(\"backup\", [backupItem1, backupItem2, backupItem3, backupItem4]))\nfinishingChecks = FormControl(\"Final setup\", CheckBoxGroup(\"finishing\", [finishingItem1, finishingItem2, finishingItem3, finishingItem4]))\nhardwareItem1 = CheckBoxItem(\"Inspect the laptop, charger, ports, and accessories\", \"Confirm everything arrived undamaged.\", \"inspect\")\nhardwareItem2 = CheckBoxItem(\"Connect to Wi-Fi and plug in power\", \"Use a trusted network during setup.\", \"connect\")\nhardwareItem3 = CheckBoxItem(\"Install operating system and driver updates\", \"Restart when updates are complete.\", \"updates\")\nhardwareItem4 = CheckBoxItem(\"Set display, keyboard, trackpad, and power preferences\", \"Adjust settings for comfort and battery life.\", \"preferences\")\nsecurityItem1 = CheckBoxItem(\"Create a strong device password or PIN\", \"Avoid reusing passwords from other accounts.\", \"password\")\nsecurityItem2 = CheckBoxItem(\"Enable fingerprint or face sign-in\", \"Use biometric sign-in if your laptop supports it.\", \"biometric\")\nsecurityItem3 = CheckBoxItem(\"Turn on disk encryption\", \"Protect files if the laptop is lost or stolen.\", \"encryption\")\nsecurityItem4 = CheckBoxItem(\"Enable automatic updates and firewall protection\", \"Keep system security protections active.\", \"protection\")\naccountsItem1 = CheckBoxItem(\"Sign in with your primary account\", \"Use your Apple, Microsoft, Google, or Linux account as needed.\", \"primaryAccount\")\naccountsItem2 = CheckBoxItem(\"Set up email, calendar, and contacts\", \"Verify sync is working correctly.\", \"email\")\naccountsItem3 = CheckBoxItem(\"Connect cloud storage\", \"Examples include OneDrive, iCloud Drive, Google Drive, or Dropbox.\", \"cloud\")\naccountsItem4 = CheckBoxItem(\"Connect printer, Bluetooth, and external devices\", \"Test each device after pairing.\", \"devices\")\nsoftwareItem1 = CheckBoxItem(\"Install a web browser and password manager\", \"Import bookmarks only from trusted sources.\", \"browser\")\nsoftwareItem2 = CheckBoxItem(\"Install productivity apps\", \"Add office, notes, PDF, and communication tools.\", \"productivity\")\nsoftwareItem3 = CheckBoxItem(\"Install security and remote-access tools\", \"Use approved tools for your organization.\", \"securityTools\")\nsoftwareItem4 = CheckBoxItem(\"Install work or study-specific applications\", \"Add development, design, media, or research software.\", \"specialApps\")\nbackupItem1 = CheckBoxItem(\"Enable automatic cloud backup or file sync\", \"Choose folders that contain important documents.\", \"cloudBackup\")\nbackupItem2 = CheckBoxItem(\"Create a recovery drive or recovery media\", \"Store it somewhere safe and accessible.\", \"recoveryDrive\")\nbackupItem3 = CheckBoxItem(\"Turn on device location or find-my-device\", \"This can help locate a missing laptop.\", \"findDevice\")\nbackupItem4 = CheckBoxItem(\"Test restoring one file\", \"Confirm that your backup method actually works.\", \"testRestore\")\nfinishingItem1 = CheckBoxItem(\"Set up a clean folder structure\", \"Create folders for documents, projects, photos, and downloads.\", \"folders\")\nfinishingItem2 = CheckBoxItem(\"Remove unwanted trial software and startup apps\", \"Keep the system fast and uncluttered.\", \"cleanup\")\nfinishingItem3 = CheckBoxItem(\"Review privacy and notification settings\", \"Limit unnecessary tracking and interruptions.\", \"privacy\")\nfinishingItem4 = CheckBoxItem(\"Record the serial number and keep proof of purchase\", \"Useful for support, warranty, and insurance claims.\", \"records\")","promptTokens":4388,"completionTokens":1340,"latencyMs":13740,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/realistic","repeat":3,"output":"root = Stack([onboardingCard])\nonboardingCard = Card([header, intro, checklist, help])\nheader = CardHeader(\"First-Day IT Onboarding\", \"Complete each item to get securely set up for work.\")\nintro = TextContent(\"Tick off the checklist in order. Most steps take only a few minutes.\")\nchecklist = SwitchGroup(\"firstDayChecklist\", [accountStep, passwordManagerStep, twoFactorStep, vpnStep, encryptionStep, updatesStep, channelsStep, confirmationStep], \"card\")\naccountStep = SwitchItem(\"1. Activate your work account\", \"Use the activation link from IT and confirm you can sign in.\", \"accountActivated\")\npasswordManagerStep = SwitchItem(\"2. Set up the password manager\", \"Install the approved password manager and save your recovery information.\", \"passwordManagerReady\")\ntwoFactorStep = SwitchItem(\"3. Enroll in 2FA\", \"Add your authenticator app or security key to protect your account.\", \"twoFactorEnabled\")\nvpnStep = SwitchItem(\"4. Configure the VPN\", \"Install the VPN client and test connecting from your laptop.\", \"vpnConfigured\")\nencryptionStep = SwitchItem(\"5. Verify laptop encryption\", \"Confirm device encryption is enabled before storing work files locally.\", \"encryptionVerified\")\nupdatesStep = SwitchItem(\"6. Install required updates\", \"Run operating system updates and restart if prompted.\", \"updatesInstalled\")\nchannelsStep = SwitchItem(\"7. Join your team channels\", \"Ask your manager to add you to your team chat, project spaces, and company announcements.\", \"teamChannelsJoined\")\nconfirmationStep = SwitchItem(\"8. Confirm access with IT\", \"Send IT a note once finished and report anything that did not work.\", \"itConfirmed\")\nhelp = TextCallout(\"info\", \"Need help?\", \"Contact the IT help desk if an activation link, app download, or access request is missing.\")","promptTokens":4425,"completionTokens":507,"latencyMs":6763,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/realistic","repeat":2,"output":"root = Stack([onboardingCard])\nonboardingCard = Card([header, welcome, checklistForm], \"column\", \"l\")\nheader = CardHeader(\"First-Day IT Onboarding\", \"Complete each item below to get your workstation ready.\")\nwelcome = Callout(\"info\", \"Welcome aboard!\", \"Work through this checklist in order. If anything is missing or blocked, contact the IT help desk.\")\nchecklistForm = Form(\"it-onboarding\", completeButtons, [checklistField])\nchecklistField = FormControl(\"Your first-day checklist\", checklist, \"Tick each item as you finish it.\")\nchecklist = CheckBoxGroup(\"onboarding-tasks\", checklistItems)\nchecklistItems = [activateAccount, setPasswordManager, enrollTwoFactor, configureVpn, verifyEncryption, installUpdates, joinTeamChannels, confirmHelp]\nactivateAccount = CheckBoxItem(\"Activate your company account\", \"Open your welcome email and activate your company identity account.\", \"account-activation\")\nsetPasswordManager = CheckBoxItem(\"Set up the password manager\", \"Install the approved password manager, sign in, and save your recovery information securely.\", \"password-manager\")\nenrollTwoFactor = CheckBoxItem(\"Enroll in 2FA\", \"Add an authenticator app or security key to protect your company account.\", \"two-factor\")\nconfigureVpn = CheckBoxItem(\"Connect to the VPN\", \"Install the VPN client, sign in, and verify that you can connect from your laptop.\", \"vpn\")\nverifyEncryption = CheckBoxItem(\"Check laptop encryption\", \"Confirm that full-disk encryption is enabled in your device security settings.\", \"laptop-encryption\")\ninstallUpdates = CheckBoxItem(\"Run required software updates\", \"Install pending operating system and security updates, then restart if prompted.\", \"software-updates\")\njoinTeamChannels = CheckBoxItem(\"Join your team channels\", \"Accept invitations to your chat workspace, project channels, and company announcements.\", \"team-channels\")\nconfirmHelp = CheckBoxItem(\"Save IT support details\", \"Bookmark the IT help portal and note how to request help if you get stuck.\", \"it-support\")\ncompleteButtons = Buttons([completeButton])\ncompleteButton = Button(\"I completed my IT onboarding\", Action([@ToAssistant(\"I completed my first-day IT onboarding checklist.\")]), \"primary\")","promptTokens":4425,"completionTokens":705,"latencyMs":9724,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/realistic","repeat":4,"output":"root = Stack([onboardingCard])\nonboardingCard = Card([header, intro, checklist, helpCallout], \"column\", \"l\")\nheader = CardHeader(\"First-Day IT Onboarding\", \"Work through each item and tick it off as you go.\")\nintro = TextContent(\"Welcome! These eight setup tasks will get your accounts, laptop, and team access ready for a productive first day.\")\nchecklist = Stack([accountStep, passwordStep, vpnStep, twoFactorStep, encryptionStep, updatesStep, channelsStep, supportStep], \"column\", \"s\")\naccountStep = SwitchGroup(\"accountActivation\", [accountItem], \"card\")\npasswordStep = SwitchGroup(\"passwordManager\", [passwordItem], \"card\")\nvpnStep = SwitchGroup(\"vpnSetup\", [vpnItem], \"card\")\ntwoFactorStep = SwitchGroup(\"twoFactorAuthentication\", [twoFactorItem], \"card\")\nencryptionStep = SwitchGroup(\"encryptionCheck\", [encryptionItem], \"card\")\nupdatesStep = SwitchGroup(\"laptopUpdates\", [updatesItem], \"card\")\nchannelsStep = SwitchGroup(\"teamChannels\", [channelsItem], \"card\")\nsupportStep = SwitchGroup(\"itSupport\", [supportItem], \"card\")\naccountItem = SwitchItem(\"Activate your company account\", \"Use the welcome email invitation to sign in and confirm your profile details.\", \"activateAccount\")\npasswordItem = SwitchItem(\"Set up the password manager\", \"Install the approved password manager, create your vault, and save your recovery information securely.\", \"setupPasswordManager\")\nvpnItem = SwitchItem(\"Configure the VPN\", \"Install the company VPN client and verify you can connect to internal resources.\", \"configureVpn\")\ntwoFactorItem = SwitchItem(\"Enable 2FA\", \"Register your authenticator app or security key, then test a sign-in using two-factor authentication.\", \"enable2fa\")\nencryptionItem = SwitchItem(\"Confirm laptop encryption\", \"Check that full-disk encryption is enabled and that your recovery key has been backed up according to company policy.\", \"checkEncryption\")\nupdatesItem = SwitchItem(\"Run system updates\", \"Install pending operating system, security, browser, and endpoint protection updates.\", \"runUpdates\")\nchannelsItem = SwitchItem(\"Join the right team channels\", \"Ask your manager for access to your department chat channels, project spaces, shared drives, and calendar groups.\", \"joinChannels\")\nsupportItem = SwitchItem(\"Save IT support details\", \"Bookmark the IT help portal and note the support contact method for urgent issues.\", \"saveSupport\")\nhelpCallout = Callout(\"info\", \"Need help?\", \"If a setup step is blocked, contact IT support rather than skipping it. They can verify access, reset invitations, or help with your device.\")","promptTokens":4425,"completionTokens":719,"latencyMs":9134,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/realistic","repeat":1,"output":"root = Stack([header, onboardingCard], \"column\", \"l\")\nheader = Stack([title, intro], \"column\", \"s\")\ntitle = TextContent(\"First-Day IT Onboarding\", \"large-heavy\")\nintro = TextContent(\"Work through each item below and tick it off as you go. Reach out to IT support if anything does not work.\")\nonboardingCard = Card([cardHeader, onboardingForm], \"column\", \"m\")\ncardHeader = CardHeader(\"Your setup checklist\", \"Complete these eight essentials before starting work.\")\nonboardingForm = Form(\"first-day-it-onboarding\", completeButtons, [checklistField])\nchecklistField = FormControl(\"Setup tasks\", checklist, \"Tick each task once it is complete.\")\nchecklist = CheckBoxGroup(\"it-onboarding-checklist\", [activateAccount, setPasswordManager, enrollTwoFactor, configureVpn, verifyEncryption, installUpdates, joinTeamChannels, confirmSupport])\nactivateAccount = CheckBoxItem(\"Activate your company account\", \"Use the welcome email to sign in and confirm your profile details.\", \"account-activation\")\nsetPasswordManager = CheckBoxItem(\"Set up the password manager\", \"Install the approved password manager and save your recovery information securely.\", \"password-manager\")\nenrollTwoFactor = CheckBoxItem(\"Enroll in two-factor authentication\", \"Add an authenticator app or security key to protect your account.\", \"two-factor-auth\")\nconfigureVpn = CheckBoxItem(\"Configure the VPN\", \"Install the VPN client and confirm that you can connect successfully.\", \"vpn-setup\")\nverifyEncryption = CheckBoxItem(\"Check laptop encryption\", \"Confirm that full-disk encryption is enabled in your device settings.\", \"laptop-encryption\")\ninstallUpdates = CheckBoxItem(\"Run required software updates\", \"Install pending operating system, security, and company software updates.\", \"software-updates\")\njoinTeamChannels = CheckBoxItem(\"Join your team channels\", \"Ask your manager which chat, project, and announcement channels you need.\", \"team-channels\")\nconfirmSupport = CheckBoxItem(\"Save IT support contact details\", \"Bookmark the support portal and note the help desk contact method.\", \"it-support\")\ncompleteButtons = Buttons([completeButton])\ncompleteButton = Button(\"Mark onboarding complete\", Action([@ToAssistant(\"I completed my first-day IT onboarding checklist.\")]), \"primary\")","promptTokens":4425,"completionTokens":608,"latencyMs":22275,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/adversarial","repeat":0,"output":"root = Stack([page])\npage = Stack([title, scopeNote, deploySteps], \"column\", \"l\")\ntitle = CardHeader(\"Production deployment runbook\", \"A safe Docker Compose deployment workflow with commands, reasons, and skip-risk for every step.\")\nscopeNote = Callout(\"warning\", \"Before you begin\", \"These commands deploy a containerized application to a Linux server using Docker Compose. Replace example.com, deploy, and registry.example.com/my-app with your real domain, Linux user, and container image.\")\ndeploySteps = Accordion([stepOne, stepTwo, stepThree, stepFour, stepFive, stepSix, stepSeven])\nstepOne = AccordionItem(\"prepare-server\", \"1. Prepare and secure the server\", [prepareWhy, prepareSkip, prepareCommands])\nprepareWhy = TextContent(\"**Why it matters:** A deployment needs a stable host with current security patches, a non-root deployment account, and the firewall opened only for required traffic. This reduces avoidable outages and exposure before application code is running.\")\nprepareSkip = TextCallout(\"danger\", \"If you skip it\", \"You may deploy onto an unpatched server, expose unnecessary ports, or accidentally run the application as root. That makes compromises and recovery significantly harder.\")\nprepareCommands = CodeBlock(\"bash\", \"ssh root@YOUR_SERVER_IP\\napt update && apt upgrade -y\\nadduser deploy\\nusermod -aG sudo deploy\\nufw allow OpenSSH\\nufw allow 80/tcp\\nufw allow 443/tcp\\nufw --force enable\\nexit\\nssh deploy@YOUR_SERVER_IP\")\nstepTwo = AccordionItem(\"install-runtime\", \"2. Install Docker and Docker Compose\", [runtimeWhy, runtimeSkip, runtimeCommands])\nruntimeWhy = TextContent(\"**Why it matters:** Docker makes the runtime reproducible: the same image tested in CI is the one started on the server. Docker Compose keeps the application, database, cache, and network configuration versioned together.\")\nruntimeSkip = TextCallout(\"danger\", \"If you skip it\", \"Manual package installation can produce dependency drift between local, staging, and production environments. Deployments become harder to repeat and roll back.\")\nruntimeCommands = CodeBlock(\"bash\", \"sudo apt update\\nsudo apt install -y ca-certificates curl gnupg\\nsudo install -m 0755 -d /etc/apt/keyrings\\ncurl -fsSL https://download.docker.com/linux/ubuntu/gpg | sudo gpg --dearmor -o /etc/apt/keyrings/docker.gpg\\nsudo chmod a+r /etc/apt/keyrings/docker.gpg\\necho \\\"deb [arch=$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/docker.gpg] https://download.docker.com/linux/ubuntu $(. /etc/os-release && echo $VERSION_CODENAME) stable\\\" | sudo tee /etc/apt/sources.list.d/docker.list > /dev/null\\nsudo apt update\\nsudo apt install -y docker-ce docker-ce-cli containerd.io docker-buildx-plugin docker-compose-plugin\\nsudo usermod -aG docker $USER\\nnewgrp docker\\ndocker --version\\ndocker compose version\")\nstepThree = AccordionItem(\"create-config\", \"3. Create the deployment directory and production configuration\", [configWhy, configSkip, configCommands])\nconfigWhy = TextContent(\"**Why it matters:** Production configuration separates secrets and environment-specific settings from application code. Keeping this directory organized also makes backups, audits, and incident response faster.\")\nconfigSkip = TextCallout(\"danger\", \"If you skip it\", \"Secrets can end up committed to source control, environment values may be missing, and the application may start with development defaults or fail at runtime.\")\nconfigCommands = CodeBlock(\"bash\", \"mkdir -p ~/apps/my-app\\ncd ~/apps/my-app\\ncat > .env <<'EOF'\\nAPP_ENV=production\\nAPP_URL=https://example.com\\nDATABASE_URL=postgresql://myapp:CHANGE_THIS_DATABASE_PASSWORD@db:5432/myapp\\nAPP_SECRET=CHANGE_THIS_TO_A_LONG_RANDOM_SECRET\\nEOF\\nchmod 600 .env\\nopenssl rand -hex 32\")\nstepFour = AccordionItem(\"compose-file\", \"4. Create and validate the Compose definition\", [composeWhy, composeSkip, composeCommands])\ncomposeWhy = TextContent(\"**Why it matters:** The Compose file declares exactly which image runs, which ports are exposed, how persistent volumes are mounted, and which services must start together. Validation catches syntax and variable errors before the live deployment.\")\ncomposeSkip = TextCallout(\"danger\", \"If you skip it\", \"A malformed service definition can cause downtime, data loss from missing volumes, or accidental exposure of internal services such as PostgreSQL.\")\ncomposeCommands = CodeBlock(\"bash\", \"cd ~/apps/my-app\\ncat > compose.yml <<'EOF'\\nservices:\\n app:\\n image: registry.example.com/my-app:1.0.0\\n env_file: .env\\n restart: unless-stopped\\n ports:\\n - \\\"127.0.0.1:3000:3000\\\"\\n depends_on:\\n db:\\n condition: service_healthy\\n db:\\n image: postgres:16-alpine\\n environment:\\n POSTGRES_DB: myapp\\n POSTGRES_USER: myapp\\n POSTGRES_PASSWORD: CHANGE_THIS_DATABASE_PASSWORD\\n volumes:\\n - postgres_data:/var/lib/postgresql/data\\n restart: unless-stopped\\n healthcheck:\\n test: [\\\"CMD-SHELL\\\", \\\"pg_isready -U myapp -d myapp\\\"]\\n interval: 10s\\n timeout: 5s\\n retries: 5\\nvolumes:\\n postgres_data:\\nEOF\\ndocker compose --env-file .env config\")\nstepFive = AccordionItem(\"pull-and-backup\", \"5. Back up data and pull the release image\", [pullWhy, pullSkip, pullCommands])\npullWhy = TextContent(\"**Why it matters:** A verified backup provides a recovery point before schema migrations or application changes. Pulling the immutable release image before restarting reduces the maintenance window and confirms registry access.\")\npullSkip = TextCallout(\"danger\", \"If you skip it\", \"A failed migration or incompatible release may leave the database unusable with no current recovery point. Image-pull failures can also occur after services have already been stopped.\")\npullCommands = CodeBlock(\"bash\", \"cd ~/apps/my-app\\nmkdir -p backups\\ndocker compose exec -T db pg_dump -U myapp myapp | gzip > backups/myapp-$(date +%F-%H%M%S).sql.gz\\ndocker compose pull\\ndocker image inspect registry.example.com/my-app:1.0.0 --format '{{.Id}}'\")\nstepSix = AccordionItem(\"deploy-release\", \"6. Start the release and run migrations\", [releaseWhy, releaseSkip, releaseCommands])\nreleaseWhy = TextContent(\"**Why it matters:** Starting services in detached mode applies the declared production state. Database migrations must run as part of the release so the application code and database schema remain compatible.\")\nreleaseSkip = TextCallout(\"danger\", \"If you skip it\", \"The new containers will not be running, or the application may fail with missing tables, columns, indexes, or incompatible schema assumptions.\")\nreleaseCommands = CodeBlock(\"bash\", \"cd ~/apps/my-app\\ndocker compose up -d\\ndocker compose ps\\ndocker compose logs --tail=100 app\\n# Run the migration command supported by your application:\\ndocker compose exec app ./bin/migrate\\n# Examples for common frameworks:\\n# docker compose exec app npx prisma migrate deploy\\n# docker compose exec app python manage.py migrate\\n# docker compose exec app bundle exec rails db:migrate\")\nstepSeven = AccordionItem(\"verify-and-monitor\", \"7. Verify health, enable HTTPS, and monitor the deployment\", [verifyWhy, verifySkip, verifyCommands])\nverifyWhy = TextContent(\"**Why it matters:** A successful container start does not prove users can reach the service or that critical flows work. Health checks, HTTPS verification, and log monitoring confirm that the release is functioning from both server and user perspectives.\")\nverifySkip = TextCallout(\"danger\", \"If you skip it\", \"You can declare a deployment complete while users receive errors, TLS is invalid, migrations are failing, or a container is repeatedly restarting.\")\nverifyCommands = CodeBlock(\"bash\", \"cd ~/apps/my-app\\ndocker compose ps\\ncurl -i http://127.0.0.1:3000/health\\ncurl -I https://example.com\\ndocker compose logs -f --tail=100 app\\n# Confirm automatic restart behavior:\\ndocker inspect --format='{{.HostConfig.RestartPolicy.Name}}' $(docker compose ps -q app)\\n# Roll back to the prior image tag if verification fails:\\nsed -i 's|registry.example.com/my-app:1.0.0|registry.example.com/my-app:0.9.0|' compose.yml\\ndocker compose up -d\\ndocker compose logs --tail=100 app\")","promptTokens":4407,"completionTokens":2081,"latencyMs":20512,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/adversarial","repeat":1,"output":"root = Stack([intro, scope, warning, deployment], \"column\", \"l\")\nintro = Card([CardHeader(\"Production deployment runbook\", \"Dockerized web application on an Ubuntu server\"), TextContent(\"This guide deploys a containerized application behind Caddy with automatic HTTPS. Replace every value in ALL_CAPS before running commands.\", \"default\")])\nscope = Callout(\"info\", \"Assumptions\", \"You have an Ubuntu 22.04+ server, a domain name pointed at the server IP, SSH access, and a project that can be built with Docker.\")\nwarning = Callout(\"warning\", \"Before you begin\", \"Deployment commands can interrupt an existing service. If this is an existing production server, take a backup and confirm the correct host before making changes.\")\ndeployment = Accordion([step1, step2, step3, step4, step5, step6, step7, step8, step9])\nstep1 = AccordionItem(\"step-1\", \"1. Connect to the server and verify its identity\", [step1Why, step1Skip, step1Commands])\nstep1Why = TextContent(\"Why it matters: This confirms you are deploying to the intended machine and gives you a baseline for disk, memory, OS version, and network identity. Production incidents often begin with commands run against the wrong server.\", \"default\")\nstep1Skip = TextCallout(\"warning\", \"If you skip it\", \"You may overwrite the wrong environment, discover too late that storage is full, or deploy to a host that cannot serve your domain.\")\nstep1Commands = CodeBlock(\"bash\", \"ssh DEPLOY_USER@SERVER_IP\\nhostnamectl\\nuname -a\\ndf -h\\nfree -h\\ncurl -4 ifconfig.me; echo\")\nstep2 = AccordionItem(\"step-2\", \"2. Update the operating system and install required packages\", [step2Why, step2Skip, step2Commands])\nstep2Why = TextContent(\"Why it matters: Security updates reduce exposure to known vulnerabilities. Docker provides reproducible application runtime environments, Git retrieves source code, and Caddy will terminate HTTPS traffic.\", \"default\")\nstep2Skip = TextCallout(\"warning\", \"If you skip it\", \"Your server may retain known security vulnerabilities, lack required deployment tooling, or behave differently from the build environment.\")\nstep2Commands = CodeBlock(\"bash\", \"sudo apt update\\nsudo apt upgrade -y\\nsudo apt install -y ca-certificates curl git ufw\\ncurl -fsSL https://get.docker.com -o get-docker.sh\\nsudo sh get-docker.sh\\nsudo usermod -aG docker $USER\\nnewgrp docker\\ndocker --version\\ndocker compose version\")\nstep3 = AccordionItem(\"step-3\", \"3. Configure the firewall and SSH access\", [step3Why, step3Skip, step3Commands])\nstep3Why = TextContent(\"Why it matters: A firewall limits exposed services to SSH, HTTP, and HTTPS. This is a basic production boundary: application ports should not be publicly reachable when a reverse proxy is handling traffic.\", \"default\")\nstep3Skip = TextCallout(\"warning\", \"If you skip it\", \"Unnecessary ports may remain exposed to the internet. If you enable the firewall without allowing SSH first, you can lock yourself out of the server.\")\nstep3Commands = CodeBlock(\"bash\", \"sudo ufw allow OpenSSH\\nsudo ufw allow 80/tcp\\nsudo ufw allow 443/tcp\\nsudo ufw enable\\nsudo ufw status verbose\")\nstep4 = AccordionItem(\"step-4\", \"4. Create a dedicated application directory and retrieve the source\", [step4Why, step4Skip, step4Commands])\nstep4Why = TextContent(\"Why it matters: A predictable directory keeps source, environment configuration, and deployment files organized. Cloning a specific repository branch makes the release source traceable and repeatable.\", \"default\")\nstep4Skip = TextCallout(\"warning\", \"If you skip it\", \"Files can become scattered across the server, deployments become difficult to audit, and future operators may not know what version is running.\")\nstep4Commands = CodeBlock(\"bash\", \"sudo mkdir -p /opt/APP_NAME\\nsudo chown -R $USER:$USER /opt/APP_NAME\\ncd /opt/APP_NAME\\ngit clone --branch main https://github.com/ORG/REPOSITORY.git .\\ngit status\\ngit log -1 --oneline\")\nstep5 = AccordionItem(\"step-5\", \"5. Create and protect production environment variables\", [step5Why, step5Skip, step5Commands])\nstep5Why = TextContent(\"Why it matters: Credentials, API keys, database URLs, and application secrets must be supplied separately from source code. A strong application secret protects signed sessions, tokens, and encrypted values where applicable.\", \"default\")\nstep5Skip = TextCallout(\"warning\", \"If you skip it\", \"The app may fail at startup, use unsafe defaults, expose credentials in Git, or invalidate authentication and encryption guarantees.\")\nstep5Commands = CodeBlock(\"bash\", \"cd /opt/APP_NAME\\ncat > .env <<'EOF'\\nNODE_ENV=production\\nAPP_URL=https://APP_DOMAIN\\nDATABASE_URL=REPLACE_WITH_PRODUCTION_DATABASE_URL\\nAPP_SECRET=REPLACE_WITH_A_LONG_RANDOM_SECRET\\nEOF\\nopenssl rand -base64 48\\nchmod 600 .env\\nls -l .env\")\nstep6 = AccordionItem(\"step-6\", \"6. Create the production Docker Compose and Caddy configuration\", [step6Why, step6Skip, step6Commands])\nstep6Why = TextContent(\"Why it matters: Docker Compose defines the exact production process, restart behavior, and internal port mapping. Caddy accepts public traffic on ports 80 and 443, obtains TLS certificates automatically, and proxies requests to the application without exposing its internal port.\", \"default\")\nstep6Skip = TextCallout(\"warning\", \"If you skip it\", \"Your application may not restart after a reboot, may be publicly exposed on an internal port, or may serve traffic without HTTPS.\")\nstep6Commands = CodeBlock(\"yaml\", \"cd /opt/APP_NAME\\ncat > compose.prod.yml <<'EOF'\\nservices:\\n app:\\n build: .\\n env_file:\\n - .env\\n restart: unless-stopped\\n expose:\\n - \\\"3000\\\"\\n\\n caddy:\\n image: caddy:2-alpine\\n restart: unless-stopped\\n ports:\\n - \\\"80:80\\\"\\n - \\\"443:443\\\"\\n volumes:\\n - ./Caddyfile:/etc/caddy/Caddyfile:ro\\n - caddy_data:/data\\n - caddy_config:/config\\n depends_on:\\n - app\\n\\nvolumes:\\n caddy_data:\\n caddy_config:\\nEOF\\ncat > Caddyfile <<'EOF'\\nAPP_DOMAIN {\\n reverse_proxy app:3000\\n}\\nEOF\\nsed -i 's/APP_DOMAIN/your-domain.example/g' Caddyfile\")\nstep7 = AccordionItem(\"step-7\", \"7. Build, validate, and start the release\", [step7Why, step7Skip, step7Commands])\nstep7Why = TextContent(\"Why it matters: Building on the server creates the deployable artifact from the checked-out commit. Compose validation catches malformed configuration before startup. Detached startup allows services to continue running after your SSH session closes.\", \"default\")\nstep7Skip = TextCallout(\"warning\", \"If you skip it\", \"You may start an outdated image, miss a configuration error, or leave the application dependent on an interactive SSH session.\")\nstep7Commands = CodeBlock(\"bash\", \"cd /opt/APP_NAME\\ndocker compose --env-file .env -f compose.prod.yml config\\ndocker compose --env-file .env -f compose.prod.yml build --pull\\ndocker compose --env-file .env -f compose.prod.yml up -d\\ndocker compose -f compose.prod.yml ps\\ndocker compose -f compose.prod.yml logs --tail=100 app\")\nstep8 = AccordionItem(\"step-8\", \"8. Verify health, HTTPS, and application behavior\", [step8Why, step8Skip, step8Commands])\nstep8Why = TextContent(\"Why it matters: A successful container start does not prove that the application is usable. These checks verify process status, proxy routing, certificate issuance, HTTP response behavior, and recent error logs.\", \"default\")\nstep8Skip = TextCallout(\"warning\", \"If you skip it\", \"A broken release can remain undetected until users report it. Common failures include missing environment variables, failed migrations, DNS errors, and TLS certificate issues.\")\nstep8Commands = CodeBlock(\"bash\", \"cd /opt/APP_NAME\\ndocker compose -f compose.prod.yml ps\\ndocker compose -f compose.prod.yml logs --tail=100 caddy\\ncurl -I http://your-domain.example\\ncurl -I https://your-domain.example\\ncurl -sS https://your-domain.example | head -20\\nsudo ss -tulpn | grep -E ':80|:443'\")\nstep9 = AccordionItem(\"step-9\", \"9. Document rollback and deploy future releases safely\", [step9Why, step9Skip, step9Commands])\nstep9Why = TextContent(\"Why it matters: Every production deployment needs a quick, tested reversal path. Recording the previous Git revision lets you return to known-good code if the new release fails health checks or causes unexpected behavior.\", \"default\")\nstep9Skip = TextCallout(\"warning\", \"If you skip it\", \"Recovery during an incident becomes slower and riskier. Operators may guess at previous versions or make additional untested changes under pressure.\")\nstep9Commands = CodeBlock(\"bash\", \"# Deploy a later release\\ncd /opt/APP_NAME\\ngit rev-parse HEAD > /tmp/APP_NAME-previous-release.txt\\ngit fetch origin\\ngit checkout main\\ngit pull --ff-only origin main\\ndocker compose --env-file .env -f compose.prod.yml build --pull\\ndocker compose --env-file .env -f compose.prod.yml up -d\\ndocker compose -f compose.prod.yml logs --tail=100 app\\n\\n# Roll back if needed\\ncd /opt/APP_NAME\\ngit checkout \\\"$(cat /tmp/APP_NAME-previous-release.txt)\\\"\\ndocker compose --env-file .env -f compose.prod.yml build\\ndocker compose --env-file .env -f compose.prod.yml up -d\\ncurl -I https://your-domain.example\")","promptTokens":4407,"completionTokens":2412,"latencyMs":23103,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/minimal","repeat":1,"output":"root = Stack([\n Card([\n CardHeader(\"Contact Us\", \"We'd love to hear from you. Send us a message and we'll get back to you as soon as possible.\"),\n Separator(\"horizontal\"),\n ContactForm\n ]),\n SuccessCallout\n], \"column\", \"l\")\n\nSuccessCallout = Callout(\"success\", \"Message Sent\", \"Thank you! Your message has been successfully delivered.\", $showSuccess)\n\nContactForm = Form(\"contact_form\", SubmitButtons, [\n NameField,\n EmailField,\n MessageField\n])\n\nNameField = FormControl(\"Full Name\", Input(\"name\", \"John Doe\", \"text\", {required: true}, $name), \"Please enter your name\")\nEmailField = FormControl(\"Email Address\", Input(\"email\", \"john@example.com\", \"email\", {required: true, email: true}, $email), \"We'll use this to reply\")\nMessageField = FormControl(\"Message\", TextArea(\"message\", \"How can we help?\", 5, {required: true, minLength: 10}, $message), \"Minimum 10 characters\")\n\nSubmitButtons = Buttons([\n SubmitBtn,\n ClearBtn\n], \"row\")\n\nSubmitBtn = Button(\"Send Message\", Action([@Set($showSuccess, true), @Set($name, \"\"), @Set($email, \"\"), @Set($message, \"\")]), \"primary\")\nClearBtn = Button(\"Clear\", Action([@Set($name, \"\"), @Set($email, \"\"), @Set($message, \"\")]), \"secondary\")","promptTokens":4542,"completionTokens":343,"latencyMs":4782,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/adversarial","repeat":2,"output":"root = Stack([header, scopeNote, deploymentSteps], \"column\", \"l\")\nheader = Card([title, intro], \"clear\", \"column\", \"s\")\ntitle = TextContent(\"Generic production deployment checklist\", \"large-heavy\")\nintro = TextContent(\"These commands assume a Node.js web application deployed with Docker to a Linux server. Replace values in ALL_CAPS with your real repository, image registry, server, and domain.\")\nscopeNote = Callout(\"warning\", \"Set your deployment variables first\", \"Do not run production commands until you have verified the target environment, backup policy, secret source, and rollback image. A mistaken deployment target can overwrite live data or cause avoidable downtime.\")\ndeploymentSteps = Steps([step1, step2, step3, step4, step5, step6, step7, step8, step9])\nstep1 = StepsItem(\"1. Confirm the target and create a deployment record\", \"Why it matters: Deployment mistakes often come from using the wrong cluster, server, branch, or environment. Recording the intended version gives your team an audit trail and a reliable rollback reference.\\n\\nExact commands:\\n```bash\\nexport APP_NAME=\\\"my-app\\\"\\nexport ENVIRONMENT=\\\"production\\\"\\nexport VERSION=\\\"$(git rev-parse --short HEAD)\\\"\\nexport DEPLOYED_AT=\\\"$(date -u +%Y-%m-%dT%H:%M:%SZ)\\\"\\nprintf '%s %s %s %s\\\\n' \\\"$APP_NAME\\\" \\\"$ENVIRONMENT\\\" \\\"$VERSION\\\" \\\"$DEPLOYED_AT\\\"\\ngit status --short\\ngit branch --show-current\\ngit log -1 --oneline\\n```\\n\\nExpected result: the working tree is clean, the branch is approved for release, and the displayed commit is the one you intend to deploy.\\n\\nIf you skip it: you can deploy uncommitted code, deploy a feature branch accidentally, or be unable to identify the exact version that introduced an incident.\")\nstep2 = StepsItem(\"2. Install locked dependencies and run automated checks\", \"Why it matters: A reproducible dependency install ensures that local, CI, and production builds use the versions recorded in the lockfile. Tests, linting, and type checks catch failures before users encounter them.\\n\\nExact commands:\\n```bash\\nnpm ci\\nnpm run lint\\nnpm run typecheck\\nnpm test\\n```\\n\\nFor projects without one of these scripts, inspect the available scripts first:\\n```bash\\nnpm run\\n```\\n\\nExpected result: every command exits with status 0.\\n\\nIf you skip it: a missing package, broken import, type error, security regression, or failing business rule may reach production. Using `npm install` instead of `npm ci` can also change dependency versions unexpectedly.\")\nstep3 = StepsItem(\"3. Build the release artifact and verify it locally\", \"Why it matters: The production build may behave differently from development mode because assets are bundled, environment variables are read differently, and server-side code is optimized. Build once and promote that exact artifact.\\n\\nExact commands:\\n```bash\\nnpm run build\\nnpm run start\\n```\\n\\nIn a second terminal, verify the local health endpoint and main page:\\n```bash\\ncurl -i http://localhost:3000/health\\ncurl -I http://localhost:3000/\\n```\\n\\nStop the local server after verification with `Ctrl+C`.\\n\\nIf you skip it: the first production request may reveal a build-time error, a missing environment variable, broken static assets, or an unavailable health endpoint.\")\nstep4 = StepsItem(\"4. Prepare and validate production configuration\", \"Why it matters: Secrets and production settings should be present before the application starts. Validation prevents booting a release that cannot connect to its database, payment provider, mail service, or identity provider.\\n\\nExact commands:\\n```bash\\nexport REGISTRY=\\\"registry.example.com\\\"\\nexport IMAGE=\\\"$REGISTRY/$APP_NAME:$VERSION\\\"\\nexport PROD_HOST=\\\"deploy@production.example.com\\\"\\nexport APP_DIR=\\\"/srv/$APP_NAME\\\"\\n\\nssh \\\"$PROD_HOST\\\" \\\"test -f $APP_DIR/.env && echo 'Production environment file exists'\\\"\\nssh \\\"$PROD_HOST\\\" \\\"grep -q '^DATABASE_URL=' $APP_DIR/.env && echo 'DATABASE_URL is configured'\\\"\\nssh \\\"$PROD_HOST\\\" \\\"grep -q '^APP_SECRET=' $APP_DIR/.env && echo 'APP_SECRET is configured'\\\"\\n```\\n\\nDo not print secret values in terminals, CI logs, screenshots, or chat.\\n\\nIf you skip it: the deployment may start with empty or incorrect settings, fail to connect to dependencies, expose debug behavior, or generate incompatible sessions and encryption keys.\")\nstep5 = StepsItem(\"5. Back up stateful data and verify the backup\", \"Why it matters: Application deployments sometimes include migrations or code paths that alter data. A tested backup is the safest recovery point if a migration, release, or operator action causes data loss.\\n\\nExact commands for PostgreSQL:\\n```bash\\nexport DATABASE_URL=\\\"postgresql://USER:PASSWORD@HOST:5432/DATABASE\\\"\\nexport BACKUP_FILE=\\\"$APP_NAME-$ENVIRONMENT-$VERSION.sql.gz\\\"\\npg_dump \\\"$DATABASE_URL\\\" | gzip > \\\"$BACKUP_FILE\\\"\\ngzip -t \\\"$BACKUP_FILE\\\"\\nls -lh \\\"$BACKUP_FILE\\\"\\n```\\n\\nCopy the backup to protected storage:\\n```bash\\naws s3 cp \\\"$BACKUP_FILE\\\" \\\"s3://YOUR_BACKUP_BUCKET/$APP_NAME/\\\"\\n```\\n\\nIf you skip it: a destructive migration or unexpected application bug can leave you with no practical way to restore customer data. A backup that was never validated may also be unusable when needed.\")\nstep6 = StepsItem(\"6. Build, tag, and publish an immutable container image\", \"Why it matters: An image tag tied to a commit makes deployments reproducible. The same tested image can be promoted, inspected, and rolled back without rebuilding from potentially changed source or dependencies.\\n\\nExact commands:\\n```bash\\ndocker build --pull -t \\\"$IMAGE\\\" .\\ndocker image inspect \\\"$IMAGE\\\" --format '{{.Id}}'\\ndocker push \\\"$IMAGE\\\"\\n```\\n\\nOptional security scan:\\n```bash\\ntrivy image \\\"$IMAGE\\\"\\n```\\n\\nIf you skip it: servers may build different artifacts at different times, making bugs difficult to reproduce. Reusing mutable tags such as `latest` makes rollback and incident investigation much harder.\")\nstep7 = StepsItem(\"7. Apply database migrations safely\", \"Why it matters: Schema changes must be coordinated with application code. Prefer backward-compatible, additive migrations so the old and new application versions can run during a rolling update.\\n\\nExact commands for a Prisma project:\\n```bash\\nssh \\\"$PROD_HOST\\\" \\\"cd $APP_DIR && docker compose pull\\\"\\nssh \\\"$PROD_HOST\\\" \\\"cd $APP_DIR && docker compose run --rm app npx prisma migrate deploy\\\"\\n```\\n\\nExact commands for a generic migration script:\\n```bash\\nssh \\\"$PROD_HOST\\\" \\\"cd $APP_DIR && docker compose run --rm app npm run migrate\\\"\\n```\\n\\nIf you skip it: new code may query tables or columns that do not exist. If you run unreviewed or destructive migrations, rollback may become impossible without restoring a backup.\")\nstep8 = StepsItem(\"8. Deploy the image and wait for service health\", \"Why it matters: Pulling and starting the pinned image ensures the host runs the exact release you built. Checking service status catches startup failures immediately rather than waiting for user reports.\\n\\nEnsure your production `docker-compose.yml` uses an image variable, for example `image: ${IMAGE}`. Then run:\\n```bash\\nssh \\\"$PROD_HOST\\\" \\\"cd $APP_DIR && export IMAGE='$IMAGE' && docker compose pull && docker compose up -d --remove-orphans\\\"\\nssh \\\"$PROD_HOST\\\" \\\"cd $APP_DIR && docker compose ps\\\"\\nssh \\\"$PROD_HOST\\\" \\\"cd $APP_DIR && docker compose logs --tail=100 app\\\"\\n```\\n\\nIf you skip it: the server may continue running an old image, start an unhealthy container unnoticed, or leave obsolete containers consuming ports and resources.\")\nstep9 = StepsItem(\"9. Verify production, monitor, and keep a rollback command ready\", \"Why it matters: A successful container start is not proof that users can use the service. Verify external routing, TLS, health checks, critical workflows, logs, and error rates immediately after release.\\n\\nExact commands:\\n```bash\\nexport DOMAIN=\\\"https://app.example.com\\\"\\ncurl -fsS \\\"$DOMAIN/health\\\"\\ncurl -fIs \\\"$DOMAIN/\\\"\\nssh \\\"$PROD_HOST\\\" \\\"cd $APP_DIR && docker compose logs --since=10m app\\\"\\n```\\n\\nRollback command: replace `PREVIOUS_VERSION` with the last known-good image tag, then redeploy it:\\n```bash\\nexport PREVIOUS_VERSION=\\\"PREVIOUS_VERSION\\\"\\nexport ROLLBACK_IMAGE=\\\"$REGISTRY/$APP_NAME:$PREVIOUS_VERSION\\\"\\nssh \\\"$PROD_HOST\\\" \\\"cd $APP_DIR && export IMAGE='$ROLLBACK_IMAGE' && docker compose pull && docker compose up -d\\\"\\n```\\n\\nIf you skip it: DNS, TLS, reverse-proxy routing, authentication, background jobs, or third-party integrations can fail while the container still appears healthy. Without a prepared rollback command, recovery takes longer during an incident.\")","promptTokens":4407,"completionTokens":2080,"latencyMs":22908,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/adversarial","repeat":4,"output":"root = Stack([header, deploymentGuide, note], \"column\", \"l\")\nheader = Card([title, subtitle], \"column\", \"s\")\ntitle = TextContent(\"Production deployment: Docker Compose\", \"large-heavy\")\nsubtitle = TextContent(\"A repeatable deployment workflow for a web application running behind Nginx. Commands assume Ubuntu 22.04+ and a repository containing a Dockerfile.\", \"default\")\ndeploymentGuide = Steps([step1, step2, step3, step4, step5, step6, step7, step8, step9])\nstep1 = StepsItem(\"1. Prepare the server\", \"Why it matters: production workloads need an updated, secured host with a non-root deployment account. This prevents routine application operations from requiring root access and reduces exposure to known operating-system vulnerabilities.\\n\\nRun:\\n```bash\\nsudo apt update && sudo apt -y upgrade\\nsudo apt install -y ca-certificates curl git ufw\\nsudo adduser deploy\\nsudo usermod -aG sudo deploy\\nsudo ufw allow OpenSSH\\nsudo ufw allow 80/tcp\\nsudo ufw allow 443/tcp\\nsudo ufw --force enable\\n```\\n\\nIf you skip it: you may deploy onto an unpatched host, lock yourself out by enabling a firewall without SSH access, or run application commands as root, increasing the blast radius of mistakes or compromised dependencies.\")\nstep2 = StepsItem(\"2. Install Docker and Docker Compose\", \"Why it matters: Docker packages the application and its dependencies into a consistent runtime. Docker Compose makes the deployment definition reproducible, including the application, database, networks, volumes, and restart behavior.\\n\\nRun:\\n```bash\\ncurl -fsSL https://get.docker.com -o get-docker.sh\\nsudo sh get-docker.sh\\nsudo usermod -aG docker deploy\\nsudo systemctl enable --now docker\\nsudo apt install -y docker-compose-plugin\\nnewgrp docker\\ndocker --version\\ndocker compose version\\n```\\n\\nIf you skip it: the application may work differently across servers because runtime versions, libraries, and startup configuration are installed manually or inconsistently.\")\nstep3 = StepsItem(\"3. Create a deployment directory and clone the application\", \"Why it matters: keeping the deployed source in one predictable location simplifies updates, backups, permissions, automation, and incident response.\\n\\nRun:\\n```bash\\nsudo mkdir -p /srv/myapp\\nsudo chown -R deploy:deploy /srv/myapp\\ncd /srv/myapp\\ngit clone https://github.com/YOUR_ORG/YOUR_REPOSITORY.git app\\ncd app\\ngit checkout main\\n```\\n\\nIf you skip it: code may be copied manually, difficult to audit, and easy to overwrite. Future deployments become error-prone because there is no clear source revision or directory convention.\")\nstep4 = StepsItem(\"4. Create production environment configuration\", \"Why it matters: secrets and environment-specific settings must be separated from source code. This keeps credentials out of Git history and lets the same image run in development, staging, and production with different configuration.\\n\\nRun:\\n```bash\\ncd /srv/myapp/app\\ncat > .env <<'EOF'\\nNODE_ENV=production\\nPORT=3000\\nAPP_URL=https://app.example.com\\nDATABASE_URL=postgresql://myapp:CHANGE_THIS_DATABASE_PASSWORD@db:5432/myapp\\nSESSION_SECRET=CHANGE_THIS_TO_A_LONG_RANDOM_VALUE\\nEOF\\nchmod 600 .env\\n```\\n\\nGenerate secure values with:\\n```bash\\nopenssl rand -base64 48\\n```\\n\\nIf you skip it: the application may start with development defaults, fail to connect to required services, expose insecure settings, or leak secrets by committing them to the repository.\")\nstep5 = StepsItem(\"5. Define the production Compose stack\", \"Why it matters: the Compose file declares exactly how containers are built and run. Health checks, persistent volumes, restart policies, and internal networking make failures easier to recover from and prevent data loss.\\n\\nRun:\\n```bash\\ncd /srv/myapp/app\\ncat > compose.production.yml <<'EOF'\\nservices:\\n app:\\n build: .\\n env_file: .env\\n restart: unless-stopped\\n expose:\\n - \\\"3000\\\"\\n depends_on:\\n db:\\n condition: service_healthy\\n networks:\\n - internal\\n\\n db:\\n image: postgres:16-alpine\\n environment:\\n POSTGRES_DB: myapp\\n POSTGRES_USER: myapp\\n POSTGRES_PASSWORD: CHANGE_THIS_DATABASE_PASSWORD\\n volumes:\\n - postgres_data:/var/lib/postgresql/data\\n restart: unless-stopped\\n healthcheck:\\n test: [\\\"CMD-SHELL\\\", \\\"pg_isready -U myapp -d myapp\\\"]\\n interval: 10s\\n timeout: 5s\\n retries: 5\\n networks:\\n - internal\\n\\n nginx:\\n image: nginx:1.27-alpine\\n ports:\\n - \\\"80:80\\\"\\n volumes:\\n - ./nginx.conf:/etc/nginx/conf.d/default.conf:ro\\n depends_on:\\n - app\\n restart: unless-stopped\\n networks:\\n - internal\\n\\nvolumes:\\n postgres_data:\\n\\nnetworks:\\n internal:\\nEOF\\n```\\n\\nIf you skip it: containers may not restart after a reboot, database data may disappear when a container is recreated, or services may be exposed directly to the internet without a controlled reverse proxy.\")\nstep6 = StepsItem(\"6. Configure Nginx as the reverse proxy\", \"Why it matters: Nginx accepts public HTTP traffic and forwards it to the internal application port. This keeps the app container off public ports and provides one location for request headers, compression, TLS, and rate limiting later.\\n\\nRun:\\n```bash\\ncd /srv/myapp/app\\ncat > nginx.conf <<'EOF'\\nserver {\\n listen 80;\\n server_name app.example.com;\\n\\n location / {\\n proxy_pass http://app:3000;\\n proxy_http_version 1.1;\\n proxy_set_header Host $host;\\n proxy_set_header X-Real-IP $remote_addr;\\n proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;\\n proxy_set_header X-Forwarded-Proto $scheme;\\n }\\n}\\nEOF\\n```\\n\\nIf you skip it: users must access an application port directly, common proxy headers may be missing, and adding HTTPS or traffic controls later becomes more difficult.\")\nstep7 = StepsItem(\"7. Build, start, and inspect the deployment\", \"Why it matters: building creates the production image from the checked-out source, while detached startup keeps the services running after your terminal closes. Inspecting status and logs immediately catches configuration and startup failures before users discover them.\\n\\nRun:\\n```bash\\ncd /srv/myapp/app\\ndocker compose --env-file .env -f compose.production.yml up -d --build\\ndocker compose --env-file .env -f compose.production.yml ps\\ndocker compose --env-file .env -f compose.production.yml logs --tail=100 app\\ndocker compose --env-file .env -f compose.production.yml logs --tail=100 nginx\\ncurl -I http://127.0.0.1\\n```\\n\\nIf you skip it: image-build errors, missing environment variables, failed database connections, and proxy misconfiguration can remain unnoticed until the service is unavailable.\")\nstep8 = StepsItem(\"8. Run database migrations and verify the release\", \"Why it matters: application code and database schema must remain compatible. Migrations bring the database to the version expected by the new application release; verification confirms the public service is healthy after the change.\\n\\nRun:\\n```bash\\ncd /srv/myapp/app\\ndocker compose --env-file .env -f compose.production.yml exec app npm run migrate\\ndocker compose --env-file .env -f compose.production.yml exec app npm run healthcheck\\ncurl -I http://app.example.com\\n```\\n\\nIf your project uses a different migration command, replace `npm run migrate` with its documented command, such as `npx prisma migrate deploy` or `python manage.py migrate`.\\n\\nIf you skip it: new code may query missing tables or columns, resulting in runtime errors, failed requests, or corrupted business workflows.\")\nstep9 = StepsItem(\"9. Perform future updates safely\", \"Why it matters: a consistent update procedure makes every release traceable and reduces downtime. Capturing the prior Git revision gives you a practical rollback point if the release fails verification.\\n\\nRun:\\n```bash\\ncd /srv/myapp/app\\ngit rev-parse HEAD > /srv/myapp/previous-release.txt\\ngit fetch origin\\ngit checkout main\\ngit pull --ff-only origin main\\ndocker compose --env-file .env -f compose.production.yml up -d --build\\ndocker compose --env-file .env -f compose.production.yml exec app npm run migrate\\ndocker compose --env-file .env -f compose.production.yml ps\\n```\\n\\nRollback commands:\\n```bash\\ncd /srv/myapp/app\\ngit checkout \\\"$(cat /srv/myapp/previous-release.txt)\\\"\\ndocker compose --env-file .env -f compose.production.yml up -d --build\\n```\\n\\nIf you skip it: deployments become manual and hard to reproduce. Without a recorded prior revision, rollback during an outage is slower and more uncertain.\")\nnote = Callout(\"warning\", \"Before going live\", \"Replace `app.example.com`, repository URL, and every `CHANGE_THIS` value. Point DNS for `app.example.com` to the server, then add TLS certificates—typically with Certbot—before handling real user traffic.\")","promptTokens":4407,"completionTokens":2214,"latencyMs":21733,"finishReason":"stop"} +{"model":"openai/gpt-5.6-terra","format":"openui-v2","scenario":"tasklist/adversarial","repeat":3,"output":"root = Stack([header, deploymentSteps, noteCard], \"column\", \"l\")\nheader = Card([title, subtitle], \"column\", \"s\")\ntitle = TextContent(\"Deploy a Dockerized application to Ubuntu\", \"large-heavy\")\nsubtitle = TextContent(\"A practical production deployment runbook using Docker Compose, Nginx, and HTTPS. Replace example.com, your repository URL, and image names with your real values.\", \"default\")\ndeploymentSteps = Steps([step1, step2, step3, step4, step5, step6, step7, step8, step9])\nstep1 = StepsItem(\"1. Prepare and update the server\", \"Why it matters: Deployment starts with a patched, accessible server. Updates close known security vulnerabilities, while a dedicated non-root user reduces the impact of a compromised application process.\\n\\nRun:\\nssh root@YOUR_SERVER_IP\\napt update && apt upgrade -y\\nadduser deploy\\nusermod -aG sudo deploy\\nmkdir -p /srv/myapp\\nchown -R deploy:deploy /srv/myapp\\n\\nIf you skip it: You may deploy onto an unpatched server, use root for routine work, and make later maintenance or incident recovery unnecessarily risky.\")\nstep2 = StepsItem(\"2. Configure firewall and SSH access\", \"Why it matters: A firewall exposes only required services. SSH key authentication is more resistant to password-guessing attacks than password login.\\n\\nRun:\\nufw allow OpenSSH\\nufw allow 80/tcp\\nufw allow 443/tcp\\nufw enable\\nufw status\\n\\nOptional SSH hardening after confirming key-based login works:\\nsed -i 's/^#\\\\?PasswordAuthentication.*/PasswordAuthentication no/' /etc/ssh/sshd_config\\nsystemctl reload ssh\\n\\nIf you skip it: Unnecessary ports may remain exposed, and your server may be more vulnerable to automated login attacks. Do not disable password authentication until you have verified a working SSH key login in a separate terminal.\")\nstep3 = StepsItem(\"3. Install Docker and Docker Compose\", \"Why it matters: Docker makes the application runtime reproducible across development, staging, and production. Docker Compose defines the application, database, volumes, and networks in version-controlled configuration.\\n\\nRun:\\napt install -y ca-certificates curl gnupg\\ninstall -m 0755 -d /etc/apt/keyrings\\ncurl -fsSL https://download.docker.com/linux/ubuntu/gpg | gpg --dearmor -o /etc/apt/keyrings/docker.gpg\\nchmod a+r /etc/apt/keyrings/docker.gpg\\necho \\\"deb [arch=$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/docker.gpg] https://download.docker.com/linux/ubuntu $(. /etc/os-release && echo $VERSION_CODENAME) stable\\\" | tee /etc/apt/sources.list.d/docker.list > /dev/null\\napt update\\napt install -y docker-ce docker-ce-cli containerd.io docker-buildx-plugin docker-compose-plugin\\nusermod -aG docker deploy\\n\\nIf you skip it: You cannot consistently run the containerized deployment described here. Installing software ad hoc on the host makes upgrades, rollback, and troubleshooting much harder.\")\nstep4 = StepsItem(\"4. Fetch the application and create production environment variables\", \"Why it matters: The application source or Compose configuration must be available on the server, while secrets must stay outside source control. The .env file supplies production-only values to Docker Compose.\\n\\nRun:\\nsu - deploy\\ncd /srv/myapp\\ngit clone https://github.com/YOUR_ORG/YOUR_REPOSITORY.git .\\ncp .env.example .env\\nnano .env\\n\\nExample .env values:\\nDOMAIN=example.com\\nAPP_IMAGE=ghcr.io/YOUR_ORG/YOUR_APP:latest\\nDATABASE_URL=postgresql://myapp:CHANGE_THIS_PASSWORD@db:5432/myapp\\nPOSTGRES_PASSWORD=CHANGE_THIS_TO_A_LONG_RANDOM_VALUE\\n\\nLock down secrets:\\nchmod 600 .env\\n\\nIf you skip it: The service may start with development settings, missing credentials, insecure defaults, or no database connection. Never commit .env files containing production secrets.\")\nstep5 = StepsItem(\"5. Create the production Docker Compose configuration\", \"Why it matters: Compose declares the desired production state: which image runs, which ports are internal, where persistent data lives, and how services restart after failures or reboots.\\n\\nRun:\\ncd /srv/myapp\\nnano compose.production.yml\\n\\na minimal pattern is:\\nservices:\\n app:\\n image: ${APP_IMAGE}\\n env_file: .env\\n restart: unless-stopped\\n expose:\\n - \\\"3000\\\"\\n db:\\n image: postgres:16\\n environment:\\n POSTGRES_DB: myapp\\n POSTGRES_USER: myapp\\n POSTGRES_PASSWORD: ${POSTGRES_PASSWORD}\\n volumes:\\n - postgres_data:/var/lib/postgresql/data\\n restart: unless-stopped\\nvolumes:\\n postgres_data:\\n\\nValidate the file:\\ndocker compose --env-file .env -f compose.production.yml config\\n\\nIf you skip it: Containers may not have persistent storage, restart behavior, or correct environment values. A database without a named volume can lose all data when its container is recreated.\")\nstep6 = StepsItem(\"6. Start the application and run database migrations\", \"Why it matters: Starting services lets Docker create networks, volumes, and containers. Migrations align the database schema with the version of application code being deployed.\\n\\nRun:\\ncd /srv/myapp\\ndocker compose --env-file .env -f compose.production.yml pull\\ndocker compose --env-file .env -f compose.production.yml up -d\\ndocker compose --env-file .env -f compose.production.yml ps\\ndocker compose --env-file .env -f compose.production.yml logs --tail=100 app\\n\\nIf your application has migrations, run its documented migration command, for example:\\ndocker compose --env-file .env -f compose.production.yml exec app npm run migrate\\n\\nIf you skip it: The application may be running an old image, may fail silently during startup, or may crash because its expected database tables and columns do not exist.\")\nstep7 = StepsItem(\"7. Configure Nginx as the public reverse proxy\", \"Why it matters: Nginx accepts public web traffic on ports 80 and 443, terminates TLS, and forwards requests to the application without exposing the application container directly to the internet.\\n\\nRun:\\nexit\\napt install -y nginx\\nnano /etc/nginx/sites-available/myapp\\n\\nUse this configuration, replacing example.com:\\nserver {\\n listen 80;\\n server_name example.com www.example.com;\\n location / {\\n proxy_pass http://127.0.0.1:3000;\\n proxy_set_header Host $host;\\n proxy_set_header X-Real-IP $remote_addr;\\n proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;\\n proxy_set_header X-Forwarded-Proto $scheme;\\n }\\n}\\n\\nEnable and test it:\\nln -s /etc/nginx/sites-available/myapp /etc/nginx/sites-enabled/myapp\\nrm -f /etc/nginx/sites-enabled/default\\nnginx -t\\nsystemctl reload nginx\\n\\nIf you skip it: Your app may only be accessible locally, or you may expose its internal port directly without standard TLS and proxy protections.\")\nstep8 = StepsItem(\"8. Point DNS and issue an HTTPS certificate\", \"Why it matters: DNS maps your domain to the server. HTTPS encrypts traffic, protects login sessions and API data in transit, and is expected by browsers and most production integrations.\\n\\nBefore running the commands, create A records for example.com and www.example.com pointing to YOUR_SERVER_IP.\\n\\nThen run:\\napt install -y certbot python3-certbot-nginx\\ncertbot --nginx -d example.com -d www.example.com --redirect\\ncertbot renew --dry-run\\n\\nVerify:\\ncurl -I https://example.com\\n\\nIf you skip it: Visitors may see connection failures, browser security warnings, or unencrypted HTTP traffic. Certificate renewal testing prevents discovering an expired certificate during an outage.\")\nstep9 = StepsItem(\"9. Verify, monitor, back up, and define the update procedure\", \"Why it matters: A successful container start is not proof of a healthy deployment. Verification checks the user-facing service; logs help diagnose problems; backups protect persistent data; a repeatable update process reduces downtime and makes rollback possible.\\n\\nRun health checks:\\ncurl -f https://example.com/health || curl -f https://example.com\\ndocker compose --env-file /srv/myapp/.env -f /srv/myapp/compose.production.yml ps\\ndocker compose --env-file /srv/myapp/.env -f /srv/myapp/compose.production.yml logs --tail=200 app\\n\\nCreate a database backup:\\nmkdir -p /srv/backups\\nsu - deploy -c 'cd /srv/myapp && docker compose --env-file .env -f compose.production.yml exec -T db pg_dump -U myapp myapp' > /srv/backups/myapp-$(date +%F).sql\\n\\nFor each update:\\nsu - deploy\\ncd /srv/myapp\\ngit pull\\ndocker compose --env-file .env -f compose.production.yml pull\\ndocker compose --env-file .env -f compose.production.yml up -d\\ndocker compose --env-file .env -f compose.production.yml logs --tail=100 app\\n\\nIf you skip it: You can miss broken routes, accumulating errors, failed restarts, and data-loss risks. Without tested backups and a documented update path, recovery from a bad release becomes slower and more uncertain.\")\nnoteCard = Callout(\"warning\", \"Customize before running\", \"These commands assume Ubuntu and a Dockerized web application that listens on port 3000. Confirm your application port, health-check path, migration command, repository, image registry, database credentials, and domain before executing them.\")","promptTokens":4407,"completionTokens":2280,"latencyMs":22678,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/minimal","repeat":4,"output":"root = Stack([\n Card([\n CardHeader(\"Contact Us\", \"We'd love to hear from you. Send us a message!\"),\n Separator(\"horizontal\"),\n ContactForm\n ]),\n $showSuccess ? Callout(\"success\", \"Message Sent\", \"Your message has been successfully delivered.\", $showSuccess) : null\n], \"center\")\n\nContactForm = Form(\"contact_form\", ContactButtons, [\n NameField,\n EmailField,\n MessageField\n])\n\nNameField = FormControl(\"Full Name\", Input(\"name\", \"e.g. John Doe\", \"text\", {required: true}, $name))\n\nEmailField = FormControl(\"Email Address\", Input(\"email\", \"e.g. john@example.com\", \"email\", {required: true, email: true}, $email))\n\nMessageField = FormControl(\"Your Message\", TextArea(\"message\", \"How can we help you?\", 5, {required: true, minLength: 10}, $message))\n\nContactButtons = Buttons([\n SubmitBtn,\n ResetBtn\n], \"row\")\n\nSubmitBtn = Button(\"Send Message\", Action([\n @Set($showSuccess, true),\n @Set($name, \"\"),\n @Set($email, \"\"),\n @Set($message, \"\")\n]), \"primary\")\n\nResetBtn = Button(\"Clear\", Action([\n @Set($name, \"\"),\n @Set($email, \"\"),\n @Set($message, \"\")\n]), \"secondary\")\n\n$name = \"\"\n$email = \"\"\n$message = \"\"\n$showSuccess = false","promptTokens":4543,"completionTokens":349,"latencyMs":11848,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/minimal","repeat":2,"output":"root = Stack([\n Card([\n CardHeader(\"Contact Us\", \"We'd love to hear from you. Please fill out the form below.\"),\n ContactForm\n ]),\n SuccessToast\n], \"center\")\n\nContactForm = Form(\"contact_form\", formButtons, [\n nameField,\n emailField,\n messageField\n])\n\nnameField = FormControl(\"Full Name\", Input(\"name\", \"John Doe\", \"text\", {required: true}, $name), \"Please enter your name\")\nemailField = FormControl(\"Email Address\", Input(\"email\", \"john@example.com\", \"email\", {required: true, email: true}, $email), \"We'll use this to reply to you\")\nmessageField = FormControl(\"Message\", TextArea(\"message\", \"How can we help?\", 5, {required: true, minLength: 10}, $message), \"Minimum 10 characters\")\n\nformButtons = Buttons([\n submitBtn,\n clearBtn\n], \"row\")\n\nsubmitBtn = Button(\"Send Message\", Action([@ToAssistant(\"Form submitted: \" + $name + \" (\" + $email + \"): \" + $message), @Set($showSuccess, true)]), \"primary\")\nclearBtn = Button(\"Clear\", Action([@Set($name, \"\"), @Set($email, \"\"), @Set($message, \"\"), @Set($showSuccess, false)]), \"secondary\")\n\nSuccessToast = Callout(\"success\", \"Sent!\", \"Your message has been sent successfully.\", $showSuccess)\n\n$name = \"\"\n$email = \"\"\n$message = \"\"\n$showSuccess = false","promptTokens":4543,"completionTokens":353,"latencyMs":15421,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/minimal","repeat":3,"output":"root = Stack([\n Card([\n CardHeader(\"Contact Us\", \"We'd love to hear from you. Please fill out the form below.\"),\n Separator(\"horizontal\"),\n ContactForm\n ]),\n SuccessCallout\n], \"column\", \"m\")\n\nContactForm = Form(\"contact_form\", ContactButtons, [\n NameField,\n EmailField,\n MessageField\n])\n\nNameField = FormControl(\"Full Name\", Input(\"name\", \"John Doe\", \"text\", {required: true}), \"Please enter your name\")\n\nEmailField = FormControl(\"Email Address\", Input(\"email\", \"john@example.com\", \"email\", {required: true, email: true}), \"We'll never share your email\")\n\nMessageField = FormControl(\"Message\", TextArea(\"message\", \"How can we help you?\", 5, {required: true}), \"Tell us about your inquiry\")\n\nContactButtons = Buttons([\n SubmitBtn,\n ResetBtn\n], \"row\")\n\nSubmitBtn = Button(\"Send Message\", Action([@ToAssistant(\"Form submitted: Name: $name, Email: $email, Message: $message\"), @Set($showSuccess, true)]), \"primary\")\n\nResetBtn = Button(\"Clear\", Action([@Set($name, \"\"), @Set($email, \"\"), @Set($message, \"\"), @Set($showSuccess, false)]), \"secondary\")\n\nSuccessCallout = Callout(\"success\", \"Message Sent!\", \"Thank you for reaching out. We will get back to you shortly.\", $showSuccess)","promptTokens":4543,"completionTokens":332,"latencyMs":17163,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/realistic","repeat":0,"output":"root = Stack([header, mainContent], \"column\", \"none\", \"center\")\n\nheader = Card([\n CardHeader(\"Dental Care Clinic\", \"Request an Appointment\"),\n TextContent(\"Please fill out the form below and our team will contact you to confirm your visit.\")\n], \"clear\", \"column\", \"center\")\n\nmainContent = Card([\n appointmentForm\n], \"card\")\n\nappointmentForm = Form(\"appointment_request\", formButtons, [\n nameField,\n phoneField,\n emailField,\n treatmentField,\n dateField,\n notesField\n])\n\nnameField = FormControl(\"Full Name\", Input(\"fullName\", \"e.g. John Doe\", \"text\", {required: true}))\nphoneField = FormControl(\"Phone Number\", Input(\"phone\", \"e.g. +1 234 567 890\", \"tel\", {required: true}))\nemailField = FormControl(\"Email Address\", Input(\"email\", \"john@example.com\", \"email\", {required: true}))\ntreatmentField = FormControl(\"Treatment Needed\", Select(\"treatment\", [\n SelectItem(\"checkup\", \"Routine Checkup\"),\n SelectItem(\"cleaning\", \"Professional Cleaning\"),\n SelectItem(\"whitening\", \"Teeth Whitening\"),\n SelectItem(\"root_canal\", \"Root Canal\")\n], \"Select a treatment\", {required: true}))\ndateField = FormControl(\"Preferred Date\", DatePicker(\"preferredDate\", \"single\", {required: true}))\nnotesField = FormControl(\"Additional Notes\", TextArea(\"notes\", \"Tell us about any pain or specific concerns...\", 4))\n\nformButtons = Buttons([\n submitBtn,\n resetBtn\n], \"row\")\n\nsubmitBtn = Button(\"Request Appointment\", Action([@ToAssistant(\"Patient has submitted a new appointment request. Please process it.\")] ), \"primary\")\nresetBtn = Button(\"Clear Form\", Action([@ToAssistant(\"Patient cleared the form.\")] ), \"secondary\", \"normal\", \"small\")","promptTokens":4603,"completionTokens":432,"latencyMs":15166,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/realistic","repeat":1,"output":"root = Stack([header, mainContent], \"column\", \"xl\")\n\nheader = Card([\n CardHeader(\"SmileBright Dental Clinic\", \"Appointment Request Form\"),\n TextContent(\"Please fill out the form below to request a dental appointment. Our team will contact you shortly to confirm your slot.\")\n], \"sunk\")\n\nmainContent = Card([\n appointmentForm,\n successCallout\n], \"card\")\n\nappointmentForm = Form(\"appointment_request\", formButtons, [\n nameField,\n phoneField,\n emailField,\n treatmentField,\n dateField,\n notesField\n])\n\nnameField = FormControl(\"Full Name\", Input(\"fullName\", \"e.g. John Doe\", \"text\", {required: true}))\nphoneField = FormControl(\"Phone Number\", Input(\"phone\", \"e.g. +1 234 567 890\", \"tel\", {required: true}))\nemailField = FormControl(\"Email Address\", Input(\"email\", \"e.g. john@example.com\", \"email\", {required: true}))\ntreatmentField = FormControl(\"Treatment Needed\", Select(\"treatment\", [\n SelectItem(\"checkup\", \"Routine Checkup\"),\n SelectItem(\"cleaning\", \"Professional Cleaning\"),\n SelectItem(\"whitening\", \"Teeth Whitening\"),\n SelectItem(\"root_canal\", \"Root Canal\")\n], \"select\", {required: true}))\ndateField = FormControl(\"Preferred Date\", DatePicker(\"preferredDate\", \"single\", {required: true}))\nnotesField = FormControl(\"Additional Notes\", TextArea(\"notes\", \"Tell us about any pain or specific concerns...\", 4))\n\nformButtons = Buttons([\n submitBtn,\n resetBtn\n], \"row\")\n\nsubmitBtn = Button(\"Request Appointment\", Action([@ToAssistant(\"Patient submitted a new appointment request. Please process the form data.\"), @Set($showSuccess, true)]), \"primary\")\nresetBtn = Button(\"Clear Form\", Action([@Set($showSuccess, false)]), \"secondary\")\n\nsuccessCallout = Callout(\"success\", \"Request Sent!\", \"Thank you. We have received your request and will call you soon to confirm.\", $showSuccess)\n\n$showSuccess = false","promptTokens":4603,"completionTokens":480,"latencyMs":19179,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/adversarial","repeat":1,"output":"root = Stack([\n HeaderSection,\n FeedbackFormCard,\n SuccessCallout\n], \"column\", \"xl\", \"center\")\n\nHeaderSection = Card([\n CardHeader(\"Share Your Thoughts\", \"Help us improve your experience\"),\n TextContent(\"Your feedback is invaluable to us. It only takes a minute.\", \"small\")\n], \"clear\", \"column\", \"center\")\n\nFeedbackFormCard = Card([\n Form(\"feedback_form\", FormButtons, [\n RatingField,\n IssueField,\n ContactField\n ])\n], \"card\")\n\nRatingField = FormControl(\"How satisfied were you?\", RadioGroup(\"satisfaction\", [\n RadioItem(\"5\", \"🤩 Amazing\", \"5\"),\n RadioItem(\"4\", \"🙂 Good\", \"4\"),\n RadioItem(\"3\", \"😐 Okay\", \"3\"),\n RadioItem(\"2\", \"🙁 Poor\", \"2\"),\n RadioItem(\"1\", \"😡 Terrible\", \"1\")\n], \"5\"))\n\nIssueField = FormControl(\"What could we improve?\", TextArea(\"issue\", \"Tell us briefly what went wrong...\", 3))\n\nContactField = FormControl(\"Contact Info (Optional)\", Input(\"email\", \"Email address\", \"email\"))\n\nFormButtons = Buttons([\n SubmitBtn,\n CancelBtn\n], \"row\")\n\nSubmitBtn = Button(\"Submit Feedback\", Action([@ToAssistant(\"User submitted feedback: satisfaction: $satisfaction, issue: $issue, contact: $email\"), @Set($showSuccess, true)]), \"primary\")\n\nCancelBtn = Button(\"Cancel\", Action([@ToAssistant(\"User cancelled feedback\")]), \"secondary\", \"normal\", \"small\")\n\nSuccessCallout = Callout(\"success\", \"Thank You!\", \"Your feedback has been received. We appreciate your time.\", $showSuccess)","promptTokens":4586,"completionTokens":391,"latencyMs":6752,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/adversarial","repeat":0,"output":"root = Stack([header, feedbackForm, successCallout], \"column\", \"l\", \"center\")\n\nheader = Card([\n CardHeader(\"Feedback\", \"Your voice helps us improve\"),\n TextContent(\"We value your honest thoughts. This will only take a minute.\")\n], \"sunk\")\n\nsuccessCallout = Callout(\"success\", \"Thank you!\", \"Your feedback has been submitted successfully.\", $showSuccess)\n\n$showSuccess = false\n$happiness = 3\n$feedbackType = \"suggestion\"\n$email = \"\"\n$comments = \"\"\n\nformTitle = TextContent(\"Share your experience\", \"large-heavy\")\n\nhappinessLabel = TextContent(\"How happy were you with our service?\", \"default\")\nhappinessSlider = FormControl(\"Satisfaction Score\", Slider(\"rating\", \"continuous\", 1, 5, 1, [3], \"Rate from 1 to 5\", null, $happiness), null)\n\ntypeLabel = TextContent(\"What's the main reason?\", \"default\")\ntypeSelect = FormControl(\"Feedback Category\", Select(\"category\", [\n SelectItem(\"bug\", \"Something went wrong (Bug)\"),\n SelectItem(\"suggestion\", \"I have a suggestion\"),\n SelectItem(\"compliment\", \"I loved it!\"),\n SelectItem(\"other\", \"Other\")\n], \"Select a category\", null, $feedbackType), null)\n\ncontactLabel = TextContent(\"Want a follow-up?\", \"default\")\nemailField = FormControl(\"Email Address\", Input(\"email\", \"Enter your email for a reply\", \"email\", {required: false}, $email), \"Optional\")\n\ncommentsField = FormControl(\"Tell us more\", TextArea(\"comments\", \"Briefly describe the issue or idea...\", 3, {maxLength: 200}, $comments), \"Max 200 characters\")\n\nsubmitBtn = Button(\"Submit Feedback\", Action([@Set($showSuccess, true), @Set($email, \"\"), @Set($comments, \"\"), @Set($happiness, 3)]), \"primary\")\ncancelBtn = Button(\"Cancel\", Action([@Set($showSuccess, false)]), \"secondary\")\n\nformButtons = Buttons([submitBtn, cancelBtn], \"row\")\n\nfeedbackForm = Card([\n formTitle,\n Separator(\"horizontal\"),\n happinessSlider,\n typeSelect,\n emailField,\n commentsField,\n formButtons\n], \"card\")","promptTokens":4587,"completionTokens":514,"latencyMs":13764,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/adversarial","repeat":2,"output":"root = Stack([header, feedbackSection, successToast], \"column\", \"m\", \"center\")\n\nheader = Card([\n CardHeader(\"We value your feedback\", \"Help us improve by sharing your experience.\"),\n Separator(\"horizontal\")\n], \"sunk\")\n\nfeedbackSection = Card([\n Form(\"feedbackForm\", feedbackButtons, [\n ratingField,\n issueField,\n contactField\n ])\n], \"card\")\n\nratingField = FormControl(\"How satisfied were you?\", RadioGroup(\"satisfaction\", [\n RadioItem(\"excellent\", \"Excellent\", \"excellent\"),\n RadioItem(\"good\", \"Good\", \"good\"),\n RadioItem(\"okay\", \"Okay\", \"okay\"),\n RadioItem(\"poor\", \"Poor\", \"poor\")\n], \"good\"), \"Please select one\")\n\nissueField = FormControl(\"What could be better?\", TextArea(\"issue\", \"Tell us briefly what went wrong or how we can improve...\", 3), \"Optional\")\n\ncontactField = FormControl(\"Email address\", Input(\"email\", \"name@example.com\", \"email\"), \"Only if you want a reply\")\n\nfeedbackButtons = Buttons([\n submitBtn,\n cancelBtn\n], \"row\")\n\nsubmitBtn = Button(\"Submit Feedback\", Action([@Set($showSuccess, true), @ToAssistant(\"Feedback submitted successfully\")]), \"primary\")\ncancelBtn = Button(\"Cancel\", Action([@ToAssistant(\"Cancel feedback submission\")]), \"secondary\")\n\nsuccessToast = Callout(\"success\", \"Thank you!\", \"Your feedback has been received. We appreciate your time.\", $showSuccess)","promptTokens":4586,"completionTokens":342,"latencyMs":4629,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/realistic","repeat":4,"output":"root = Stack([header, formCard, successCallout], \"column\", \"l\", \"center\")\n\nheader = Card([\n CardHeader(\"Dental Care Clinic\", \"Request an Appointment\"),\n TextContent(\"Please fill out the form below and our team will contact you shortly to confirm your visit.\")\n], \"clear\")\n\nformCard = Card([\n Form(\"appointment_request\", submitButtons, [\n nameField,\n phoneField,\n emailField,\n treatmentField,\n dateField,\n notesField\n ])\n], \"card\")\n\nnameField = FormControl(\"Full Name\", Input(\"name\", \"e.g. Jane Doe\", \"text\", {required: true}, $name), \"Required\")\nphoneField = FormControl(\"Phone Number\", Input(\"phone\", \"e.g. +1 234 567 890\", \"tel\", {required: true}, $phone), \"Required\")\nemailField = FormControl(\"Email Address\", Input(\"email\", \"jane@example.com\", \"email\", {required: true}, $email), \"Required\")\ntreatmentField = FormControl(\"Treatment Needed\", Select(\"treatment\", [\n SelectItem(\"checkup\", \"Routine Checkup\"),\n SelectItem(\"cleaning\", \"Teeth Cleaning\"),\n SelectItem(\"whitening\", \"Teeth Whitening\"),\n SelectItem(\"root_canal\", \"Root Canal\")\n], \"Select a treatment\", {required: true}, $treatment), \"Required\")\ndateField = FormControl(\"Preferred Date\", DatePicker(\"preferred_date\", \"single\", {required: true}, $prefDate))\nnotesField = FormControl(\"Additional Notes\", TextArea(\"notes\", \"Any pain, allergies, or specific concerns?\", 4, null, $notes))\n\nsubmitButtons = Buttons([\n submitBtn,\n resetBtn\n], \"row\")\n\nsubmitBtn = Button(\"Request Appointment\", Action([@ToAssistant(\"New appointment request submitted: \" + $name + \" wants \" + $treatment)]), \"primary\")\nresetBtn = Button(\"Clear Form\", Action([@Set($name, \"\"), @Set($phone, \"\"), @Set($email, \"\"), @Set($treatment, \"\"), @Set($prefDate, \"\"), @Set($notes, \"\")]), \"secondary\")\n\nsuccessCallout = Callout(\"success\", \"Request Sent\", \"Thank you! We have received your request and will call you soon.\", $showSuccess)\n\n$name = \"\"\n$phone = \"\"\n$email = \"\"\n$treatment = \"\"\n$prefDate = null\n$notes = \"\"\n$showSuccess = false","promptTokens":4603,"completionTokens":557,"latencyMs":18967,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/adversarial","repeat":4,"output":"root = Stack([header, feedbackCard, successToast], \"column\", \"m\", \"center\")\n\nheader = Card([\n CardHeader(\"Customer Feedback\", \"Help us improve your experience\"),\n TextContent(\"Your input is invaluable to us. It only takes a minute.\", \"small\")\n], \"clear\", \"column\", \"center\")\n\nsuccessToast = Callout(\"success\", \"Thank you!\", \"Your feedback has been submitted successfully.\", $showSuccess)\n\nfeedbackCard = Card([\n Form(\"feedbackForm\", submitButtons, [\n ratingField,\n issueField,\n contactField\n ])\n], \"card\", \"column\", \"start\")\n\nratingField = FormControl(\"How satisfied were you?\", RadioGroup(\"satisfaction\", [\n RadioItem(\"5\", \"Excellent\", \"5\"),\n RadioItem(\"4\", \"Good\", \"4\"),\n RadioItem(\"3\", \"Average\", \"3\"),\n RadioItem(\"2\", \"Poor\", \"2\"),\n RadioItem(\"1\", \"Terrible\", \"1\")\n], \"4\", value=$rating, rules={required: true}))\n\nissueField = FormControl(\"What went wrong? (Optional)\", TextArea(\"issue\", \"Tell us more...\", 3, value=$issueText))\n\ncontactField = FormControl(\"Email address (if you'd like a reply)\", Input(\"email\", \"name@example.com\", \"email\", value=$email, rules={email: true}))\n\nsubmitButtons = Buttons([\n Button(\"Submit Feedback\", Action([@Set($showSuccess, true), @Set($rating, null), @Set($issueText, \"\"), @Set($email, \"\"))]),\n Button(\"Cancel\", Action([@Set($rating, null), @Set($issueText, \"\"), @Set($email, \"\")]), \"secondary\", \"normal\", \"small\")\n], \"row\")","promptTokens":4586,"completionTokens":401,"latencyMs":5864,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/minimal","repeat":1,"output":"root = Stack([\n Card([\n CardHeader(\"World Geography\", \"Top 3 Largest Countries by Area\"),\n Separator(),\n Table([\n Col(\"Rank\", [\"1\", \"2\", \"3\"]),\n Col(\"Country\", [\"Russia\", \"Canada\", \"China\"]),\n Col(\"Area (sq km)\", [\"17,098,242\", \"9,984,670\", \"9,596,961\"], \"number\")\n ])\n ])\n], \"column\", \"m\")","promptTokens":4543,"completionTokens":124,"latencyMs":3752,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/minimal","repeat":0,"output":"root = Stack([\n HeaderSection,\n TableSection\n], \"column\", \"l\")\n\nHeaderSection = Card([\n CardHeader(\"World Geography\", \"Top 3 Largest Countries by Land Area\"),\n TextContent(\"A comparison of the world's largest sovereign states based on total area (including water).\")\n], \"sunk\")\n\nTableSection = Card([\n Table([\n Col(\"Rank\", rankData),\n Col(\"Country\", countryData),\n Col(\"Area (km²)\", areaData, \"number\"),\n Col(\"Continent\", continentData)\n ])\n])\n\nrankData = [\"1\", \"2\", \"3\"]\ncountryData = [\"Russia\", \"Canada\", \"China\"]\nareaData = [17098242, 9984670, 9596961]\ncontinentData = [\"Europe/Asia\", \"North America\", \"Asia\"]","promptTokens":4543,"completionTokens":207,"latencyMs":8295,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/realistic","repeat":3,"output":"root = Stack([header, mainContent], \"column\", \"none\", \"center\")\n\nheader = Card([\n CardHeader(\"Dental Clinic Appointment Request\", \"Please fill out the form below to request a visit.\"),\n TextContent(\"We will contact you shortly to confirm your appointment time.\")\n], \"clear\")\n\nmainContent = Card([\n appointmentForm\n], \"card\")\n\nappointmentForm = Form(\"appointment_request\", formButtons, [\n nameField,\n phoneField,\n emailField,\n treatmentField,\n dateField,\n notesField\n])\n\nnameField = FormControl(\"Full Name\", Input(\"fullName\", \"e.g. John Doe\", \"text\", {required: true}))\n\nphoneField = FormControl(\"Phone Number\", Input(\"phone\", \"e.g. +1 234 567 890\", \"tel\", {required: true}))\n\nemailField = FormControl(\"Email Address\", Input(\"email\", \"e.g. john@example.com\", \"email\", {required: true}))\n\ntreatmentField = FormControl(\"Treatment Needed\", Select(\"treatment\", [\n SelectItem(\"checkup\", \"Routine Checkup\"),\n SelectItem(\"cleaning\", \"Teeth Cleaning\"),\n SelectItem(\"whitening\", \"Teeth Whitening\"),\n SelectItem(\"root_canal\", \"Root Canal\"),\n], \"Select a treatment\", {required: true}))\n\ndateField = FormControl(\"Preferred Date\", DatePicker(\"preferredDate\", \"single\", {required: true}))\n\nnotesField = FormControl(\"Additional Notes\", TextArea(\"notes\", \"Tell us about any pain or specific concerns...\", 4))\n\nformButtons = Buttons([\n submitBtn,\n resetBtn\n], \"row\")\n\nsubmitBtn = Button(\"Request Appointment\", Action([@ToAssistant(\"A new appointment request has been submitted. Please process it.\"), @Set($showSuccess, true)]), \"primary\")\n\nresetBtn = Button(\"Clear Form\", Action([@Set($fullName, \"\"), @Set($phone, \"\"), @Set($email, \"\"), @Set($treatment, \"\"), @Set($preferredDate, \"\"), @Set($notes, \"\")]), \"secondary\")\n\nsuccessToast = Callout(\"success\", \"Request Sent\", \"Your appointment request has been received. We will call you soon.\", $showSuccess)\n\n// Note: $showSuccess would be managed by the application state. \n// In this declarative UI, we include it as a potential component to be rendered.\n// Since the prompt asks for the form, the core logic is in the Form component.","promptTokens":4603,"completionTokens":546,"latencyMs":32421,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/realistic","repeat":2,"output":"root = Stack([\n HeaderCard,\n FormCard,\n SuccessCallout\n], \"column\", \"l\", \"center\")\n\nHeaderCard = Card([\n CardHeader(\"Book an Appointment\", \"Request your dental visit below\"),\n TextContent(\"Please fill out the details and our team will contact you shortly to confirm your slot.\", \"small\")\n], \"sunk\")\n\nSuccessCallout = Callout(\"success\", \"Request Sent\", \"We have received your request and will call you shortly to confirm.\", $showSuccess)\n\nFormCard = Card([\n Form(\"appointment_request\", submitButtons, [\n nameField,\n phoneField,\n emailField,\n treatmentField,\n dateField,\n notesField\n ])\n], \"card\")\n\nnameField = FormControl(\"Full Name*\", Input(\"fullName\", \"e.g. John Doe\", \"text\", {required: true}))\nphoneField = FormControl(\"Phone Number*\", Input(\"phone\", \"e.g. +1 234 567 890\", \"tel\", {required: true, pattern: \"^[+]?[0-9\\\\s-]{7,15}$\"}))\nemailField = FormControl(\"Email Address\", Input(\"email\", \"john@example.com\", \"email\", {email: true}))\ntreatmentField = FormControl(\"Treatment Type*\", Select(\"treatment\", [\n SelectItem(\"checkup\", \"Routine Checkup\"),\n SelectItem(\"cleaning\", \"Professional Cleaning\"),\n SelectItem(\"whitening\", \"Teeth Whitening\"),\n SelectItem(\"root_canal\", \"Root Canal\")\n], {required: true}))\ndateField = FormControl(\"Preferred Date\", DatePicker(\"preferredDate\", \"single\", {required: true}))\nnotesField = FormControl(\"Additional Notes\", TextArea(\"notes\", \"Any pain, allergies, or specific concerns...\", 4))\n\nsubmitButtons = Buttons([\n submitBtn,\n resetBtn\n], \"row\")\n\nsubmitBtn = Button(\"Request Appointment\", Action([@Set($showSuccess, true), @ToAssistant(\"New appointment request submitted via website form\")]), \"primary\")\nresetBtn = Button(\"Clear Form\", Action([@Set($showSuccess, false)]), \"secondary\", \"normal\", \"small\")","promptTokens":4603,"completionTokens":488,"latencyMs":37586,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/minimal","repeat":3,"output":"root = Stack([\n Card([\n CardHeader(\"World Geography\", \"Top 3 Largest Countries by Land Area\"),\n Separator(\"horizontal\"),\n Table([\n Col(\"Rank\", [\"1\", \"2\", \"3\"]),\n Col(\"Country\", [\"Russia\", \"Canada\", \"China\"]),\n Col(\"Area (sq km)\", [\"17,098,242\", \"9,984,670\", \"9,596,961\"], \"number\"),\n Col(\"Continent\", [\"Europe/Asia\", \"North America\", \"Asia\"])\n ])\n ])\n], \"column\", \"m\")","promptTokens":4543,"completionTokens":145,"latencyMs":7087,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/adversarial","repeat":3,"output":"root = Stack([\n headerCard,\n feedbackFormCard,\n successMessage\n], \"column\", \"m\", \"center\")\n\nheaderCard = Card([\n CardHeader(\"We value your voice\", \"Help us improve your experience\"),\n TextContent(\"It only takes 30 seconds. Your feedback helps us build a better product for you.\")\n], \"sunk\", \"column\", \"center\")\n\nfeedbackFormCard = Card([\n Form(\"feedback_form\", feedbackButtons, [\n ratingField,\n issueField,\n contactField\n ])\n], \"card\", \"column\", \"start\")\n\nratingField = FormControl(\"How satisfied were you?\", Slider(\"satisfaction\", \"continuous\", 1, 5, 1, [3], \"1 = Poor, 5 = Excellent\"))\n\nissueField = FormControl(\"What went wrong?\", Select(\"issue_type\", [\n SelectItem(\"none\", \"Everything was great!\"),\n SelectItem(\"ui\", \"User Interface / Design\"),\n SelectItem(\"bugs\", \"Technical Bug / Error\"),\n SelectItem(\"speed\", \"Performance / Speed\"),\n SelectItem(\"feature\", \"Missing Feature\"),\n SelectItem(\"other\", \"Other\")\n], \"Please select an issue type\", \"none\"))\n\ncontactField = FormControl(\"How can we reach you? (Optional)\", Input(\"contact_email\", \"email\", \"email\", placeholder: \"email@example.com\"))\n\nfeedbackButtons = Buttons([\n submitBtn,\n cancelBtn\n], \"row\")\n\nsubmitBtn = Button(\"Send Feedback\", Action([@ToAssistant(\"Feedback submitted successfully!\"), @Set($showSuccess, true)]), \"primary\")\n\ncancelBtn = Button(\"Cancel\", Action([@Set($showSuccess, false)]), \"secondary\")\n\nsuccessMessage = Callout(\"success\", \"Thank you!\", \"Your feedback has been received. We appreciate your time.\", $showSuccess)","promptTokens":4587,"completionTokens":410,"latencyMs":24193,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/minimal","repeat":4,"output":"root = Stack([\n headerCard,\n tableCard\n], \"column\", \"l\")\n\nheaderCard = Card([\n CardHeader(\"World Geography\", \"Top 3 Largest Countries by Total Area\"),\n TextContent(\"A comparison of the largest sovereign nations based on land and water area.\")\n], \"card\")\n\ntableCard = Card([\n Table([\n Col(\"Rank\", [\"1\", \"2\", \"3\"]),\n Col(\"Country\", [\"Russia\", \"Canada\", \"China\"]),\n Col(\"Area (km²)\", [\"17,098,242\", \"9,984,670\", \"9,596,961\"], \"number\"),\n Col(\"Continent\", [\"Europe/Asia\", \"North America\", \"Asia\"])\n ])\n], \"sunk\")","promptTokens":4543,"completionTokens":183,"latencyMs":17243,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"contact-form/minimal","repeat":0,"output":"root = Stack([\n Card([\n CardHeader(\"Contact Us\", \"We'd love to hear from you. Send us a message and we'll get back to you shortly.\"),\n Separator(\"horizontal\"),\n ContactForm\n ]),\n SuccessCallout\n], \"column\", \"m\", \"center\")\n\nContactForm = Form(\"contact_form\", FormButtons, [\n NameField,\n EmailField,\n MessageField\n])\n\nFormButtons = Buttons([\n SubmitBtn,\n ResetBtn\n], \"row\")\n\nSubmitBtn = Button(\"Send Message\", Action([@Set($showSuccess, true), @ToAssistant(\"Message sent successfully!\")]), \"primary\")\nResetBtn = Button(\"Clear\", Action([@Set($name, \"\"), @Set($email, \"\"), @Set($message, \"\")]), \"secondary\")\n\nNameField = FormControl(\"Full Name\", Input(\"name\", \"Enter your name\", \"text\", {required: true}, $name))\nEmailField = FormControl(\"Email Address\", Input(\"email\", \"Enter your email\", \"email\", {required: true, email: true}, $email))\nMessageField = FormControl(\"Message\", TextArea(\"message\", \"How can we help?\", 5, {required: true, minLength: 10}, $message))\n\nSuccessCallout = Callout(\"success\", \"Sent!\", \"Your message has been sent successfully.\", $showSuccess)\n\n$name = \"\"\n$email = \"\"\n$message = \"\"\n$showSuccess = false","promptTokens":4543,"completionTokens":327,"latencyMs":81642,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/realistic","repeat":1,"output":"root = Stack([\n headerCard,\n statsGrid,\n salesTableCard,\n performanceChartCard\n], \"column\", \"l\")\n\nheaderCard = Card([\n CardHeader(\"Q3 Sales Performance\", \"Regional Results: EMEA & APAC\"),\n TextContent(\"Summary of sales representative achievement against quarterly targets.\")\n], \"sunk\")\n\nstatsGrid = Stack([\n totalRevenueCard,\n totalDealsCard,\n avgQuotaCard\n], \"row\", \"m\", \"stretch\")\n\ntotalRevenueCard = Card([\n TextContent(\"Total Revenue\", \"small\"),\n TextContent(\"$1,428,500\", \"large-heavy\")\n], \"card\")\n\ntotalDealsCard = Card([\n TextContent(\"Deals Closed\", \"small\"),\n TextContent(\"42\", \"large-heavy\")\n], \"card\")\n\navgQuotaCard = Card([\n TextContent(\"Avg. Quota Attainment\", \"small\"),\n TextContent(\"104%\", \"large-heavy\")\n], \"card\")\n\nsalesTableCard = Card([\n CardHeader(\"Representative Breakdown\"),\n Separator(\"horizontal\"),\n @Count(salesData.rows) > 0 ? Table([\n Col(\"Rep Name\", salesData.rows.name),\n Col(\"Region\", salesData.rows.region),\n Col(\"Deals\", salesData.rows.deals),\n Col(\"Revenue\", salesData.rows.revenue, \"number\"),\n Col(\"Quota %\", salesData.rows.quota_pct, \"number\")\n ]) : TextContent(\"No data available\")\n], \"card\")\n\nperformanceChartCard = Card([\n CardHeader(\"Revenue by Region\"),\n BarChart([\"EMEA\", \"APAC\"], [Series(\"Revenue\", [745000, 683500])])\n], \"card\")\n\nsalesData = {\n rows: [\n {id: 1, name: \"Elena Rossi\", region: \"EMEA\", deals: 6, revenue: 215000, quota_pct: 112},\n {id: 2, name: \"Marcus Weber\", region: \"EMEA\", deals: 5, revenue: 185000, quota_pct: 98},\n {id: 3, name: \"Sophie Dubois\", region: \"EMEA\", deals: 7, revenue: 240000, quota_pct: 120},\n {id: 4, name: \"Lukas Meyer\", region: \"EMEA\", deals: 4, revenue: 120000, quota_pct: 85},\n {id: 5, name: \"Hiroshi Tanaka\", region: \"APAC\", deals: 6, revenue: 195000, quota_pct: 105},\n {id: 6, name: \"Ji-won Kim\", region: \"APAC\", deals: 5, revenue: 165000, quota_pct: 92},\n {id: 7, name: \"Arjun Mehta\", region: \"APAC\", deals: 5, revenue: 180000, quota_pct: 102},\n {id: 8, name: \"Li Wei\", region: \"APAC\", deals: 4, revenue: 133500, quota_pct: 86}\n ]\n}","promptTokens":4578,"completionTokens":782,"latencyMs":23184,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/minimal","repeat":2,"output":"root = Stack([\n headerCard,\n tableCard\n], \"column\", \"m\")\n\nheaderCard = Card([\n CardHeader(\"Global Geography\", \"Top 3 Largest Countries by Land Area\"),\n TextContent(\"This table displays the three largest countries in the world based on total area, including land and water.\")\n], \"sunk\")\n\ncountries = [\n {id: 1, name: \"Russia\", area: 17098242, continent: \"Europe/Asia\"},\n {id: 2, name: \"Canada\", area: 9984670, continent: \"North America\"},\n {id: 3, name: \"China\", area: 9596961, continent: \"Asia\"}\n]\n\ntableCard = Card([\n Table([\n Col(\"Rank\", [\"1\", \"2\", \"3\"]),\n Col(\"Country\", countries.map(c => c.name)),\n Col(\"Area (km²)\", countries.map(c => c.area.toLocaleString()), \"number\"),\n Col(\"Continent\", countries.map(c => c.continent))\n ])\n], \"card\")","promptTokens":4542,"completionTokens":263,"latencyMs":42058,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/realistic","repeat":2,"output":"root = Stack([header, kpiRow, mainContent], \"column\", \"l\")\n\nheader = Card([\n CardHeader(\"Q3 Sales Performance\", \"Quarterly Results - EMEA & APAC Regions\"),\n Separator(\"horizontal\")\n], \"sunk\")\n\nkpiRow = Stack([\n kpiCard(\"Total Revenue\", \"$1.42M\"),\n kpiCard(\"Deals Closed\", \"42\"),\n kpiCard(\"Quota Attainment\", \"94%\")\n], \"row\", \"m\")\n\nmainContent = Card([\n Tabs([\n TabItem(\"table\", \"Detailed View\", [salesTable]),\n TabItem(\"chart\", \"Revenue by Rep\", [revenueChart])\n ])\n], \"card\")\n\nkpiCard = (title, value) => Card([\n TextContent(title, \"small\"),\n TextContent(value, \"large-heavy\")\n], \"sunk\")\n\nsalesData = [\n {id: 1, name: \"Sarah Jenkins\", region: \"EMEA\", deals: 6, revenue: 210000, quota: 200000},\n {id: 2, name: \"Liam O'Connor\", region: \"EMEA\", deals: 5, revenue: 185000, quota: 200000},\n {id: 3, name: \"Chen Wei\", region: \"APAC\", deals: 7, revenue: 245000, quota: 220000},\n {id: 4, name: \"Aiko Tanaka\", region: \"APAC\", deals: 4, revenue: 155000, quota: 180000},\n {id: 5, name: \"Marcus Schmidt\", region: \"EMEA\", deals: 8, revenue: 290000, quota: 250000},\n {id: 6, name: \"Priya Sharma\", region: \"APAC\", deals: 5, revenue: 175000, quota: 190000},\n {id: 7, name: \"Elena Rossi\", region: \"EMEA\", deals: 4, revenue: 130000, quota: 175000},\n {id: 8, name: \"David Kim\", region: \"APAC\", deals: 6, revenue: 230000, quota: 210000}\n]\n\nsalesTable = Table([\n Col(\"Rep Name\", salesData.map(r => r.name)),\n Col(\"Region\", salesData.map(r => r.region)),\n Col(\"Deals\", salesData.map(r => r.deals, \"number\")),\n Col(\"Revenue\", salesData.map(r => \"$\" + (r.revenue / 1000) + \"k\", \"number\")),\n Col(\"Quota Attainment\", salesData.map(r => {\n let pct = (r.revenue / r.quota) * 100;\n let status = pct >= 100 ? \"success\" : pct >= 90 ? \"warning\" : \"danger\";\n return Tag(pct.toFixed(1) + \"%\", null, \"sm\", status);\n }))\n])\n\nrevenueChart = BarChart(\n salesData.map(r => r.name),\n [Series(\"Revenue ($)\", salesData.map(r => r.revenue))],\n \"grouped\",\n \"Sales Representative\",\n \"USD\"\n)","promptTokens":4578,"completionTokens":816,"latencyMs":30352,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/adversarial","repeat":2,"output":"root = Stack([header, kpiRow, mainContent, recommendation], \"column\", \"l\")\n\nheader = Card([\n CardHeader(\"Cloud Provider Comparison\", \"Comprehensive technical and economic analysis of the top 5 hyperscalers\"),\n TextContent(\"A detailed breakdown of AWS, Azure, GCP, Oracle Cloud (OCI), and DigitalOcean to assist in architectural decision-making.\")\n], \"card\", \"column\", \"start\", \"stretch\")\n\nkpiRow = Stack([\n kpiAws,\n kpiAzure,\n kpiGcp,\n kpiOci,\n kpiDo\n], \"row\", \"s\", \"start\", \"stretch\", true)\n\nkpiAws = Card([TextContent(\"AWS\", \"small\"), TextContent(\"Market Leader\", \"large-heavy\")], \"sunk\")\nkpiAzure = Card([TextContent(\"Azure\", \"small\"), TextContent(\"Enterprise Std\", \"large-heavy\")], \"sunk\")\nkpiGcp = Card([TextContent(\"GCP\", \"small\"), TextContent(\"Data/AI King\", \"large-heavy\")], \"sunk\")\nkpiOci = Card([TextContent(\"OCI\", \"small\"), TextContent(\"Price/Perf\", \"large-heavy\")], \"sunk\")\nkpiDo = Card([TextContent(\"DigitalOcean\", \"small\"), TextContent(\"Dev Friendly\", \"large-heavy\")], \"sunk\")\n\nmainContent = Tabs([\n TabItem(\"specs\", \"Technical Specs\", [specTable]),\n TabItem(\"pricing\", \"Economic Model\", [pricingTable]),\n TabItem(\"performance\", \"Performance & Latency\", [performanceChart])\n])\n\nspecTable = Card([\n CardHeader(\"Detailed Feature Comparison\"),\n Table([\n Col(\"Provider\", [\"AWS\", \"Azure\", \"GCP\", \"OCI\", \"DigitalOcean\"]),\n Col(\"Regions\", [\"33+\", \"60+\", \"35+\", \"45+\", \"15+\"]),\n Col(\"Compliance\", [\"Full (HIPAA, PCI, SOC)\", \"Full (Heavy Gov/Fed)\", \"Strong (High Security)\", \"Strong\", \"Standard\"]),\n Col(\"Support\", [\"Tiered (Basic to Enterprise)\", \"Tiered (Enterprise focus)\", \"Tiered (Technical focus)\", \"High (Aggressive)\", \"Simple/Direct\"]),\n Col(\"Cold Start (FaaS)\", [\"Moderate\", \"Moderate\", \"Low (Fastest)\", \"Moderate\", \"N/A (Simpler)\"])\n ])\n])\n\npricingTable = Card([\n CardHeader(\"Economic Comparison\"),\n Table([\n Col(\"Provider\", [\"AWS\", \"Azure\", \"GCP\", \"OCI\", \"DigitalOcean\"]),\n Col(\"Free Tier\", [\"12 Months + Always Free\", \"12 Months + Always Free\", \"90-day trial + Always Free\", \"Always Free (Generous)\", \"Trial (Limited)\"]),\n Col(\"Egress Fees\", [\"High (Complex)\", \"High (Complex)\", \"Moderate\", \"Very Low\", \"Low/Included\"]),\n Col(\"Pricing Model\", [\"Granular/Complex\", \"Enterprise/Bundled\", \"Usage-based/Simple\", \"Aggressive/Low\", \"Flat/Predictable\"])\n ])\n])\n\nperformanceChart = Card([\n CardHeader(\"Cold Start Performance (ms)\"),\n HorizontalBarChart([\"AWS\", \"Azure\", \"GCP\", \"OCI\"], [250, 280, 120, 210], \"grouped\", \"Latency (Lower is better)\")\n])\n\nrecommendation = Card([\n CardHeader(\"The Verdict\"),\n Stack([\n TextContent(\"My Pick: **Google Cloud Platform (GCP)**\", \"large-heavy\"),\n TextContent(\"Why? For modern, data-driven, and container-first architectures, GCP offers the best balance of developer experience and technical performance. Their networking infrastructure and Kubernetes (GKE) implementation remain the industry gold standard. While AWS has more services, GCP's 'speed to market' and superior data/AI integration provide a significant competitive edge for scaling startups and high-performance compute workloads.\"),\n Separator(\"horizontal\"),\n TagBlock([\"Best for Scale: AWS\", \"Best for Enterprise: Azure\", \"Best for AI/ML: GCP\", \"Best for Value: OCI\", \"Best for Simplicity: DigitalOcean\"])\n ], \"column\", \"s\")\n], \"card\", \"column\", \"start\", \"stretch\")","promptTokens":4591,"completionTokens":969,"latencyMs":6537,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/adversarial","repeat":0,"output":"root = Stack([header, kpiSection, comparisonSection, recommendationSection], \"column\", \"l\")\n\nheader = Card([\n CardHeader(\"Cloud Provider Comparison 2024\", \"Deep dive into AWS, Azure, GCP, Oracle, and DigitalOcean\"),\n TextContent(\"A comprehensive technical and financial comparison of the top 5 cloud ecosystems.\")\n], \"sunk\")\n\nkpiSection = Stack([\n Card([\n CardHeader(\"Market Sentiment & Reliability\"),\n Stack([\n Card([\n TextContent(\"Highest Availability\", \"small\"),\n TextContent(\"AWS & Azure\", \"large-heavy\")\n ]),\n Card([\n TextContent(\"Best Developer Experience\", \"small\"),\n TextContent(\"GCP & DigitalOcean\", \"large-heavy\")\n ]),\n Card([\n TextContent(\"Best Value for Scale\", \"small\"),\n TextContent(\"Oracle Cloud\", \"large-heavy\")\n ])\n ], \"row\", \"m\", \"stretch\")\n ])\n], \"none\")\n\ncomparisonSection = Tabs([\n TabItem(\"specs\", \"Technical Specs\", [specsTable]),\n TabItem(\"finance\", \"Financials\", [financeTable]),\n TabItem(\"charts\", \"Market Trends\", [marketCharts])\n])\n\nspecsTable = Card([\n CardHeader(\"Technical Specifications\"),\n Table([\n Col(\"Provider\", providerNames),\n Col(\"Global Regions\", regionCounts),\n Col(\"Uptime SLA\", uptimeSLA),\n Col(\"Cold Start (Serverless)\", coldStarts),\n Col(\"Compliance\", complianceCertifications),\n Col(\"Support Tiers\", supportLevels)\n ])\n])\n\nfinanceTable = Card([\n CardHeader(\"Pricing & Cost Structure\"),\n Table([\n Col(\"Provider\", providerNames),\n Col(\"Free Tier\", freeTierInfo),\n Col(\"Egress Fees\", egressCost),\n Col(\"Pricing Model\", pricingModels)\n ])\n])\n\nmarketCharts = Card([\n CardHeader(\"Comparison Visualizations\"),\n Stack([\n BarChart([\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"DO\"], [\n Series(\"Market Share %\", [32, 23, 11, 3, 2])\n ], \"horizontal\"),\n HorizontalBarChart([\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"DO\"], [\n Series(\"Complexity Score (1-10)\", [9, 8, 6, 7, 3])\n ])\n ], \"column\", \"l\")\n])\n\nrecommendationSection = Card([\n CardHeader(\"Expert Recommendation\"),\n TextContent(\"Based on the data, the 'best' choice depends on your specific use case:\", \"default\"),\n Separator(\"horizontal\"),\n Stack([\n Card([\n TextContent(\"For Enterprise Scale\", \"small\"),\n TextContent(\"AWS or Azure. Their massive region footprint and compliance depth are unmatched for global corporations.\", \"default\")\n ]),\n Card([\n TextContent(\"For AI & Data Science\", \"small\"),\n TextContent(\"GCP. Their integration with TensorFlow and TPU hardware provides a superior developer workflow for ML.\", \"default\")\n ]),\n Card([\n TextContent(\"For Startups/SMEs\", \"small\"),\n TextContent(\"DigitalOcean. The simplicity, predictable pricing, and low egress fees allow for rapid scaling without 'bill shock'.\", \"default\")\n ])\n ], \"row\", \"m\", \"stretch\")\n], \"card\")\n\nproviderNames = [\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"DigitalOcean\"]\nregionCounts = [\"33+\", \"60+\", \"37+\", \"45+\", \"15+\"]\nuptimeSLA = [\"99.99%\", \"99.99%\", \"99.95%\", \"99.99%\", \"99.9%\"]\ncoldStarts = [\"Medium\", \"Medium\", \"Low\", \"High\", \"N/A\"]\ncomplianceCertifications = [\"HIPAA, PCI, SOC, FedRAMP\", \"HIPAA, PCI, SOC, FedRAMP\", \"HIPAA, PCI, SOC\", \"HIPAA, PCI, SOC\", \"PCI, SOC\"]\nsupportLevels = [\"Basic to Enterprise\", \"Basic to Enterprise\", \"Basic to Enterprise\", \"Basic to Enterprise\", \"Community/Paid\"]\n\nfreeTierInfo = [\"12 Months + Always Free\", \"12 Months + Always Free\", \"Always Free (Limited)\", \"Always Free (High Resource)\", \"Fixed Credit/Trial\"]\negressCost = [\"High/Variable\", \"High/Variable\", \"Moderate\", \"Low\", \"Very Low/Predictable\"]\npricingModels = [\"Complex/Granular\", \"Enterprise/Hybrid\", \"Resource-Centric\", \"Aggressive/Low\", \"Flat/Simple\"]\n\nmarketCharts = [\n BarChart([\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"DO\"], [Series(\"Market Share %\", [32, 23, 11, 3, 2])], \"horizontal\"),\n HorizontalBarChart([\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"DO\"], [Series(\"Complexity Score (1-10)\", [9, 8, 6, 7, 3])])\n]","promptTokens":4591,"completionTokens":1171,"latencyMs":26369,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/adversarial","repeat":3,"output":"root = Stack([header, summaryCard, comparisonTabs, recommendationCard], \"column\", \"l\")\n\nheader = Card([\n CardHeader(\"Cloud Provider Comparison\", \"Deep dive into AWS, Azure, GCP, Oracle, and DigitalOcean\"),\n TextContent(\"Comprehensive analysis of infrastructure, pricing, and performance metrics for 2024.\")\n], \"sunk\")\n\nsummaryCard = Card([\n CardHeader(\"Executive Summary\"),\n Stack([\n TextContent(\"The 'best' provider depends on your workload. Enterprise needs Azure/AWS, Startups prefer GCP/DigitalOcean, and Cost-Optimization favors Oracle.\"),\n Separator(\"horizontal\")\n ])\n], \"clear\")\n\ncomparisonTabs = Tabs([\n TabItem(\"specs\", \"Full Specifications\", [specTable]),\n TabItem(\"pricing\", \"Pricing & Free Tier\", [pricingTable]),\n TabItem(\"performance\", \"Performance & Reliability\", [performanceCharts])\n])\n\nspecTable = Table([\n Col(\"Provider\", providerNames),\n Col(\"Primary Focus\", primaryFocus),\n Col(\"Global Regions\", regionCounts),\n Col(\"Compliance\", complianceLevels),\n Col(\"Support Tiers\", supportModels)\n])\n\npricingTable = Table([\n Col(\"Provider\", providerNames),\n Col(\"Free Tier\", freeTierInfo),\n Col(\"Egress Fees\", egressCost),\n Col(\"Cold Start (Avg)\", coldStartTimes)\n])\n\nperformanceCharts = Stack([\n Card([\n CardHeader(\"Reliability Comparison (Uptime %)\"),\n BarChart([\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"DigitalOcean\"], [Series(\"Uptime\", [99.99, 99.95, 99.9, 99.99, 99.99])])\n ]),\n Card([\n CardHeader(\"Global Footprint (Region Count)\"),\n HorizontalBarChart([\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"DigitalOcean\"], [Series(\"Regions\", [33, 60, 40, 45, 32])])\n ])\n], \"column\", \"m\")\n\nrecommendationCard = Card([\n CardHeader(\"AI Recommendation\"),\n TextContent(\"Based on the current landscape, my pick is: **Google Cloud Platform (GCP)**\"),\n Separator(\"horizontal\"),\n TextContent(\"Reasoning: GCP offers the best balance of cutting-edge Kubernetes orchestration, superior data/AI integration, and a more intuitive developer experience compared to the legacy complexity of AWS or Azure. For cost-sensitive, simple deployments, DigitalOcean is the winner, but for scalable, intelligent applications, GCP is the sweet spot.\")\n], \"card\")\n\nproviderNames = [\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"DigitalOcean\"]\nprimaryFocus = [\"General Purpose / Scale\", \"Enterprise / Windows\", \"Data / AI / K8s\", \"Database / Cost\", \"Simplicity / Developer\"]\nregionCounts = [\"33+\", \"60+\", \"40+\", \"45+\", \"32+\"]\ncomplianceLevels = [\"Full (HIPAA/PCI/SOC)\", \"Full (Gov/Industry)\", \"High (Security-first)\", \"High (DB focused)\", \"Standard\"]\nsupportModels = [\"Basic to Enterprise\", \"Enterprise-centric\", \"Developer to Enterprise\", \"Tiered/DB-centric\", \"Community/Paid\"]\nfreeTierInfo = [\"12 Months / Always Free\", \"12 Months / Credit-based\", \"Always Free / Credits\", \"Trial Period\", \"Generous Credits\"]\negressCost = [\"High / Complex\", \"High / Complex\", \"Moderate\", \"Low\", \"Very Low\"]\ncoldStartTimes = [\"~200ms\", \"~250ms\", \"~150ms\", \"~300ms\", \"~100ms\"]","promptTokens":4590,"completionTokens":844,"latencyMs":13194,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/realistic","repeat":0,"output":"root = Stack([header, kpis, tabs], \"column\", \"m\")\n\nheader = Card([\n CardHeader(\"Q3 Sales Performance\", \"Regional Sales Team Results\")\n], \"sunk\")\n\nkpis = Stack([\n kpiTotalRev,\n kpiTotalDeals,\n kpiAvgQuotaAttainment\n], \"row\", \"m\")\n\nkpiTotalRev = Card([\n TextContent(\"Total Q3 Revenue\", \"small\"),\n TextContent(\"$2,480,000\", \"large-heavy\")\n])\n\nkpiTotalDeals = Card([\n TextContent(\"Deals Closed\", \"small\"),\n TextContent(\"42\", \"large-heavy\")\n])\n\nkpiAvgQuotaAttainment = Card([\n TextContent(\"Avg. Quota Attainment\", \"small\"),\n TextContent(\"104%\", \"large-heavy\")\n])\n\ntabs = Tabs([\n TabItem(\"emea\", \"EMEA Region\", [emeaTable]),\n TabItem(\"apac\", \"APAC Region\", [apacTable]),\n TabItem(\"all\", \"All Reps\", [allTable])\n])\n\nallTable = Card([\n CardHeader(\"All Sales Representatives\"),\n Table([\n Col(\"Rep Name\", data.rows.name),\n Col(\"Region\", data.rows.region),\n Col(\"Deals\", data.rows.deals),\n Col(\"Revenue\", data.rows.revenue),\n Col(\"Quota Attainment\", @Each(data.rows, \"r\", Tag(r.attainment + \"%\", null, \"sm\", r.attainment >= 100 ? \"success\" : \"warning\")))\n ])\n], \"clear\")\n\nemeaTable = Card([\n CardHeader(\"EMEA Team\"),\n Table([\n Col(\"Rep Name\", data.rows.name),\n Col(\"Deals\", data.rows.deals),\n Col(\"Revenue\", data.rows.revenue),\n Col(\"Quota Attainment\", @Each(data.rows, \"r\", Tag(r.attainment + \"%\", null, \"sm\", r.attainment >= 100 ? \"success\" : \"warning\")))\n ])\n], \"clear\")\n\napacTable = Card([\n CardHeader(\"APAC Team\"),\n Table([\n Col(\"Rep Name\", data.rows.name),\n Col(\"Deals\", data.rows.deals),\n Col(\"Revenue\", data.rows.revenue),\n Col(\"Quota Attainment\", @Each(data.rows, \"r\", Tag(r.attainment + \"%\", null, \"sm\", r.attainment >= 100 ? \"success\" : \"warning\")))\n ])\n], \"clear\")\n\ndata = {\n \"rows\": [\n {\"name\": \"Sarah Jenkins\", \"region\": \"EMEA\", \"deals\": 6, \"revenue\": 450000, \"attainment\": 112},\n {\"name\": \"Marc Dubois\", \"region\": \"EMEA\", \"deals\": 5, \"revenue\": 380000, \"attainment\": 95},\n {\"name\": \"Elena Rossi\", \"region\": \"EMEA\", \"deals\": 7, \"revenue\": 520000, \"attainment\": 128},\n {\"name\": \"Lars Nielsen\", \"region\": \"EMEA\", \"deals\": 4, \"revenue\": 290000, \"attainment\": 72},\n {\"name\": \"Kenji Sato\", \"region\": \"APAC\", \"deals\": 6, \"revenue\": 410000, \"attainment\": 102},\n {\"name\": \"Mei Lin\", \"region\": \"APAC\", \"deals\": 8, \"revenue\": 590000, \"attainment\": 145},\n {\"name\": \"Arjun Patel\", \"region\": \"APAC\", \"deals\": 5, \"revenue\": 320000, \"attainment\": 88},\n {\"name\": \"Chloe Tan\", \"region\": \"APAC\", \"deals\": 5, \"revenue\": 310000, \"attainment\": 78}\n]\n}","promptTokens":4578,"completionTokens":958,"latencyMs":59105,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/adversarial","repeat":4,"output":"root = Stack([header, dashboard, comparison_section, recommendation, comparison_table], \"column\", \"xl\")\n\nheader = Card([\n CardHeader(\"Cloud Provider Comparison Matrix\", \"Deep dive into AWS, Azure, GCP, Oracle, and IBM\"),\n TextContent(\"A comprehensive analysis of global cloud infrastructure across performance, cost, and reliability metrics.\")\n], \"sunk\")\n\ndashboard = Stack([\n Card([\n CardHeader(\"Market Share & Presence\", \"Global Reach Overview\"),\n RadialChart([\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"IBM\"], [33, 23, 11, 5, 3])\n ]),\n Card([\n CardHeader(\"Performance Benchmarks\", \"Cold Start Times (ms)\"),\n BarChart([\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"IBM\"], [Series(\"Lambda/Functions\", [250, 310, 220, 450, 500])])\n ])\n], \"row\", \"m\", \"center\", \"stretch\", true)\n\ncomparison_section = Tabs([\n TabItem(\"specs\", \"Technical Specs\", [spec_card]),\n TabItem(\"costs\", \"Pricing Models\", [cost_card]),\n TabItem(\"compliance\", \"Compliance & Trust\", [compliance_card])\n])\n\nspec_card = Card([\n CardHeader(\"Infrastructure & Performance\"),\n Table([\n Col(\"Provider\", [\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"IBM\"]),\n Col(\"Regions\", [\"30+\", \"60+\", \"35+\", \"45+\", \"20+\"]),\n Col(\"Uptime SLA\", [\"99.99%\", \"99.99%\", \"99.99%\", \"99.99%\", \"99.95%\"]),\n Col(\"Cold Start\", [\"Med\", \"Med\", \"Low\", \"High\", \"High\"])\n ])\n])\n\ncost_card = Card([\n CardHeader(\"Cost Structures\"),\n Table([\n Col(\"Provider\", [\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"IBM\"]),\n Col(\"Free Tier\", [\"12 Mo\", \"Always\", \"Always\", \"Limited\", \"Limited\"]),\n Col(\"Egress Fees\", [\"High\", \"High\", \"Med\", \"Low\", \"Med\"]),\n Col(\"Pricing Model\", [\"Complex\", \"Enterprise\", \"Granular\", \"Aggressive\", \"Fixed\"])\n ])\n])\n\ncompliance_card = Card([\n CardHeader(\"Compliance & Security\"),\n Table([\n Col(\"Provider\", [\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"IBM\"]),\n Col(\"Certifications\", [\"SOC/HIPAA/PCI\", \"SOC/HIPAA/PCI\", \"SOC/HIPAA/PCI\", \"SOC/HIPAA\", \"SOC/HIPAA\"]),\n Col(\"Support Tiers\", [\"4 Tiers\", \"4 Tiers\", \"3 Tiers\", \"3 Tiers\", \"2 Tiers\"])\n ])\n])\n\ncomparison_table = Card([\n CardHeader(\"Detailed Feature Breakdown\"),\n Table([\n Col(\"Feature\", [\"Compute\", \"Storage\", \"Network\", \"AI/ML\", \"Serverless\"]),\n Col(\"AWS\", [\"EC2\", \"S3\", \"Direct Connect\", \"SageMaker\", \"Lambda\"]),\n Col(\"Azure\", [\"VMs\", \"Blob\", \"ExpressRoute\", \"Azure ML\", \"Functions\"]),\n Col(\"GCP\", [\"GCE\", \"Cloud Storage\", \"Interconnect\", \"Vertex AI\", \"Cloud Functions\"]),\n Col(\"Oracle\", [\"OCI Compute\", \"Object Storage\", \"FastConnect\", \"OCI AI\", \"Functions\"]),\n Col(\"IBM\", [\"Bare Metal\", \"Cloud Object\", \"Direct Link\", \"Watson\", \"Code Engine\"])\n ])\n])\n\nrecommendation = Card([\n CardHeader(\"The AI Assistant's Choice\"),\n Stack([\n TextContent(\"Winner: Google Cloud Platform (GCP)\", \"large-heavy\"),\n TextContent(\"Why? For most modern, data-driven, and AI-centric applications, GCP offers the best balance of developer experience, superior networking, and industry-leading AI/ML integration (Vertex AI). While AWS has more services, GCP's 'opinionated' approach reduces operational complexity and provides more predictable performance for containerized workloads (Kubernetes/GKE).\"),\n TagBlock([\"Best for AI\", \"Best Networking\", \"Developer Friendly\"])\n ], \"column\", \"s\")\n], \"card\", \"column\", \"m\", \"start\", \"center\")","promptTokens":4591,"completionTokens":1016,"latencyMs":18711,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/realistic","repeat":3,"output":"root = Stack([dashboard_title, kpi_row, main_content], \"column\", \"xl\")\n\ndashboard_title = Card([\n CardHeader(\"Q3 Sales Performance\", \"Regional Results: EMEA & APAC\")\n], \"clear\")\n\nkpi_row = Stack([\n kpi_total_revenue,\n kpi_total_deals,\n kpi_avg_quota_attainment\n], \"row\", \"m\")\n\nkpi_total_revenue = Card([\n TextContent(\"Total Revenue\", \"small\"),\n TextContent(\"$1,420,000\", \"large-heavy\")\n], \"sunk\")\n\nkpi_total_deals = Card([\n TextContent(\"Deals Closed\", \"small\"),\n TextContent(\"42\", \"large-heavy\")\n], \"sunk\")\n\nkpi_avg_quota_attainment = Card([\n TextContent(\"Avg. Quota Attainment\", \"small\"),\n TextContent(\"94%\", \"large-heavy\")\n], \"sunk\")\n\nmain_content = Tabs([\n TabItem(\"table_view\", \"Detailed Table\", [sales_table]),\n TabItem(\"chart_view\", \"Revenue by Region\", [revenue_chart])\n])\n\nsales_data = [\n {id: 1, name: \"Sarah Jenkins\", region: \"EMEA\", deals: 6, revenue: 245000, quota: 220000},\n {id: 2, name: \"Marcus Chen\", region: \"APAC\", deals: 5, revenue: 185000, quota: 200000},\n {id: 3, name: \"Elena Rodriguez\", region: \"EMEA\", deals: 7, revenue: 310000, quota: 250000},\n {id: 4, name: \"Hiroshi Tanaka\", region: \"APAC\", deals: 4, revenue: 140000, quota: 180000},\n {id: 5, name: \"Liam O'Connor\", region: \"EMEA\", deals: 5, revenue: 195000, quota: 210000},\n {id: 6, name: \"Priya Sharma\", region: \"APAC\", deals: 8, revenue: 290000, quota: 230000},\n {id: 7, name: \"David Müller\", region: \"EMEA\", deals: 4, revenue: 120000, quota: 190000},\n {id: 8, name: \"Chloe Zhao\", region: \"APAC\", deals: 3, revenue: 135000, quota: 170000}\n]\n\nsales_table = Card([\n Table([\n Col(\"Rep Name\", sales_data.map(r => r.name)),\n Col(\"Region\", sales_data.map(r => r.region)),\n Col(\"Deals\", sales_data.map(r => r.deals), \"number\"),\n Col(\"Revenue\", sales_data.map(r => r.revenue), \"number\"),\n Col(\"Quota Attainment\", @Each(sales_data, \"r\", \n Tag(\n (r.revenue / r.quota * 100) + \"%\", \n null, \n \"sm\", \n (r.revenue / r.quota) >= 1.0 ? \"success\" : (r.revenue / r.quota) >= 0.8 ? \"warning\" : \"danger\"\n )\n ))\n ])\n])\n\nrevenue_chart = Card([\n CardHeader(\"Revenue Distribution\"),\n BarChart(\n [\"EMEA\", \"APAC\"], \n [\n Series(\"Revenue\", [\n @Sum(@Filter(sales_data, \"region\", \"==\", \"EMEA\"), \"revenue\"),\n @Sum(@Filter(sales_data, \"region\", \"==\", \"APAC\"), \"revenue\")\n ])\n ]\n )\n])","promptTokens":4578,"completionTokens":943,"latencyMs":57699,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/realistic","repeat":4,"output":"root = Stack([header, kpis, mainContent], \"column\", \"l\")\n\nheader = Card([CardHeader(\"Q3 Sales Performance\", \"Regional Breakdown: EMEA & APAC\")])\n\nkpiItems = [\n Card([TextContent(\"Total Revenue\", \"small\"), TextContent(\"$1,420,500\", \"large-heavy\")]),\n Card([TextContent(\"Deals Closed\", \"small\"), TextContent(\"42\", \"large-heavy\")]),\n Card([TextContent(\"Avg Quota Attainment\", \"small\"), TextContent(\"94%\", \"large-heavy\")])\n]\n\nkpis = Stack(kpiItems, \"row\", \"m\")\n\nmainContent = Tabs([\n TabItem(\"table\", \"Detailed View\", [salesTable]),\n TabItem(\"chart\", \"Revenue Distribution\", [revenueChart])\n])\n\nsalesData = [\n {id: 1, name: \"Sarah Jenkins\", region: \"EMEA\", deals: 6, revenue: 245000, quota: 220000},\n {id: 2, name: \"Marcus Chen\", region: \"APAC\", deals: 5, revenue: 185000, quota: 200000},\n {id: 3, name: \"Elena Rodriguez\", region: \"EMEA\", deals: 7, revenue: 310000, quota: 250000},\n {id: 4, name: \"Hiroshi Tanaka\", region: \"APAC\", deals: 4, revenue: 155000, quota: 180000},\n {id: 5, name: \"Alistair Cook\", region: \"EMEA\", deals: 5, revenue: 195000, quota: 210000},\n {id: 6, name: \"Priya Sharma\", region: \"APAC\", deals: 8, revenue: 325000, quota: 240000},\n {id: 7, name: \"Lars Nielsen\", region: \"EMEA\", deals: 3, revenue: 110000, quota: 190000},\n {id: 8, name: \"Ji-won Kim\", region: \"APAC\", deals: 6, revenue: 215000, quota: 200000}\n]\n\nsalesTable = Card([\n Table([\n Col(\"Rep Name\", salesData.map(r => r.name)),\n Col(\"Region\", salesData.map(r => r.region)),\n Col(\"Deals\", salesData.map(r => r.deals), \"number\"),\n Col(\"Revenue\", salesData.map(r => r.revenue), \"number\"),\n Col(\"Quota Attainment\", @Each(salesData, \"r\", Tag(\n (r.revenue / r.quota * 100).toFixed(1) + \"%\", \n null, \n \"sm\", \n (r.revenue / r.quota) >= 1 ? \"success\" : (r.revenue / r.quota >= 0.8 ? \"warning\" : \"danger\")\n )))\n ])\n])\n\nrevenueChart = Card([\n CardHeader(\"Revenue by Representative\"),\n BarChart(\n salesData.map(r => r.name), \n [Series(\"Revenue\", salesData.map(r => r.revenue))],\n \"grouped\",\n \"Sales Rep\",\n \"USD ($)\"\n )\n])","promptTokens":4578,"completionTokens":819,"latencyMs":50453,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/minimal","repeat":4,"output":"root = Stack([header, kpiRow, chartSection, tableSection], \"column\", \"l\")\n\nheader = Card([\n CardHeader(\"Revenue Analytics\", \"Monthly performance overview for the last 6 months\"),\n Separator(\"horizontal\")\n], \"sunk\")\n\nkpiRow = Stack([\n totalRevCard,\n avgRevCard,\n growthCard\n], \"row\", \"m\")\n\ntotalRevCard = Card([\n TextContent(\"Total Revenue\", \"small\"),\n TextContent(\"$42,500\", \"large-heavy\")\n], \"card\")\n\navgRevCard = Card([\n TextContent(\"Monthly Avg\", \"small\"),\n TextContent(\"$7,083\", \"large-heavy\")\n], \"card\")\n\ngrowthCard = Card([\n TextContent(\"Growth (MoM)\", \"small\"),\n TextContent(\"+12.5%\", \"large-heavy\")\n], \"card\")\n\nchartSection = Card([\n CardHeader(\"Revenue Trend\"),\n BarChart(months, [revenueSeries], \"grouped\", \"Month\", \"USD\")\n], \"card\")\n\ntableSection = Card([\n CardHeader(\"Detailed Breakdown\"),\n Table([\n Col(\"Month\", months),\n Col(\"Revenue\", revenues, \"number\"),\n Col(\"Status\", statuses, \"string\")\n ])\n], \"card\")\n\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"]\nrevenues = [5200, 6100, 5800, 7200, 8500, 9700]\nrevenueSeries = Series(\"Actual Revenue\", revenues)\nstatuses = [\"Stable\", \"Growth\", \"Dip\", \"Growth\", \"Surge\", \"Surge\"]","promptTokens":4546,"completionTokens":398,"latencyMs":11058,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/minimal","repeat":2,"output":"root = Stack([\n HeaderCard,\n StatsRow,\n ChartCard,\n DataCard\n], \"column\", \"l\")\n\nHeaderCard = Card([\n CardHeader(\"Revenue Analytics\", \"Performance overview for the last 6 months\"),\n Separator(\"horizontal\")\n], \"card\")\n\nStatsRow = Stack([\n TotalRevenueCard,\n AvgMonthlyCard,\n GrowthCard\n], \"row\", \"m\", \"start\", \"center\", true)\n\nTotalRevenueCard = Card([\n TextContent(\"Total Revenue\", \"small\"),\n TextContent(\"$42,500\", \"large-heavy\")\n], \"sunk\")\n\nAvgMonthlyCard = Card([\n TextContent(\"Avg Monthly\", \"small\"),\n TextContent(\"$7,083\", \"large-heavy\")\n], \"sunk\")\n\nGrowthCard = Card([\n TextContent(\"MoM Growth\", \"small\"),\n TextContent(\"+12.5%\", \"large-heavy\")\n], \"sunk\")\n\nChartCard = Card([\n CardHeader(\"Monthly Revenue Trend\"),\n RevenueBarChart\n], \"card\")\n\nRevenueBarChart = BarChart(\n [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"],\n [Series(\"Revenue\", [5200, 6100, 5800, 7200, 8500, 9700])],\n \"grouped\",\n \"Month\",\n \"Amount ($)\"\n)\n\nDataCard = Card([\n CardHeader(\"Detailed Breakdown\"),\n RevenueTable\n], \"card\")\n\nRevenueTable = Table([\n Col(\"Month\", [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"]),\n Col(\"Revenue\", [\"$5,200\", \"$6,100\", \"$5,800\", \"$7,200\", \"$8,500\", \"$9,700\"], \"number\"),\n Col(\"Status\", @Each([\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"], \"m\", Tag(m == \"Jun\" ? \"success\" : \"neutral\", null, \"sm\", m == \"Jun\" ? \"success\" : \"neutral\")))\n])","promptTokens":4546,"completionTokens":509,"latencyMs":15505,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/minimal","repeat":1,"output":"root = Stack([\n header,\n kpiRow,\n chartSection,\n tableSection\n], \"column\", \"xl\")\n\nheader = Card([\n CardHeader(\"Revenue Analytics\", \"Monthly performance overview for the last 6 months\")\n], \"clear\")\n\nkpiRow = Stack([\n totalRevenueCard,\n avgRevenueCard,\n growthCard\n], \"row\", \"m\")\n\ntotalRevenueCard = Card([\n TextContent(\"Total Revenue (6m)\", \"small\"),\n TextContent(\"$42,500\", \"large-heavy\")\n], \"sunk\")\n\navgRevenueCard = Card([\n TextContent(\"Avg. Monthly\", \"small\"),\n TextContent(\"$7,083\", \"large-heavy\")\n], \"sunk\")\n\ngrowthCard = Card([\n TextContent(\"Growth Trend\", \"small\"),\n Tag(\"Up 12%\", null, \"sm\", \"success\")\n], \"sunk\")\n\nchartSection = Card([\n CardHeader(\"Revenue Trend\", \"Monthly breakdown\"),\n revenueChart\n], \"card\")\n\nrevenueChart = BarChart(\n [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"],\n [Series(\"Actual Revenue\", [5200, 6100, 5800, 7200, 8500, 9700])]\n)\n\ntableSection = Card([\n CardHeader(\"Detailed Breakdown\"),\n revenueTable\n], \"card\")\n\nrevenueTable = Table([\n Col(\"Month\", [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"]),\n Col(\"Revenue\", [\"$5,200\", \"$6,100\", \"$5,800\", \"$7,200\", \"$8,500\", \"$9,700\"], \"number\"),\n Col(\"Status\", @Each([\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"], \"m\", Tag(m, null, \"sm\", m == \"Jun\" ? \"success\" : \"neutral\")))\n])","promptTokens":4546,"completionTokens":475,"latencyMs":30797,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/minimal","repeat":3,"output":"root = Stack([\n HeaderCard,\n KpiRow,\n ChartCard,\n TableCard\n], \"column\", \"l\")\n\nHeaderCard = Card([\n CardHeader(\"Revenue Analytics\", \"Monthly performance overview\")\n], \"sunk\")\n\nKpiRow = Stack([\n TotalRevenueCard,\n AvgMonthlyCard,\n GrowthCard\n], \"row\", \"m\", \"start\", \"evenly\", true)\n\nTotalRevenueCard = Card([\n TextContent(\"Total Revenue (6m)\", \"small\"),\n TextContent(\"$42,500\", \"large-heavy\")\n], \"card\")\n\nAvgMonthlyCard = Card([\n TextContent(\"Avg Monthly\", \"small\"),\n TextContent(\"$7,083\", \"large-heavy\")\n], \"card\")\n\nGrowthCard = Card([\n TextContent(\"MoM Growth\", \"small\"),\n TextContent(\"+12.5%\", \"large-heavy\")\n], \"card\")\n\nChartCard = Card([\n CardHeader(\"Revenue Trend\", \"Monthly breakdown\"),\n BarChart(\n [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"],\n [Series(\"Revenue\", [5200, 6100, 5800, 7200, 8500, 9700])]\n )\n], \"card\")\n\nTableCard = Card([\n CardHeader(\"Detailed Monthly Data\", \"Raw figures per month\"),\n Table([\n Col(\"Month\", [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"]),\n Col(\"Revenue\", [\"$5,200\", \"$6,100\", \"$5,800\", \"$7,200\", \"$8,500\", \"$9,700\"], \"number\"),\n Col(\"Status\", @Each([\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"], \"m\", Tag(\"m\", null, \"sm\", m == \"Jun\" ? \"success\" : \"neutral\")))\n ])\n], \"card\")","promptTokens":4546,"completionTokens":471,"latencyMs":29559,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/realistic","repeat":0,"output":"root = Stack([\n headerCard,\n kpiStack,\n chartCard,\n tableCard\n], \"column\", \"xl\")\n\nheaderCard = Card([\n CardHeader(\"Website Traffic Analysis\", \"Traffic sources for last month\"),\n TextContent(\"A detailed breakdown of how users discovered your website across different channels.\")\n])\n\nkpiStack = Stack([\n totalVisitorsCard,\n conversionCard,\n avgSessionCard\n], \"row\", \"m\")\n\ntotalVisitorsCard = Card([\n TextContent(\"Total Visitors\", \"small\"),\n TextContent(\"12,482\", \"large-heavy\")\n], \"s\")\n\nconversionCard = Card([\n TextContent(\"Conversion Rate\", \"small\"),\n TextContent(\"3.2%\", \"large-heavy\")\n], \"s\")\n\navgSessionCard = Card([\n TextContent(\"Avg. Session\", \"small\"),\n TextContent(\"2m 45s\", \"large-heavy\")\n], \"s\")\n\nchartCard = Card([\n CardHeader(\"Traffic Distribution\"),\n Tabs([\n TabItem(\"pie\", \"Pie Chart\", [\n Stack([\n PieChart(trafficLabels, trafficValues, \"donut\"),\n Stack([\n TrafficLegendItem(\"Organic Search\", \"#4F46E5\"),\n TrafficLegendItem(\"Paid Ads\", \"#10B981\"),\n TrafficLegendItem(\"Social Media\", \"#F59E0B\"),\n TrafficLegendItem(\"Direct\", \"#3B82F6\"),\n TrafficLegendItem(\"Referrals\", \"#8B5CF6\")\n ], \"column\", \"xs\")\n ], \"row\", \"xl\", \"center\")\n ]),\n TabItem(\"bar\", \"Bar Chart\", [\n BarChart(trafficLabels, [Series(\"Visitors\", trafficValues)], \"grouped\", \"Source\", \"Count\")\n ])\n ])\n])\n\ntableCard = Card([\n CardHeader(\"Source Breakdown\"),\n Table([\n Col(\"Source\", trafficLabels),\n Col(\"Visitors\", trafficCounts, \"number\"),\n Col(\"Percentage\", trafficPercentages, \"string\"),\n Col(\"Trend\", trafficTrends)\n ])\n])\n\ntrafficLabels = [\"Organic Search\", \"Paid Ads\", \"Social Media\", \"Direct\", \"Referrals\"]\ntrafficValues = [5617, 2496, 1872, 1747, 750]\ntrafficCounts = [5617, 2496, 1872, 1747, 750]\ntrafficPercentages = [\"45%\", \"20%\", \"15%\", \"14%\", \"6%\"]\ntrafficTrends = @Each(trafficValues, \"v\", Tag(v > 2000 ? \"↑\" : \"↓\", null, \"sm\", v > 2000 ? \"success\" : \"neutral\"))\n\nTrafficLegendItem = (label, color) => Stack([\n TextContent(label, \"small\"),\n TextContent(color, \"small\") // Placeholder for visual representation\n], \"row\", \"xs\", \"center\")\n\n// Note: In a real environment, the legend would use colored circles/divs. \n// Here we represent the structure.\n\n// Re-defining Legend for valid syntax\nlegendItem1 = Stack([TextContent(\"Organic Search\", \"small\"), TextContent(\"45%\", \"small\")], \"row\", \"xs\")\nlegendItem2 = Stack([TextContent(\"Paid Ads\", \"small\"), TextContent(\"20%\", \"small\")], \"row\", \"xs\")\nlegendItem3 = Stack([TextContent(\"Social Media\", \"small\"), TextContent(\"15%\", \"small\")], \"row\", \"xs\")\nlegendItem4 = Stack([TextContent(\"Direct\", \"small\"), TextContent(\"14%\", \"small\")], \"row\", \"xs\")\nlegendItem5 = Stack([TextContent(\"Referrals\", \"small\"), TextContent(\"6%\", \"small\")], \"row\", \"xs\")\n\nTrafficLegendItem = (label, color) => Stack([TextContent(label, \"small\"), TextContent(\"●\", \"small\")], \"row\", \"xs\") \n\n// Finalizing the Legend list for the chartCard\nlegendList = [legendItem1, legendItem2, legendItem3, legendItem4, legendItem5]\n\n// Correcting chartCard to use the list\nchartCard = Card([\n CardHeader(\"Traffic Distribution\"),\n Tabs([\n TabItem(\"pie\", \"Pie Chart\", [\n Stack([\n PieChart(trafficLabels, trafficValues, \"donut\"),\n legendList\n ], \"row\", \"xl\", \"center\")\n ]),\n TabItem(\"bar\", \"Bar Chart\", [\n BarChart(trafficLabels, [Series(\"Visitors\", trafficValues)], \"grouped\", \"Source\", \"Count\")\n ])\n ])\n])","promptTokens":4567,"completionTokens":1088,"latencyMs":19462,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/realistic","repeat":3,"output":"root = Stack([\n headerCard,\n kpiSection,\n chartSection,\n tableSection\n], \"column\", \"l\")\n\nheaderCard = Card([\n CardHeader(\"Website Traffic Analysis\", \"Monthly Breakdown: Last Month\"),\n TextContent(\"A detailed overview of your acquisition channels and user engagement metrics.\")\n], \"sunk\")\n\nkpiSection = Stack([\n totalVisitsCard,\n conversionCard,\n bounceRateCard\n], \"row\", \"m\", \"start\", \"center\", true)\n\ntotalVisitsCard = Card([\n TextContent(\"Total Visits\", \"small\"),\n TextContent(\"42,850\", \"large-heavy\")\n], \"card\")\n\nconversionCard = Card([\n TextContent(\"Avg. Conversion\", \"small\"),\n TextContent(\"3.2%\", \"large-heavy\")\n], \"card\")\n\nbounceRateCard = Card([\n TextContent(\"Bounce Rate\", \"small\"),\n TextContent(\"44.1%\", \"large-heavy\")\n], \"card\")\n\nchartSection = Card([\n CardHeader(\"Traffic Source Distribution\"),\n Tabs([\n TabItem(\"pie\", \"Pie Chart\", [trafficPieChart]),\n TabItem(\"bar\", \"Bar Chart\", [trafficBarChart])\n ])\n], \"card\")\n\ntrafficData = [\n {\"label\": \"Organic Search\", \"value\": 18500},\n {\"label\": \"Paid Ads\", \"value\": 9200},\n {\"label\": \"Social Media\", \"value\": 6400},\n {\"label\": \"Direct\", \"value\": 5150},\n {\"label\": \"Referrals\", \"value\": 3600}\n]\n\ntrafficLabels = [\"Organic Search\", \"Paid Ads\", \"Social Media\", \"Direct\", \"Referrals\"]\ntrafficValues = [18500, 9200, 6400, 5150, 3600]\n\ntrafficPieChart = PieChart(trafficLabels, trafficValues, \"donut\")\n\ntrafficBarChart = BarChart(trafficLabels, [Series(\"Visits\", trafficValues)], \"grouped\", \"Channel\", \"Number of Visits\")\n\ntableSection = Card([\n CardHeader(\"Channel Detail Table\"),\n Table([\n Col(\"Source\", trafficLabels),\n Col(\"Visits\", trafficValues, \"number\"),\n Col(\"Share\", @Each(trafficValues, \"v\", TextContent((\"\" + (v / 42850 * 100)).substring(0, 5) + \"%\", \"small\")))\n ])\n], \"card\")","promptTokens":4567,"completionTokens":593,"latencyMs":9202,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/minimal","repeat":0,"output":"root = Stack([\n headerCard,\n kpiRow,\n chartCard,\n tableCard\n], \"column\", \"m\")\n\nheaderCard = Card([\n CardHeader(\"Revenue Analytics\", \"Performance overview for the last 6 months\")\n], \"sunk\")\n\nkpiRow = Stack([\n totalRevenueKpi,\n avgMonthlyKpi,\n growthKpi\n], \"row\", \"m\")\n\ntotalRevenueKpi = Card([\n TextContent(\"Total Revenue\", \"small\"),\n TextContent(\"$42,500\", \"large-heavy\")\n])\n\navgMonthlyKpi = Card([\n TextContent(\"Avg Monthly\", \"small\"),\n TextContent(\"$7,083\", \"large-heavy\")\n])\n\ngrowthKpi = Card([\n TextContent(\"Growth (6mo)\", \"small\"),\n TextContent(\"+12.5%\", \"large-heavy\")\n])\n\nchartCard = Card([\n CardHeader(\"Monthly Revenue Trend\"),\n BarChart(\n [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"],\n [Series(\"Revenue\", [5200, 6100, 5800, 7200, 8500, 9700])]\n )\n])\n\ntableCard = Card([\n CardHeader(\"Detailed Breakdown\"),\n Table([\n Col(\"Month\", [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"]),\n Col(\"Revenue\", [5200, 6100, 5800, 7200, 8500, 9700], \"number\"),\n Col(\"Status\", @Each([\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\"], \"m\", Tag(m, null, \"sm\", m == \"Jun\" ? \"success\" : \"neutral\")))\n ])\n])","promptTokens":4546,"completionTokens":440,"latencyMs":62203,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/realistic","repeat":2,"output":"root = Stack([\n headerCard,\n kpiRow,\n contentArea\n], \"column\", \"l\")\n\nheaderCard = Card([\n CardHeader(\"Website Traffic Breakdown\", \"Last Month Performance Analysis\"),\n TextContent(\"A detailed look at your acquisition channels to help optimize marketing spend and SEO strategy.\")\n], \"sunk\")\n\nkpiRow = Stack([\n kpiCard(\"Total Visits\", \"124,502\"),\n kpiCard(\"Conversion Rate\", \"3.2%\"),\n kpiCard(\"Avg. Session\", \"2m 45s\")\n], \"row\", \"m\", \"start\", \"center\", true)\n\nkpiCard = Card([\n TextContent(\"Label\", \"small\"),\n TextContent(\"Value\", \"large-heavy\")\n], \"s\", \"start\", \"center\")\n\ncontentArea = Stack([\n Tabs([\n TabItem(\"visual\", \"Visual Overview\", [chartCard]),\n TabItem(\"data\", \"Detailed Data\", [tableCard])\n ])\n], \"none\")\n\nchartCard = Card([\n CardHeader(\"Traffic Distribution\"),\n PieChart([\"Organic Search\", \"Paid Ads\", \"Social\", \"Direct\", \"Referrals\"], [42, 25, 15, 10, 8], \"donut\")\n])\n\ntableCard = Card([\n CardHeader(\"Channel Breakdown\"),\n Table([\n Col(\"Channel\", [\"Organic Search\", \"Paid Ads\", \"Social\", \"Direct\", \"Referrals\"]),\n Col(\"Visits\", [52291, 31125, 18675, 12450, 9961], \"number\"),\n Col(\"Percentage\", [\"42%\", \"25%\", \"15%\", \"10%\", \"8%\"]),\n Col(\"Trend\", [@Each([\"up\", \"down\", \"up\", \"neutral\", \"up\"], \"t\", Tag(t, null, \"sm\", t == \"up\" ? \"success\" : t == \"down\" ? \"danger\" : \"neutral\"))])\n ])\n])","promptTokens":4567,"completionTokens":474,"latencyMs":28879,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/adversarial","repeat":2,"output":"root = Stack([header, kpiRow, mainContent, dataSection], \"column\", \"l\")\n\nheader = Card([CardHeader(\"Monthly Performance Overview\", \"Revenue/Metric Analysis for H1 2024\")])\n\nkpiRow = Stack([\n totalCard,\n avgCard,\n growthCard\n], \"row\", \"m\")\n\ntotalCard = Card([\n TextContent(\"Total (Jan-Jul)\", \"small\"),\n TextContent(\"38,200\", \"large-heavy\")\n])\n\navgCard = Card([\n TextContent(\"Monthly Average\", \"small\"),\n TextContent(\"5,457\", \"large-heavy\")\n])\n\ngrowthCard = Card([\n TextContent(\"Peak Month\", \"small\"),\n TextContent(\"Jun (7,100)\", \"large-heavy\")\n])\n\nmainContent = Tabs([\n chartTab,\n summaryTab\n])\n\nchartTab = TabItem(\"chart\", \"Trend Chart\", [\n Card([\n CardHeader(\"Monthly Trend\"),\n LineChart([\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"], [Series(\"Value\", [4200, 3800, 5100, 4900, 6300, 7100, 6800])])\n ])\n])\n\nsummaryTab = TabItem(\"summary\", \"Bar Comparison\", [\n Card([\n CardHeader(\"Monthly Comparison\"),\n BarChart([\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"], [Series(\"Value\", [4200, 3800, 5100, 4900, 6300, 7100, 6800])])\n ])\n])\n\ndataSection = Card([\n CardHeader(\"Raw Data Verification\"),\n Table([\n Col(\"Month\", [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]),\n Col(\"Value\", [4200, 3800, 5100, 4900, 6300, 7100, 6800], \"number\")\n ])\n])","promptTokens":4600,"completionTokens":517,"latencyMs":5964,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/realistic","repeat":1,"output":"root = Stack([header, kpiRow, chartsSection, tableSection], \"column\", \"l\")\n\nheader = Card([CardHeader(\"Website Traffic Breakdown\", \"Performance report for last month\")])\n\nkpiRow = Stack([totalTrafficCard, conversionCard, bounceRateCard], \"row\", \"m\", \"start\", \"stretch\", true)\n\ntotalTrafficCard = Card([TextContent(\"Total Sessions\", \"small\"), TextContent(\"42,850\", \"large-heavy\")])\nconversionCard = Card([TextContent(\"Avg. Conversion Rate\", \"small\"), TextContent(\"3.2%\", \"large-heavy\")])\nbounceRateCard = Card([TextContent(\"Bounce Rate\", \"small\"), TextContent(\"44.1%\", \"large-heavy\")])\n\nchartsSection = Stack([pieCard, barCard], \"row\", \"m\", \"start\", \"stretch\", true)\n\ntrafficData = [\n { \"source\": \"Organic Search\", \"value\": 18500 },\n { \"source\": \"Paid Ads\", \"value\": 10200 },\n { \"source\": \"Direct\", \"value\": 7400 },\n { \"source\": \"Social\", \"value\": 4250 },\n { \"source\": \"Referrals\", \"value\": 2500 }\n]\n\npieCard = Card([\n CardHeader(\"Traffic Distribution (%)\"),\n PieChart(\n trafficData.map(d => d.source), \n trafficData.map(d => d.value), \n \"donut\"\n )\n])\n\nbarCard = Card([\n CardHeader(\"Sessions by Source\"),\n HorizontalBarChart(\n trafficData.map(d => d.source),\n [Series(\"Sessions\", trafficData.map(d => d.value))],\n \"grouped\"\n )\n])\n\ntableSection = Card([\n CardHeader(\"Detailed Source Data\"),\n Table([\n Col(\"Source\", trafficData.map(d => d.source)),\n Col(\"Sessions\", trafficData.map(d => d.value), \"number\"),\n Col(\"Percentage\", trafficData.map(d => ((d.value / 42850) * 100).toFixed(1) + \"%\")),\n Col(\"Status\", @Each(trafficData, \"t\", Tag(t.value > 10000 ? \"high\" : \"normal\", null, \"sm\", t.value > 10000 ? \"success\" : \"neutral\")))\n ])\n])","promptTokens":4567,"completionTokens":572,"latencyMs":40987,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/adversarial","repeat":1,"output":"root = Stack([header, kpiSection, chartSection, dataSection], \"column\", \"l\")\n\nheader = Card([CardHeader(\"Monthly Revenue Performance\", \"Yearly Overview\")])\n\nkpiSection = Stack([totalRevenueCard, avgRevenueCard], \"row\", \"m\")\n\ntotalRevenueCard = Card([\n TextContent(\"Total Revenue\", \"small\"),\n TextContent(\"$48,200\", \"large-heavy\")\n], \"s\")\n\navgRevenueCard = Card([\n TextContent(\"Monthly Average\", \"small\"),\n TextContent(\"$6,885\", \"large-heavy\")\n], \"s\")\n\nchartSection = Card([\n CardHeader(\"Revenue Trends\"),\n Tabs([\n TabItem(\"line\", \"Line Chart\", [lineChart]),\n TabItem(\"bar\", \"Bar Chart\", [barChart])\n ])\n], \"s\")\n\nlineChart = LineChart(months, [Series(\"Revenue\", revenues)], \"natural\", \"Month\", \"USD\")\n\nbarChart = BarChart(months, [Series(\"Revenue\", revenues)], \"grouped\", \"Month\", \"USD\")\n\ndataSection = Card([\n CardHeader(\"Raw Data Verification\"),\n Table([\n Col(\"Month\", months),\n Col(\"Revenue ($)\", revenues, \"number\")\n ])\n], \"s\")\n\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nrevenues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]","promptTokens":4601,"completionTokens":364,"latencyMs":7923,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/adversarial","repeat":0,"output":"root = Stack([header, kpis, charts, rawData], \"column\", \"l\")\n\nheader = Card([CardHeader(\"Monthly Performance Overview\", \"Data tracking from January to July\")], \"sunk\")\n\nkpi_values = [4200, 3800, 5100, 4900, 6300, 7100, 6800]\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\ntotal_revenue = @Sum(kpi_values)\navg_revenue = @Average(kpi_values)\n\nkpis = Stack([\n Card([\n TextContent(\"Total Revenue\", \"small\"),\n TextContent(\"$\" + \"\" + total_revenue, \"large-heavy\")\n ]),\n Card([\n TextContent(\"Monthly Average\", \"small\"),\n TextContent(\"$\" + \"\" + avg_revenue, \"large-heavy\")\n ]),\n Card([\n TextContent(\"Peak Month\", \"small\"),\n TextContent(\"Jun\", \"large-heavy\")\n ])\n], \"row\", \"m\")\n\ncharts = Tabs([\n TabItem(\"trend\", \"Trend Line\", [\n Card([\n CardHeader(\"Revenue Trend Over Time\"),\n LineChart(months, [Series(\"Revenue\", kpi_values)], \"natural\")\n ])\n ]),\n TabItem(\"dist\", \"Distribution\", [\n Card([\n CardHeader(\"Monthly Comparison\"),\n BarChart(months, [Series(\"Revenue\", kpi_values)], \"grouped\")\n ])\n ])\n])\n\ndataTable = Table([\n Col(\"Month\", months),\n Col(\"Revenue ($)\", kpi_values, \"number\")\n])\n\nrawData = Card([\n CardHeader(\"Raw Data Verification\"),\n dataTable\n])","promptTokens":4600,"completionTokens":416,"latencyMs":16269,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/minimal","repeat":0,"output":"root = Stack([\n headerCard,\n warningCallout,\n actionCard\n], \"column\", \"l\", \"center\")\n\nheaderCard = Card([\n CardHeader(\"Subscription Status\", \"Account: Premium Trial\"),\n TextContent(\"Your current plan is nearing its end. Please upgrade to maintain uninterrupted access to all features.\")\n], \"sunk\")\n\nwarningCallout = Callout(\"warning\", \"Trial Expiring Soon\", \"Your trial period will expire in 3 days. Upgrade now to avoid service interruption.\", true)\n\nactionCard = Card([\n TextContent(\"Choose a plan to continue\", \"large-heavy\"),\n Stack([\n upgradeBtn,\n settingsBtn\n ], \"row\", \"m\", \"center\")\n], \"card\")\n\nupgradeBtn = Button(\"Upgrade Now\", Action([@ToAssistant(\"I want to upgrade my subscription\")]), \"primary\", \"medium\")\n\nsettingsBtn = Button(\"Manage Billing\", Action([@OpenUrl(\"https://example.com/billing\")]), \"secondary\", \"medium\")","promptTokens":4543,"completionTokens":225,"latencyMs":5234,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/adversarial","repeat":4,"output":"root = Stack([header, kpiRow, charts, rawDataTable], \"column\", \"l\")\n\nheader = Card([\n CardHeader(\"Monthly Performance Overview\", \"Revenue/Metric analysis for the first half of the year\")\n], \"sunk\")\n\nkpiRow = Stack([\n totalCard,\n avgCard,\n peakCard\n], \"row\", \"m\")\n\ntotalVal = 38200\navgVal = 5457\npeakVal = 7100\n\ntotalCard = Card([\n TextContent(\"Total Volume\", \"small\"),\n TextContent(\"\" + totalVal, \"large-heavy\")\n])\n\navgCard = Card([\n TextContent(\"Monthly Average\", \"small\"),\n TextContent(\"\" + avgVal, \"large-heavy\")\n])\n\npeakCard = Card([\n TextContent(\"Peak Month\", \"small\"),\n TextContent(\"Jun (\" + peakVal + \")\", \"large-heavy\")\n])\n\nmonths = [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]\nvalues = [4200, 3800, 5100, 4900, 6300, 7100, 6800]\n\ncharts = Stack([\n Card([\n CardHeader(\"Trend Analysis\"),\n Tabs([\n TabItem(\"line\", \"Line Chart\", [LineChart(months, [Series(\"Value\", values)], \"linear\")]),\n TabItem(\"bar\", \"Bar Chart\", [BarChart(months, [Series(\"Value\", values)])])\n ])\n ]),\n Card([\n CardHeader(\"Distribution\"),\n PieChart(months, values, \"donut\")\n ])\n], \"row\", \"m\", wrap=true)\n\ndataRows = [\n {\"month\": \"Jan\", \"value\": 4200},\n {\"month\": \"Feb\", \"value\": 3800},\n {\"month\": \"Mar\", \"value\": 5100},\n {\"month\": \"Apr\", \"value\": 4900},\n {\"month\": \"May\", \"value\": 6300},\n {\"month\": \"Jun\", \"value\": 7100},\n {\"month\": \"Jul\", \"value\": 6800}\n]\n\nrawDataTable = Card([\n CardHeader(\"Raw Data Verification\"),\n Table([\n Col(\"Month\", dataRows.month),\n Col(\"Value\", dataRows.value, \"number\")\n ])\n])","promptTokens":4600,"completionTokens":576,"latencyMs":6331,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/realistic","repeat":4,"output":"root = Stack([header, kpiRow, chartSection, tableSection], \"column\", \"xl\")\n\nheader = Card([\n CardHeader(\"Website Traffic Analysis\", \"Last Month's Acquisition Channels\"),\n TextContent(\"A detailed breakdown of how users discovered your website across different acquisition channels.\")\n], \"sunk\")\n\nkpiRow = Stack([\n totalVisitsCard,\n conversionCard,\n bounceRateCard\n], \"row\", \"m\", \"start\", \"center\", true)\n\ntotalVisitsCard = Card([\n TextContent(\"Total Visits\", \"small\"),\n TextContent(\"42,850\", \"large-heavy\")\n], \"card\")\n\nconversionCard = Card([\n TextContent(\"Avg. Conversion Rate\", \"small\"),\n TextContent(\"3.2%\", \"large-heavy\")\n], \"card\")\n\nbounceRateCard = Card([\n TextContent(\"Bounce Rate\", \"small\"),\n TextContent(\"44.1%\", \"large-heavy\")\n], \"card\")\n\nchartSection = Card([\n CardHeader(\"Traffic Distribution\", \"Percentage of total visits\"),\n Tabs([\n TabItem(\"pie\", \"Pie Chart\", [trafficPieChart]),\n TabItem(\"radial\", \"Radial View\", [trafficRadialChart])\n ])\n], \"card\")\n\ntrafficPieChart = PieChart([\"Organic Search\", \"Paid Ads\", \"Social\", \"Direct\", \"Referrals\"], [45, 25, 10, 15, 5], \"donut\")\n\ntrafficRadialChart = RadialChart([\"Organic\", \"Paid\", \"Social\", \"Direct\", \"Referral\"], [45, 25, 10, 15, 5])\n\ntableSection = Card([\n CardHeader(\"Channel Details\", \"Raw volume and performance\"),\n Table([\n Col(\"Channel\", [\"Organic Search\", \"Paid Ads\", \"Social\", \"Direct\", \"Referrals\"]),\n Col(\"Visits\", [19282, 10712, 4285, 6427, 2144], \"number\"),\n Col(\"Trend\", [\n Tag(\"↑ 12%\", null, \"sm\", \"success\"),\n Tag(\"↓ 2%\", null, \"sm\", \"danger\"),\n Tag(\"→ 0%\", null, \"sm\", \"neutral\"),\n Tag(\"↑ 5%\", null, \"sm\", \"success\"),\n Tag(\"↑ 8%\", null, \"sm\", \"success\")\n ])\n ])\n], \"card\")","promptTokens":4567,"completionTokens":569,"latencyMs":35175,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/minimal","repeat":3,"output":"root = Stack([\n warningCallout,\n mainContent\n], \"column\", \"m\")\n\nwarningCallout = Callout(\"warning\", \"Trial Expiration Warning\", \"Your premium trial period is scheduled to expire in 3 days. Please upgrade to a subscription plan to avoid any service interruption.\", true)\n\nmainContent = Card([\n CardHeader(\"Account Status\", \"Subscription: Trial Tier\"),\n Separator(\"horizontal\"),\n Stack([\n statusCard,\n actionCard\n ], \"row\", \"m\", \"stretch\")\n], \"card\")\n\nstatusCard = Card([\n TextContent(\"Trial Remaining\", \"small\"),\n TextContent(\"3 Days\", \"large-heavy\"),\n Separator(\"horizontal\"),\n TagBlock([\"Limited Access\", \"Auto-expires soon\"])\n], \"sunk\")\n\nactionCard = Card([\n TextContent(\"Take Action\", \"small\"),\n TextContent(\"Upgrade now to keep your current settings and unlock all features.\", \"default\"),\n Buttons([\n upgradeBtn,\n viewPlansBtn\n ], \"row\")\n], \"sunk\")\n\nupgradeBtn = Button(\"Upgrade Now\", Action([@ToAssistant(\"I want to upgrade my subscription immediately. Please show me the pricing plans.\")]), \"primary\")\nviewPlansBtn = Button(\"View Plans\", Action([@ToAssistant(\"Show me the different subscription plans available.\")]), \"secondary\")","promptTokens":4543,"completionTokens":304,"latencyMs":10130,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/minimal","repeat":2,"output":"root = Stack([\n warningCallout,\n contentCard\n], \"column\", \"m\")\n\nwarningCallout = Callout(\"warning\", \"Trial Period Expiring\", \"Your premium trial will expire in 3 days. Upgrade now to maintain uninterrupted access to all features.\", true)\n\ncontentCard = Card([\n CardHeader(\"Subscription Status\", \"Trial Plan\"),\n Separator(\"horizontal\"),\n Stack([\n TextContent(\"Days Remaining\", \"small\"),\n TextContent(\"3\", \"large-heavy\")\n ], \"row\", \"m\", \"center\", \"start\"),\n Separator(\"horizontal\"),\n Stack([\n upgradeBtn,\n viewPlanBtn\n ], \"row\", \"m\", \"center\", \"end\")\n], \"card\", \"column\", \"m\")\n\nupgradeBtn = Button(\"Upgrade Now\", Action([@ToAssistant(\"I want to upgrade my subscription\")]), \"primary\")\n\nviewPlanBtn = Button(\"View Plans\", Action([@ToAssistant(\"Show me the subscription plans\")]), \"secondary\")","promptTokens":4543,"completionTokens":224,"latencyMs":15033,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/realistic","repeat":1,"output":"root = Stack([\n maintenanceHeader,\n maintenanceStatus,\n impactDetails,\n customerCommunication\n], \"column\", \"xl\")\n\nmaintenanceHeader = Card([\n CardHeader(\"Scheduled System Maintenance\", \"Upcoming Service Window\"),\n TextCallout(\"info\", \"Planned Maintenance\", \"We are performing routine updates to improve our platform.\")\n], \"sunk\")\n\nmaintenanceStatus = Card([\n Stack([\n TextContent(\"Maintenance Window\", \"large-heavy\"),\n TextContent(\"Sunday, 02:00 - 06:00 UTC\", \"default\")\n ], \"row\", \"m\", \"between\")\n], \"card\")\n\nimpactDetails = Card([\n CardHeader(\"Service Impact\"),\n TextContent(\"During this window, certain services will be temporarily unavailable. All other platform features will remain fully operational.\"),\n Separator(\"horizontal\"),\n Stack([\n impactCard1,\n impactCard2,\n impactCard3\n ], \"row\", \"m\", \"start\", \"stretch\", true)\n], \"card\")\n\nimpactCard1 = Card([\n Tag(\"unavailable\", null, \"sm\", \"danger\"),\n TextContent(\"Payments\", \"large-heavy\"),\n TextContent(\"Processing and payouts will be paused.\")\n], \"sunk\", \"column\", \"m\")\n\nimpactCard2 = Card([\n Tag(\"unavailable\", null, \"sm\", \"danger\"),\n TextContent(\"Exports\", \"large-heavy\"),\n TextContent(\"Data exports and reporting will be disabled.\")\n], \"sunk\", \"column\", \"m\")\n\nimpactCard3 = Card([\n Tag(\"available\", null, \"sm\", \"success\"),\n TextContent(\"Core Platform\", \"large-heavy\"),\n TextContent(\"Dashboard and browsing remain active.\")\n], \"sunk\", \"column\", \"m\")\n\ncustomerCommunication = Card([\n CardHeader(\"Customer Communication Templates\"),\n TextContent(\"Use the following templates to notify your users via email or in-app notifications.\"),\n Tabs([\n emailTab,\n inAppTab\n ])\n], \"card\")\n\nemailTab = TabItem(\"email\", \"Email Template\", [\n Card([\n TextContent(\"Subject: Scheduled Maintenance Notification\", \"small-heavy\"),\n Separator(),\n MarkDownRenderer(\"Hello,\\n\\nTo ensure the continued stability and performance of our services, we have scheduled routine maintenance for **this Sunday between 02:00 and 06:00 UTC**.\\n\\n**What to expect:**\\n- **Payments:** Will be temporarily unavailable.\\n- **Data Exports:** Will be temporarily unavailable.\\n- **All other features:** Will continue to work as normal.\\n\\nThis is a planned update and not an unexpected outage. We apologize for any inconvenience this may cause.\\n\\nBest regards,\\nThe Team\")\n ], \"clear\")\n])\n\ninAppTab = TabItem(\"inapp\", \"In-App Banner\", [\n Card([\n TextContent(\"Banner Text:\", \"small-heavy\"),\n CodeBlock(\"markdown\", \"⚠️ **Scheduled Maintenance:** Payments and Exports will be unavailable this Sunday (02:00-06:00 UTC). All other services remain active.\")\n ], \"clear\")\n])\n\nmaintenanceStatus = Card([\n Stack([\n TextContent(\"Maintenance Window\", \"large-heavy\"),\n TextContent(\"Sunday, 02:00 - 06:00 UTC\", \"default\")\n ], \"row\", \"m\", \"between\")\n], \"card\")","promptTokens":4578,"completionTokens":783,"latencyMs":8622,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/minimal","repeat":4,"output":"root = Stack([\n warningCallout,\n contentStack\n], \"column\", \"m\")\n\nwarningCallout = Callout(\"warning\", \"Trial Expiring Soon\", \"Your free trial period will expire in 3 days. Please upgrade your plan to avoid any service interruption.\", true)\n\ncontentStack = Stack([\n headerCard,\n actionCard\n], \"column\", \"m\")\n\nheaderCard = Card([\n CardHeader(\"Subscription Status\", \"Current Plan: Free Trial\"),\n TextContent(\"Your current access includes all premium features, but your account will revert to the 'Basic' tier on October 27th, 2023.\")\n], \"sunk\")\n\nactionCard = Card([\n CardHeader(\"Take Action Now\", \"Secure your continuous access\"),\n Stack([\n TextContent(\"Choose a plan that fits your needs and keep your workflow uninterrupted.\"),\n upgradeBtn,\n viewPlansBtn\n ], \"column\", \"m\", \"center\", \"start\")\n], \"card\")\n\nupgradeBtn = Button(\"Upgrade Now\", Action([@ToAssistant(\"I want to upgrade my subscription immediately. Please show me the checkout options.\")]), \"primary\", \"normal\", \"large\")\n\nviewPlansBtn = Button(\"View All Plans\", Action([@ToAssistant(\"Show me the different subscription tiers and pricing details.\")]), \"secondary\", \"normal\", \"medium\")","promptTokens":4543,"completionTokens":300,"latencyMs":24351,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"chart/adversarial","repeat":3,"output":"root = Stack([dashboard, dataTable], \"column\", \"xl\")\n\ndashboard = Stack([header, kpis, trendChart], \"column\", \"l\")\n\nheader = Card([CardHeader(\"Monthly Performance Overview\", \"Revenue/Metric analysis for Jan - Jul\")])\n\nkpis = Stack([kpiJan, kpiFeb, kpiMar, kpiApr, kpiMay, kpiJun, kpiJul], \"row\", \"m\", \"start\", \"center\", true)\n\nkpiJan = Card([TextContent(\"January\", \"small\"), TextContent(\"4200\", \"large-heavy\")])\nkpiFeb = Card([TextContent(\"February\", \"small\"), TextContent(\"3800\", \"large-heavy\")])\nkpiMar = Card([TextContent(\"March\", \"small\"), TextContent(\"5100\", \"large-heavy\")])\nkpiApr = Card([TextContent(\"April\", \"small\"), TextContent(\"4900\", \"large-heavy\")])\nkpiMay = Card([TextContent(\"May\", \"small\"), TextContent(\"6300\", \"large-heavy\")])\nkpiJun = Card([TextContent(\"June\", \"small\"), TextContent(\"7100\", \"large-heavy\")])\nkpiJul = Card([TextContent(\"July\", \"small\"), TextContent(\"6800\", \"large-heavy\")])\n\ntrendChart = Card([\n CardHeader(\"Trend Analysis\"),\n Tabs([\n TabItem(\"line\", \"Line Chart\", [LineChart([\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"], [Series(\"Value\", [4200, 3800, 5100, 4900, 6300, 7100, 6800])])]),\n TabItem(\"bar\", \"Bar Chart\", [BarChart([\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"], [Series(\"Value\", [4200, 3800, 5100, 4900, 6300, 7100, 6800])])])\n ])\n])\n\ndataTable = Card([\n CardHeader(\"Raw Data Verification\"),\n Table([\n Col(\"Month\", [\"Jan\", \"Feb\", \"Mar\", \"Apr\", \"May\", \"Jun\", \"Jul\"]),\n Col(\"Value\", [4200, 3800, 5100, 4900, 6300, 7100, 6800], \"number\")\n ])\n])","promptTokens":4601,"completionTokens":601,"latencyMs":42821,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/realistic","repeat":4,"output":"root = Stack([\n header_card,\n status_section,\n impact_details,\n customer_comms_section\n], \"column\", \"xl\")\n\nheader_card = Card([\n CardHeader(\"Scheduled System Maintenance\", \"Service Update\"),\n TextContent(\"Please be advised of upcoming planned maintenance to improve our platform performance.\", \"default\")\n], \"sunk\")\n\nstatus_section = Card([\n TextContent(\"Maintenance Schedule\", \"large-heavy\"),\n Separator(\"horizontal\"),\n Stack([\n Stack([\n TextContent(\"Date\", \"small\"),\n TextContent(\"Sunday, October 27, 2024\", \"default\")\n ], \"column\"),\n Stack([\n TextContent(\"Window (UTC)\", \"small\"),\n TextContent(\"02:00 - 06:00\", \"default\")\n ], \"column\"),\n Stack([\n TextContent(\"Duration\", \"small\"),\n TextContent(\"4 Hours\", \"default\")\n ], \"column\")\n ], \"row\", \"l\", \"start\", \"between\")\n], \"card\")\n\nimpact_details = Card([\n CardHeader(\"Service Impact Analysis\"),\n TextContent(\"During this window, service availability will vary by module.\", \"small\"),\n Separator(\"horizontal\"),\n Stack([\n impact_card_unavailable,\n impact_card_available\n ], \"row\", \"m\", \"start\", \"stretch\", true)\n], \"card\")\n\nimpact_card_unavailable = Card([\n Tag(\"Unavailable\", null, \"sm\", \"danger\"),\n TextContent(\"Payments\", \"large-heavy\"),\n TextContent(\"All transaction processing and billing services.\", \"small\")\n], \"card\", \"sunk\")\n\nimpact_card_available = Card([\n Tag(\"Operational\", null, \"sm\", \"success\"),\n TextContent(\"Core Platform\", \"large-heavy\"),\n TextContent(\"Dashboard, reporting, and user management.\", \"small\")\n], \"card\", \"sunk\")\n\ncustomer_comms_section = Card([\n CardHeader(\"Communication Templates\"),\n TextContent(\"Use the following templates to notify your users via email or in-app notifications.\", \"small\"),\n Separator(\"horizontal\"),\n Tabs([\n TabItem(\"email\", \"Email Template\", [email_content]),\n TabItem(\"inapp\", \"In-App Banner\", [inapp_content])\n ])\n], \"card\")\n\nemail_content = Stack([\n TextContent(\"Subject: Upcoming Scheduled Maintenance - [Date]\", \"small-heavy\"),\n Separator(\"horizontal\"),\n MarkDownRenderer(\"## Planned Service Maintenance\\n\\nHello,\\n\\nTo ensure the continued reliability of our services, we have scheduled a **planned maintenance window** for this coming Sunday.\\n\\n**When:** Sunday, 02:00 - 06:00 UTC\\n**What to expect:**\\n- **Payments:** Will be temporarily unavailable.\\n- **Exports:** Data export functions will be paused.\\n- **Everything else:** All other platform features will remain fully operational.\\n\\nWe apologize for any inconvenience and appreciate your patience as we improve our systems.\")\n], \"column\", \"m\")\n\ninapp_content = Stack([\n TextContent(\"Banner Text\", \"small-heavy\"),\n Separator(\"horizontal\"),\n Card([\n Callout(\"info\", \"Scheduled Maintenance\", \"Payments and exports will be unavailable this Sunday, 02:00-06:00 UTC. Most services remain active.\", null),\n TextContent(\"This is a planned update to improve system stability.\", \"small\")\n ], \"clear\")\n], \"column\", \"m\")","promptTokens":4578,"completionTokens":821,"latencyMs":8168,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"data-table/adversarial","repeat":1,"output":"root = Stack([header, kpiRow, comparisonTabs, recommendationCard], \"column\", \"l\")\n\nheader = Card([\n CardHeader(\"Cloud Provider Comparison\", \"Deep dive into AWS, Azure, GCP, Oracle, and IBM\"),\n TextContent(\"Comprehensive analysis of infrastructure, cost, and performance metrics for the top 5 global cloud providers.\")\n], \"sunk\")\n\nkpiRow = Stack([\n kpiAws,\n kpiAzure,\n kpiGcp,\n kpiOracle,\n kpiIbm\n], \"row\", \"s\", \"center\", \"stretch\", true)\n\nkpiAws = Card([TextContent(\"AWS\", \"small\"), TextContent(\"Market Leader\", \"large-heavy\")], \"clear\")\nkpiAzure = Card([TextContent(\"Azure\", \"small\"), TextContent(\"Enterprise Choice\", \"large-heavy\")], \"clear\")\nkpiGcp = Card([TextContent(\"GCP\", \"small\"), TextContent(\"Data/AI Powerhouse\", \"large-heavy\")], \"clear\")\nkpiOracle = Card([TextContent(\"Oracle\", \"small\"), TextContent(\"DB Specialist\", \"large-heavy\")], \"clear\")\nkpiIbm = Card([TextContent(\"IBM\", \"small\"), TextContent(\"Hybrid/Mainframe\", \"large-heavy\")], \"clear\")\n\ncomparisonTabs = Tabs([\n tabSpecs,\n tabPerformance,\n tabPricing\n])\n\ntabSpecs = TabItem(\"specs\", \"Full Specifications\", [\n Card([\n CardHeader(\"Technical & Compliance Matrix\"),\n Table([\n Col(\"Provider\", providerNames),\n Col(\"Global Regions\", regionCounts),\n Col(\"Compliance\", complianceList),\n Col(\"Support Tiers\", supportTiers),\n Col(\"Primary Strength\", strengthList)\n ])\n ])\n])\n\ntabPerformance = TabItem(\"perf\", \"Performance & Latency\", [\n Card([\n CardHeader(\"Compute & Network Performance\"),\n Tabs([\n TabItem(\"cold\", \"Cold Starts (Serverless)\", [\n BarChart([\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"IBM\"], [\n Series(\"Avg Cold Start (ms)\", [250, 310, 220, 450, 500])\n ])\n ]),\n TabItem(\"uptime\", \"Uptime SLA (%)\", [\n RadialChart([\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"IBM\"], [99.99, 99.95, 99.99, 99.9, 99.95])\n ])\n ])\n ])\n])\n\ntabPricing = TabItem(\"pricing\", \"Cost & Egress Analysis\", [\n Card([\n CardHeader(\"Cost Structure Comparison\"),\n Table([\n Col(\"Provider\", providerNames),\n Col(\"Free Tier\", freeTierList),\n Col(\"Egress Fees\", egressFees),\n Col(\"Complexity\", complexityList)\n ])\n ])\n])\n\nrecommendationCard = Card([\n CardHeader(\"AI Assistant Recommendation\"),\n TextContent(\"Based on the current landscape, my selection depends on your specific workload profile:\", \"small-heavy\"),\n Separator(\"horizontal\"),\n Stack([\n TextContent(\"🚀 For Scale & Ecosystem: **AWS** - Unmatched breadth of services and third-party integrations.\"),\n TextContent(\"🏢 For Enterprise/Windows: **Azure** - Seamless integration with Active Directory and Microsoft stack.\"),\n TextContent(\"🤖 For ML/Data Science: **GCP** - Superior Kubernetes support and specialized AI/ML hardware (TPUs).\"),\n TextContent(\"🗄️ For Database Heavy: **Oracle** - Most optimized for massive relational workloads.\"),\n TextContent(\"🛡️ For Regulated Hybrid: **IBM** - Best for strict data sovereignty and mainframe integration.\")\n ], \"column\", \"s\")\n], \"card\")\n\nproviderNames = [\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"IBM\"]\nregionCounts = [\"33+\", \"60+\", \"39+\", \"45+\", \"48+\"]\ncomplianceList = [\n [\"SOC, HIPAA, PCI, FedRAMP\", \"SOC, HIPAA, PCI, FedRAMP\", \"SOC, HIPAA, PCI, FedRAMP\", \"SOC, HIPAA, PCI, FedRAMP\", \"SOC, HIPAA, PCI, FedRAMP\"],\n [\"High\", \"High\", \"High\", \"Medium\", \"High\"]\n]\nsupportTiers = [\"Basic, Developer, Business, Enterprise\", \"Basic, Developer, Standard, Professional\", \"Basic, Role-based, Enterprise\", \"Standard, Premier\", \"Standard, Premium\"]\nstrengthList = [\"Ecosystem\", \"Enterprise Integration\", \"Data/AI\", \"Database\", \"Hybrid Cloud\"]\n\nfreeTierList = [\"12mo Free/Always\", \"12mo Free/Always\", \"Always Free/Trial\", \"Trial Only\", \"Limited Trial\"]\negressFees = [\"High/Complex\", \"Moderate\", \"Moderate/Tiered\", \"Low/Specialized\", \"Moderate\"]\ncomplexityList = [\"Very High\", \"High\", \"Medium\", \"Medium\", \"High\"]\n\n// Helper for the table structure logic\ncomplianceList = [\n \"SOC, HIPAA, PCI, FedRAMP\",\n \"SOC, HIPAA, PCI, FedRAMP\",\n \"SOC, HIPAA, PCI, FedRAMP\",\n \"SOC, HIPAA, PCI, FedRAMP\",\n \"SOC, HIPAA, PCI, FedRAMP\"\n]\nsupportTiers = [\n \"Basic, Dev, Biz, Ent\",\n \"Basic, Dev, Std, Prof\",\n \"Basic, Role, Ent\",\n \"Std, Premier\",\n \"Std, Premium\"\n]\nstrengthList = [\"Ecosystem\", \"MS Integration\", \"Data/AI\", \"Database\", \"Hybrid\"]\n\n// Re-mapping columns for the Table component to match the schema\n// Table takes Col[] where Col is (label, data, type?)\n// The data must be a flat array of the same length as labels\nCol(\"Provider\", [\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"IBM\"])\nCol(\"Regions\", [\"33+\", \"60+\", \"39+\", \"45+\", \"48+\"])\nCol(\"Compliance\", [\"Full\", \"Full\", \"Full\", \"Full\", \"Full\"])\nCol(\"Support\", [\"4 Tiers\", \"4 Tiers\", \"3 Tiers\", \"2 Tiers\", \"2 Tiers\"])\nCol(\"Strength\", [\"Scale\", \"Enterprise\", \"Intelligence\", \"DB\", \"Hybrid\"])\n\n// Redefining columns for the pricing table\nCol(\"Provider\", [\"AWS\", \"Azure\", \"GCP\", \"Oracle\", \"IBM\"])\nCol(\"Free Tier\", [\"12mo/Always\", \"12mo/Always\", \"Always Free\", \"Trial\", \"Trial\"])\nCol(\"Egress\", [\"High\", \"Mid\", \"Mid\", \"Low\", \"Mid\"])\nCol(\"Complexity\", [\"High\", \"High\", \"Mid\", \"Mid\", \"High\"])","promptTokens":4591,"completionTokens":1564,"latencyMs":141010,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/minimal","repeat":1,"output":"root = Stack([\n header,\n warningCallout,\n statusCards,\n actionSection\n], \"column\", \"m\")\n\nheader = Card([\n CardHeader(\"Subscription Status\", \"Account: Premium Trial\")\n], \"sunk\")\n\nwarningCallout = Callout(\"warning\", \"Trial Expiring Soon\", \"Your premium trial period will expire in 3 days. Upgrade now to avoid service interruption.\", true)\n\nstatusCards = Stack([\n kpiCard,\n kpiCard2\n], \"row\", \"m\")\n\nkpiCard = Card([\n TextContent(\"Days Remaining\", \"small\"),\n TextContent(\"3\", \"large-heavy\")\n], \"card\")\n\nkpiCard2 = Card([\n TextContent(\"Current Plan\", \"small\"),\n TextContent(\"Trial (Pro)\", \"large-heavy\")\n], \"card\")\n\nactionSection = Card([\n TextContent(\"Don't lose access to your advanced features, custom workflows, and priority support.\", \"small\"),\n Separator(\"horizontal\"),\n Buttons([upgradeBtn, viewPlansBtn], \"row\")\n], \"card\")\n\nupgradeBtn = Button(\"Upgrade Now\", Action([@ToAssistant(\"I want to upgrade my subscription\")]), \"primary\")\nviewPlansBtn = Button(\"View Pricing\", Action([@OpenUrl(\"https://example.com/pricing\")]), \"secondary\")","promptTokens":4543,"completionTokens":301,"latencyMs":49870,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/adversarial","repeat":2,"output":"root = Stack([\n ErrorContainer,\n WarningCallout,\n FailureDetails,\n RetryButton\n], \"column\", \"xl\", \"center\", \"center\")\n\nErrorContainer = Card([\n BigErrorText,\n Separator(\"horizontal\", true),\n FailureDetails,\n RetryButton\n], \"card\", \"column\", \"center\", \"center\", \"xl\")\n\nBigErrorText = TextContent(\"CRITICAL SYSTEM FAILURE\", \"large-heavy\")\n\nWarningCallout = Callout(\"error\", \"FATAL ERROR\", \"The operation could not be completed. The system has encountered an unrecoverable exception.\", true)\n\nFailureDetails = Stack([\n TextContent(\"Error Code: 0x8004210B\", \"small\"),\n TextContent(\"Status: DISCONNECTED\", \"small\"),\n TagBlock([\"UNRECOVERABLE\", \"TIMEOUT\", \"CRITICAL\"])\n], \"column\", \"xs\", \"center\")\n\nRetryButton = Button(\"RETRY OPERATION\", Action([@ToAssistant(\"Please try the operation again immediately.\")]), \"primary\", \"normal\", \"large\")","promptTokens":4556,"completionTokens":241,"latencyMs":2605,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/realistic","repeat":0,"output":"root = Stack([\n maintenanceBanner,\n mainContent\n], \"column\", \"none\", \"stretch\", \"center\")\n\nmaintenanceBanner = Card([\n CardHeader(\"Scheduled Maintenance\", \"System Update Notification\"),\n TextCallout(\"info\", \"Planned Maintenance Window\", \"We will be performing scheduled system improvements this Sunday.\"),\n Separator(\"horizontal\"),\n Stack([\n Stack([\n TextContent(\"When:\", \"small-heavy\"),\n TextContent(\"Sunday, 02:00 - 06:00 UTC\", \"default\")\n ], \"row\", \"s\", \"start\"),\n Stack([\n TextContent(\"Impact:\", \"small-heavy\"),\n TextContent(\"Payments & Exports: Unavailable\", \"default\"),\n TextContent(\"All other services: Fully Operational\", \"default\")\n ], \"row\", \"s\", \"start\")\n ], \"column\", \"xs\", \"start\")\n], \"sunk\")\n\nmainContent = Stack([\n Card([\n CardHeader(\"Customer Communication Toolkit\"),\n TextContent(\"Use the resources below to notify your users about the upcoming maintenance window.\", \"small\"),\n Separator(),\n Tabs([\n TabItem(\"email\", \"Email Template\", [emailTemplate]),\n TabItem(\"status\", \"Status Page Widget\", [statusWidget]),\n TabItem(\"social\", \"Social Media\", [socialMediaPosts])\n ])\n ])\n], \"column\", \"m\")\n\nemailTemplate = Stack([\n TextContent(\"Email Template Preview\", \"large-heavy\"),\n Card([\n TextContent(\"Subject: Scheduled Maintenance: [Date] [Time]\", \"small-heavy\"),\n Separator(),\n MarkDownRenderer(\"Hi {{customer_name}},\\n\\nWe are performing scheduled maintenance to improve our platform. \\n\\n**Window:** Sunday, 02:00 - 06:00 UTC\\n**What to expect:** During this time, **Payments and Exports** will be temporarily unavailable. All other features will remain fully functional.\\n\\nWe apologize for any inconvenience and appreciate your patience as we work to improve our service.\\n\\nBest regards,\\n[Company Name] Team\")\n ], \"clear\"),\n Buttons([\n Button(\"Copy Template\", Action([@ToAssistant(\"Copy the email template for me\")]), \"primary\"),\n Button(\"Customize\", Action([@ToAssistant(\"Help me customize this email\")]), \"secondary\")\n ])\n], \"column\", \"m\")\n\nstatusWidget = Stack([\n TextContent(\"Status Page Widget Preview\", \"large-heavy\"),\n Card([\n Stack([\n TextContent(\"System Status\", \"large\"),\n Tag(\"Planned\", \"neutral\", \"sm\")\n ], \"row\", \"between\", \"center\"),\n Separator(),\n Stack([\n Stack([\n TextContent(\"Core Services\", \"default\"),\n Tag(\"Operational\", \"success\", \"sm\")\n ], \"row\", \"between\", \"center\"),\n Stack([\n TextContent(\"Payments\", \"default\"),\n Tag(\"Maintenance\", \"warning\", \"sm\")\n ], \"row\", \"between\", \"center\"),\n Stack([\n TextContent(\"Exports\", \"default\"),\n Tag(\"Maintenance\", \"warning\", \"sm\")\n ], \"row\", \"between\", \"center\")\n ], \"column\", \"s\")\n ])\n], \"column\", \"m\")\n\nsocialMediaPosts = Stack([\n TextContent(\"Social Media Snippets\", \"large-heavy\"),\n Accordion([\n AccordionItem(\"twitter\", \"Twitter/X (Short)\", [\n Card([\n TextContent(\"📢 Scheduled Maintenance Notice\\n\\nWe'll be performing system updates this Sunday (02:00-06:00 UTC). Payments and Exports will be offline, but everything else stays up! 🛠️\\n\\nThanks for your patience!\", \"small\")\n ])\n ]),\n AccordionItem(\"slack\", \"Slack/Discord (Community)\", [\n Card([\n TextContent(\"Hey @everyone! Just a heads-up: We have planned maintenance this Sunday between 02:00 and 06:00 UTC. 🛠️\\n\\n🚫 **Unavailable:** Payments & Exports\\n✅ **Available:** Everything else!\\n\\nWe'll post an update here once we're back to 100%.\", \"small\")\n ])\n ])\n ])\n], \"column\", \"m\")","promptTokens":4579,"completionTokens":986,"latencyMs":39235,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/realistic","repeat":2,"output":"root = Stack([\n maintenanceBanner,\n mainContent\n], \"column\", \"none\", \"center\")\n\nmaintenanceBanner = Callout(\"info\", \"Scheduled Maintenance\", \"We will be performing planned system maintenance this Sunday from 02:00 to 06:00 UTC. Please see details below.\", $bannerVisible)\n\nmainContent = Stack([\n Card([\n CardHeader(\"Maintenance Schedule\", \"Sunday, 02:00 - 06:00 UTC\"),\n Separator(\"horizontal\"),\n Stack([\n Stack([\n TextContent(\"Status\", \"small\"),\n TextContent(\"Planned\", \"default\")\n ], \"column\", \"none\", \"center\"),\n Stack([\n TextContent(\"Duration\", \"small\"),\n TextContent(\"4 Hours\", \"default\")\n ], \"column\", \"none\", \"center\"),\n Stack([\n TextContent(\"Type\", \"small\"),\n TextContent(\"System Update\", \"default\")\n ], \"column\", \"none\", \"center\")\n ], \"row\", \"xl\", \"center\", \"evenly\"),\n Separator(\"horizontal\"),\n MarkDownRenderer(\"### ⚠️ Impacted Services\\n- **Payments:** Unavailable\\n- **Data Exports:** Unavailable\\n- **Dashboard & API:** Fully Operational\\n- **User Authentication:** Fully Operational\", \"sunk\"),\n Separator(\"horizontal\"),\n TextContent(\"We are performing these updates to improve our infrastructure. This is a scheduled procedure and not an unexpected outage. We apologize for any inconvenience.\", \"small\")\n ], \"card\"),\n \n Stack([\n Card([\n CardHeader(\"Service Availability Matrix\"),\n Table([\n Col(\"Service\", [\"Core Platform\", \"Payments\", \"Exports\", \"API Access\", \"Reporting\"]),\n Col(\"Status\", [\"✅ Online\", \"❌ Maintenance\", \"❌ Maintenance\", \"✅ Online\", \"✅ Online\"], \"string\"),\n Col(\"Impact\", [\"None\", \"No transactions\", \"No downloads\", \"None\", \"None\"], \"string\")\n ])\n ], \"card\"),\n \n Card([\n CardHeader(\"Customer Support\"),\n TextContent(\"If you have urgent questions regarding your account during this window, please contact our support team.\"),\n Buttons([\n Button(\"Contact Support\", Action([@ToAssistant(\"I have a question about the upcoming maintenance\")]), \"secondary\"),\n Button(\"View Status Page\", Action([@OpenUrl(\"https://status.example.com\")]), \"tertiary\")\n ], \"row\")\n ], \"sunk\")\n ], \"column\", \"m\")\n], \"column\", \"m\", \"center\")\n\n$bannerVisible = true","promptTokens":4579,"completionTokens":590,"latencyMs":31211,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/adversarial","repeat":1,"output":"root = Stack([\n errorHeader,\n errorCard,\n retrySection\n], \"column\", \"xl\", \"center\", \"center\")\n\nerrorHeader = Card([\n TextContent(\"CRITICAL SYSTEM FAILURE\", \"large-heavy\"),\n Separator(\"horizontal\"),\n TextCallout(\"danger\", \"FATAL ERROR\", \"The operation could not be completed. The system has encountered an unrecoverable error state.\")\n], \"card\", \"column\", \"center\", \"center\", \"xl\")\n\nerrorCard = Card([\n Stack([\n TextContent(\"ERROR CODE: 0x882-FATAL\", \"large\"),\n TextContent(\"Status: Immediate Action Required\", \"default\"),\n Separator(\"horizontal\"),\n TextContent(\"All processes have been halted to prevent data corruption. Please review the logs or contact your administrator immediately.\", \"small\")\n ], \"column\", \"m\", \"center\", \"center\")\n], \"sunk\", \"column\", \"start\", \"stretch\")\n\nretrySection = Stack([\n TextContent(\"Would you like to attempt a system reboot?\", \"default\"),\n Buttons([\n Button(\"RETRY OPERATION\", Action([@ToAssistant(\"Please attempt to retry the operation immediately.\")]), \"primary\", \"normal\", \"large\"),\n Button(\"ABORT AND EXIT\", Action([@ToAssistant(\"Abort the process and exit safely.\")]), \"destructive\", \"normal\", \"large\")\n ], \"row\", \"m\")\n], \"row\", \"m\", \"center\", \"center\")","promptTokens":4556,"completionTokens":332,"latencyMs":12685,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/minimal","repeat":1,"output":"root = Stack([\n Card([\n CardHeader(\"Invoice Details\", \"Invoice #INV-2023-0042\"),\n TextContent(\"Thank you for your business. Please find your invoice summary below. You can download the full PDF version using the button provided.\"),\n Separator(\"horizontal\"),\n Stack([\n InvoiceSummary,\n DownloadSection\n ], \"row\", \"m\", \"start\", \"between\", true)\n ])\n], \"column\", \"l\")\n\nInvoiceSummary = Stack([\n Card([\n TextContent(\"Summary\", \"small-heavy\"),\n TextContent(\"Total Amount: $1,250.00\", \"large\"),\n TextContent(\"Status: Paid\", \"small\")\n ], \"s\"),\n Card([\n TextContent(\"Due Date\", \"small-heavy\"),\n TextContent(\"Oct 24, 2023\", \"default\"),\n TextContent(\"Payment Method: Visa ****4242\", \"small\")\n ], \"s\")\n], \"row\", \"s\", \"start\", \"center\", true)\n\nDownloadSection = Card([\n Stack([\n IconDownload,\n DownloadButton\n ], \"row\", \"s\", \"center\", \"start\")\n], \"clear\")\n\nIconDownload = TextContent(\"📄\", \"large\")\n\nDownloadButton = Button(\"Download PDF Invoice\", Action([@OpenUrl(\"https://example.com/invoices/INV-2023-0042.pdf\")]), \"primary\", \"normal\", \"medium\")","promptTokens":4541,"completionTokens":345,"latencyMs":4225,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/minimal","repeat":2,"output":"root = Stack([\n Card([\n CardHeader(\"Invoice #INV-2024-001\", \"Date: October 24, 2023\"),\n Separator(\"horizontal\"),\n Stack([\n SummaryItem(\"Customer\", \"Acme Corp\"),\n SummaryItem(\"Amount Due\", \"$1,250.00\"),\n SummaryItem(\"Status\", Tag(\"Paid\", null, \"sm\", \"success\")),\n ], \"row\", \"m\", \"between\"),\n Separator(\"horizontal\"),\n TextContent(\"Thank you for your business! You can download your official tax invoice using the button below.\"),\n ActionSection\n ], \"card\"),\n InvoiceActions\n], \"column\", \"l\")\n\nSummaryItem = Card([\n Stack([\n TextContent(\"Label\", \"small\"),\n TextContent(\"Value\", \"default\"),\n ], \"column\", \"xs\")\n], \"clear\", \"column\", \"start\")\n\nInvoiceActions = Card([\n ActionSection\n], \"clear\")\n\nActionSection = Buttons([\n DownloadBtn,\n EmailBtn\n], \"row\")\n\nDownloadBtn = Button(\"Download Invoice\", Action([@OpenUrl(\"https://example.com/files/invoice_inv_2024_001.pdf\")]), \"primary\")\nEmailBtn = Button(\"Email Copy\", Action([@ToAssistant(\"Please email a copy of invoice #INV-2024-001 to acme@example.com\")]), \"secondary\")","promptTokens":4541,"completionTokens":332,"latencyMs":6056,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/adversarial","repeat":0,"output":"root = Stack([\n ErrorOverlay,\n MainContent\n], \"column\", \"none\", \"center\", \"center\")\n\nErrorOverlay = Card([\n ErrorIcon,\n ErrorTitle,\n ErrorMessage,\n RetryButton\n], \"card\", \"column\", \"center\", \"center\", \"xl\")\n\nErrorIcon = TextContent(\"❌\", \"large-heavy\")\n\nErrorTitle = TextContent(\"CRITICAL SYSTEM FAILURE\", \"large-heavy\")\n\nErrorMessage = TextContent(\"THE REQUEST COULD NOT BE PROCESSED. THE OPERATION HAS FAILED COMPLETELY.\", \"default\")\n\nRetryButton = Button(\"RETRY OPERATION\", Action([@ToAssistant(\"Please try the operation again\")]), \"destructive\", \"large\")\n\nMainContent = Card([\n CardHeader(\"System Status\", \"Operational (Degraded)\"),\n Separator(\"horizontal\"),\n TextContent(\"While the error above is critical, the rest of the dashboard is still loading components...\")\n], \"sunk\")\n\n// Note: The 'flashing' effect is simulated via the high-contrast 'destructive' variant and heavy typography in the UI engine.","promptTokens":4557,"completionTokens":238,"latencyMs":25801,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/realistic","repeat":3,"output":"root = Stack([\n maintenanceHeader,\n statusDashboard,\n notificationCard,\n customerNoticeCard,\n actionButtons\n], \"column\", \"l\")\n\nmaintenanceHeader = Card([\n CardHeader(\"System Maintenance Notice\", \"Scheduled Infrastructure Upgrade\"),\n TextCallout(\"info\", \"Planned Maintenance Window\", \"Services will be partially unavailable during this time.\")\n], \"sunk\")\n\nstatusDashboard = Card([\n CardHeader(\"Service Availability Impact\"),\n Stack([\n kpiPayments,\n kpiExports,\n kpiCoreServices\n ], \"row\", \"m\", \"stretch\", \"center\", \"space-between\"),\n Separator(\"horizontal\"),\n Stack([\n statusTable\n ])\n], \"card\")\n\nkpiPayments = Card([\n TextContent(\"Payments\", \"small\"),\n TextContent(\"UNAVAILABLE\", \"large-heavy\"),\n Tag(\"maintenance\", null, \"sm\", \"warning\")\n], \"sunk\", \"column\", \"center\")\n\nkpiExports = Card([\n TextContent(\"Exports\", \"small\"),\n TextContent(\"UNAVAILABLE\", \"large-heavy\"),\n Tag(\"maintenance\", null, \"sm\", \"warning\")\n], \"sunk\", \"column\", \"center\")\n\nkpiCoreServices = Card([\n TextContent(\"Core Services\", \"small\"),\n TextContent(\"OPERATIONAL\", \"large-heavy\"),\n Tag(\"active\", null, \"sm\", \"success\")\n], \"sunk\", \"column\", \"center\")\n\nstatusTable = Table([\n Col(\"Service Module\", [\"Payments Gateway\", \"Data Export Engine\", \"User Dashboard\", \"API Access\", \"Reporting\"]),\n Col(\"Status\", [\"Down\", \"Down\", \"Up\", \"Up\", \"Up\"]),\n Col(\"Type\", [\"Planned\", \"Planned\", \"Normal\", \"Normal\", \"Normal\"])\n])\n\nnotificationCard = Card([\n CardHeader(\"Maintenance Details\"),\n Stack([\n TextContent(\"📅 **Time:** Sunday, 02:00 - 06:00 UTC\", \"default\"),\n TextContent(\"🛠️ **Reason:** Routine system optimization and security patching.\", \"default\"),\n TextContent(\"✅ **Scope:** This is a planned maintenance window to improve platform stability.\", \"default\")\n ], \"column\", \"s\")\n])\n\ncustomerNoticeCard = Card([\n CardHeader(\"Customer Communication Template\"),\n MarkDownRenderer(\"### 📢 Scheduled Maintenance Notice\\n\\n**Dear Customer,**\\n\\nPlease be advised that we have scheduled routine maintenance for this **Sunday between 02:00 and 06:00 UTC**.\\n\\n**What to expect:**\\n- **Payments:** Will be temporarily unavailable.\\n- **Data Exports:** Will be temporarily unavailable.\\n- **All other features:** Will continue to work as normal.\\n\\nThis is a planned upgrade to ensure we continue providing you with the best possible service. We apologize for any inconvenience.\\n\\n*Thank you for your patience!*\", \"card\")\n])\n\nactionButtons = Buttons([\n Button(\"Copy Notice\", Action([@ToAssistant(\"Please copy the customer notice text for me to use in an email.\")]), \"primary\"),\n Button(\"Schedule Reminder\", Action([@ToAssistant(\"Set a reminder for this maintenance window.\")]), \"secondary\")\n], \"row\")","promptTokens":4579,"completionTokens":737,"latencyMs":48486,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/minimal","repeat":4,"output":"root = Stack([\n Card([\n CardHeader(\"Invoice Details\", \"INV-2023-0042\"),\n Separator(),\n Stack([\n TextContent(\"Order Summary\", \"large\"),\n TextContent(\"Thank you for your recent purchase. You can download your official tax invoice below.\"),\n ], \"column\", \"m\"),\n Separator(),\n Stack([\n TextContent(\"Status\", \"small\"),\n Tag(\"Paid\", null, \"sm\", \"success\"),\n ], \"row\", \"m\"),\n Stack([\n TextContent(\"Amount Due\", \"small\"),\n TextContent(\"$142.50\", \"large-heavy\"),\n ], \"row\", \"m\"),\n Separator(),\n downloadBtn,\n ]),\n], \"center\")\n\ndownloadBtn = Button(\"Download Invoice\", Action([@OpenUrl(\"https://example.com/invoices/inv-2023-0042.pdf\")]), \"primary\", \"normal\", \"large\")","promptTokens":4542,"completionTokens":225,"latencyMs":11861,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/adversarial","repeat":4,"output":"root = Stack([\n ErrorHeader,\n ErrorContent,\n RetryButton\n], \"column\", \"xl\", \"center\", \"center\")\n\nErrorHeader = Card([\n TextContent(\"CRITICAL SYSTEM FAILURE\", \"large-heavy\"),\n Separator(\"horizontal\", true),\n TextCallout(\"danger\", \"FATAL ERROR\", \"The operation could not be completed. The system has encountered an unrecoverable error state.\")\n], \"card\", \"column\", \"center\", \"center\", \"m\")\n\nErrorContent = Card([\n Stack([\n TextContent(\"❌\", \"large-heavy\"),\n TextContent(\"ERROR CODE: 0x8842-FATAL\", \"large\"),\n TextContent(\"Status: Terminated\", \"default\"),\n TextContent(\"All processes have been halted to prevent data corruption.\", \"small\")\n ], \"column\", \"m\", \"center\", \"center\")\n], \"sunk\", \"column\", \"center\", \"center\")\n\nRetryButton = Buttons([\n Button(\"RETRY OPERATION\", Action([@ToAssistant(\"Please try the operation again. I am ready to retry.\")]), \"primary\", \"normal\", \"large\"),\n Button(\"ABORT AND EXIT\", Action([@OpenUrl(\"https://google.com\")]), \"secondary\", \"destructive\", \"large\")\n], \"column\", \"m\")","promptTokens":4557,"completionTokens":296,"latencyMs":33384,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/minimal","repeat":3,"output":"root = Stack([\n headerCard,\n invoiceDetails,\n actionSection\n], \"column\", \"l\")\n\nheaderCard = Card([\n CardHeader(\"Invoice #INV-2024-001\", \"Issued on Oct 24, 2024\"),\n Separator(\"horizontal\")\n], \"sunk\")\n\ninvoiceDetails = Card([\n Stack([\n Card([\n TextContent(\"Customer Information\", \"small-heavy\"),\n Separator(\"horizontal\"),\n TextContent(\"Alex Rivers\"),\n TextContent(\"alex.rivers@example.com\"),\n TextContent(\"+1 (555) 012-3456\")\n ], \"s\"),\n Card([\n TextContent(\"Billing Summary\", \"small-heavy\"),\n Separator(\"horizontal\"),\n TextContent(\"Total Amount\", \"small\"),\n TextContent(\"$1,250.00\", \"large-heavy\"),\n Tag(\"PAID\", null, \"sm\", \"success\")\n ], \"s\")\n ], \"row\", \"m\", \"start\", \"stretch\", true)\n], \"clear\")\n\nactionSection = Card([\n Stack([\n TextContent(\"Need a copy for your records?\"),\n downloadBtn\n ], \"row\", \"m\", \"center\", \"end\")\n], \"clear\")\n\ndownloadBtn = Button(\"Download Invoice (PDF)\", Action([@OpenUrl(\"https://example.com/invoices/INV-2024-001.pdf\")]), \"primary\", \"normal\", \"medium\")","promptTokens":4542,"completionTokens":349,"latencyMs":22592,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/realistic","repeat":3,"output":"root = Stack([\n header,\n subscriptionStatus,\n actionSection\n], \"column\", \"xl\")\n\nheader = Card([\n CardHeader(\"Subscription Management\", \"Manage your current plan and billing preferences\"),\n Separator(\"horizontal\")\n], \"card\")\n\nsubscriptionStatus = Card([\n Stack([\n Stack([\n TextContent(\"Current Plan\", \"small\"),\n Tag(\"Monthly Pro\", null, \"md\", \"info\")\n ], \"column\", \"xs\"),\n Stack([\n TextContent(\"Next Billing Date\", \"small\"),\n TextContent(\"October 24, 2023\", \"default\")\n ], \"column\", \"xs\")\n ], \"row\", \"between\")\n], \"sunk\")\n\nactionSection = Stack([\n TextContent(\"Next Steps\", \"large-heavy\"),\n TextContent(\"Choose an option below to manage your subscription status.\", \"small\"),\n Separator(\"horizontal\"),\n Stack([\n upgradeCard,\n cancelCard\n ], \"row\", \"m\", \"start\", \"center\", true)\n], \"column\", \"l\")\n\nupgradeCard = Card([\n Stack([\n TextContent(\"Upgrade to Annual\", \"large\"),\n TextContent(\"Save 20% by switching to a yearly plan.\", \"small\"),\n Separator(\"horizontal\"),\n Button(\"Switch to Annual\", Action([@ToAssistant(\"I want to upgrade to the annual plan\")]), \"primary\", \"normal\", \"large\")\n ], \"column\", \"m\", \"center\", \"center\")\n], \"card\")\n\ncancelCard = Card([\n Stack([\n TextContent(\"Cancel Subscription\", \"large\"),\n TextContent(\"You will lose access to Pro features at the end of your cycle.\", \"small\"),\n Separator(\"horizontal\"),\n Button(\"Cancel Plan\", Action([@ToAssistant(\"I want to cancel my subscription\")]), \"secondary\", \"destructive\", \"large\")\n ], \"column\", \"m\", \"center\", \"center\")\n], \"clear\")","promptTokens":4570,"completionTokens":446,"latencyMs":4731,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/minimal","repeat":0,"output":"root = Stack([\n Card([\n CardHeader(\"Invoice Summary\", \"Order #INV-88291\"),\n Separator(\"horizontal\"),\n Stack([\n TextContent(\"Total Amount Due\", \"small\"),\n TextContent(\"$1,240.50\", \"large-heavy\")\n ], \"row\", \"m\", \"start\", \"between\"),\n Separator(\"horizontal\"),\n TextContent(\"Status: Paid\", \"default\"),\n Tag(\"success\", \"Paid\", \"sm\"),\n Separator(\"horizontal\"),\n downloadBtn\n ], \"card\"),\n invoiceDetails\n], \"column\", \"l\")\n\ndownloadBtn = Button(\"Download Invoice\", Action([@OpenUrl(\"https://example.com/invoices/INV-88291.pdf\")]), \"primary\", \"normal\", \"large\")\n\ninvoiceDetails = Card([\n CardHeader(\"Billing Details\"),\n Stack([\n Stack([\n TextContent(\"Bill To\", \"small-heavy\"),\n TextContent(\"Acme Corp\\n123 Business Rd\\nSan Francisco, CA\")\n ], \"column\"),\n Stack([\n TextContent(\"Ship To\", \"small-heavy\"),\n TextContent(\"Acme Warehouse\\n456 Logistics Way\\nOakland, CA\")\n ], \"column\")\n ], \"row\", \"m\", \"start\", \"between\"),\n Separator(\"horizontal\"),\n invoiceTable\n], \"sunk\")\n\ninvoiceTable = Table([\n Col(\"Description\", [\"Cloud Hosting (Monthly)\", \"Domain Registration\", \"SSL Certificate\"]),\n Col(\"Quantity\", [1, 1, 1], \"number\"),\n Col(\"Unit Price\", [\"$1,100.00\", \"$120.00\", \"$20.50\"], \"number\"),\n Col(\"Total\", [\"$1,100.00\", \"$120.00\", \"$20.50\"], \"number\")\n])","promptTokens":4542,"completionTokens":441,"latencyMs":48505,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/adversarial","repeat":0,"output":"root = Button(\"Click Me\")","promptTokens":4557,"completionTokens":8,"latencyMs":613,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"callout/adversarial","repeat":3,"output":"root = Stack([\n errorHeader,\n errorCard,\n retrySection\n], \"column\", \"xl\", \"center\", \"center\")\n\nerrorHeader = Card([\n TextContent(\"CRITICAL SYSTEM FAILURE\", \"large-heavy\"),\n TextContent(\"The requested operation could not be completed. All processes have been halted.\", \"default\")\n], \"card\", \"column\", \"center\", \"center\", \"center\")\n\nerrorCard = Card([\n Callout(\"error\", \"FATAL ERROR\", \"An unrecoverable error occurred during execution. Please check your connection or system status.\"),\n Separator(\"horizontal\"),\n Stack([\n TextContent(\"Error Code: 0x8004210B\", \"small\"),\n TextContent(\"Timestamp: \" + @Now(), \"small\")\n ], \"row\", \"none\", \"center\", \"between\")\n], \"sunk\")\n\nretrySection = Stack([\n TextContent(\"Would you like to attempt a system recovery?\", \"large\"),\n retryButtons\n], \"column\", \"m\", \"center\", \"center\")\n\nretryButtons = Buttons([\n Button(\"RETRY IMMEDIATELY\", Action([@ToAssistant(\"Please retry the failed operation immediately.\")]), \"primary\", \"normal\", \"large\"),\n Button(\"ABORT AND EXIT\", Action([@ToAssistant(\"Abort the current process and exit safely.\")]), \"secondary\", \"destructive\", \"large\")\n], \"row\", \"m\")","promptTokens":4557,"completionTokens":319,"latencyMs":56831,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/adversarial","repeat":2,"output":"root = Button(\"Click Me\")","promptTokens":4558,"completionTokens":8,"latencyMs":1034,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/adversarial","repeat":1,"output":"root = Button(\"Click Me\")","promptTokens":4558,"completionTokens":8,"latencyMs":1346,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/adversarial","repeat":4,"output":"root = Button(\"Click Me\")","promptTokens":4558,"completionTokens":8,"latencyMs":450,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/adversarial","repeat":3,"output":"root = Button(\"Click Me\")","promptTokens":4558,"completionTokens":8,"latencyMs":868,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/realistic","repeat":4,"output":"root = Stack([\n SubscriptionSummary,\n Separator(\"horizontal\"),\n ActionCardStack\n], \"column\", \"xl\")\n\nSubscriptionSummary = Card([\n CardHeader(\"Your Current Plan\", \"Monthly Pro Plan\"),\n Stack([\n StatusTag,\n TextContent(\"Next billing date: October 15, 2023\", \"small\"),\n Separator(\"horizontal\"),\n PriceDetail\n ], \"column\", \"m\")\n], \"sunk\")\n\nStatusTag = Tag(\"Active\", null, \"md\", \"success\")\n\nPriceDetail = Stack([\n TextContent(\"$19.99 / month\", \"large-heavy\"),\n TextContent(\"Automatic renewal enabled\", \"small\")\n], \"row\", \"m\", \"start\")\n\nActionCardStack = Stack([\n UpgradeCard,\n CancelCard\n], \"row\", \"m\", \"start\", \"wrap\")\n\nUpgradeCard = Card([\n CardHeader(\"Save 20% with Annual\", \"Switch to a yearly plan and get two months free.\"),\n Stack([\n TextContent(\"$192.00 / year\", \"large-heavy\"),\n TextContent(\"Billed annually\", \"small\")\n ], \"column\", \"m\"),\n UpgradeButton\n], \"card\", \"column\", \"m\")\n\nUpgradeButton = Button(\"Upgrade to Annual\", Action([@ToAssistant(\"I want to upgrade to the annual plan\")]), \"primary\", \"medium\")\n\nCancelCard = Card([\n CardHeader(\"Manage Subscription\", \"If you cancel, you will lose access to premium features at the end of your billing cycle.\"),\n TextContent(\"You can re-subscribe at any time.\", \"small\"),\n CancelButton\n], \"card\", \"column\", \"m\")\n\nCancelButton = Button(\"Cancel Subscription\", Action([@ToAssistant(\"I want to cancel my subscription\")]), \"secondary\", \"destructive\", \"medium\")","promptTokens":4571,"completionTokens":417,"latencyMs":16080,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/realistic","repeat":0,"output":"root = Stack([\n subscriptionHeader,\n subscriptionStatus,\n Separator(\"horizontal\"),\n actionSection\n], \"column\", \"xl\", \"center\", \"center\")\n\nsubscriptionHeader = Card([\n CardHeader(\"Subscription Management\", \"Manage your current plan and billing cycle\"),\n TextContent(\"Your current subscription is active and renewing on the 15th of every month.\")\n], \"sunk\")\n\nsubscriptionStatus = Card([\n Stack([\n TextContent(\"Current Plan\", \"small\"),\n Tag(\"Monthly Pro\", null, \"md\", \"info\")\n ], \"row\", \"m\", \"start\", \"start\"),\n Separator(\"horizontal\"),\n Stack([\n TextContent(\"Next Billing Date\", \"small\"),\n TextContent(\"October 15, 2023\", \"default\")\n ], \"row\", \"m\", \"start\", \"start\")\n], \"card\")\n\nactionSection = Stack([\n TextContent(\"What would you like to do next?\", \"large-heavy\"),\n TextContent(\"Choose an option below to manage your account status.\", \"small\"),\n Separator(\"horizontal\"),\n Stack([\n upgradeCard,\n cancelCard\n ], \"row\", \"m\", \"center\", \"center\", true)\n], \"column\", \"xl\", \"center\", \"center\")\n\nupgradeCard = Card([\n Stack([\n TextContent(\"Upgrade to Annual\", \"large\"),\n Tag(\"BEST VALUE\", null, \"sm\", \"success\")\n ], \"column\", \"s\", \"center\", \"center\"),\n TextContent(\"Save 20% by switching to a yearly plan.\", \"small\"),\n Separator(\"horizontal\"),\n Button(\"Switch to Annual\", Action([@ToAssistant(\"I want to upgrade to the annual plan\")]), \"primary\", \"medium\")\n], \"card\", \"column\", \"m\", \"center\", \"center\")\n\ncancelCard = Card([\n Stack([\n TextContent(\"Cancel Subscription\", \"large\"),\n Tag(\"WARNING\", null, \"sm\", \"danger\")\n ], \"column\", \"s\", \"center\", \"center\"),\n TextContent(\"You will lose access to Pro features at the end of your cycle.\", \"small\"),\n Separator(\"horizontal\"),\n Button(\"Cancel Plan\", Action([@ToAssistant(\"I want to cancel my subscription\")]), \"secondary\", \"medium\", \"destructive\")\n], \"card\", \"column\", \"m\", \"center\", \"center\")","promptTokens":4571,"completionTokens":542,"latencyMs":33234,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/realistic","repeat":2,"output":"root = Stack([\n HeaderSection,\n SubscriptionStatus,\n ActionSection\n], \"column\", \"xl\", \"center\")\n\nHeaderSection = Card([\n CardHeader(\"Subscription Management\", \"Manage your current plan and billing cycle\"),\n Separator(\"horizontal\"),\n TextContent(\"Your current plan is active. You are currently on the **Monthly Pro Plan** and will be billed $29.00 on October 15, 2023.\")\n], \"sunk\")\n\nSubscriptionStatus = Card([\n Stack([\n TextContent(\"Current Status: Active\", \"large-heavy\"),\n Tag(\"Monthly\", null, \"md\", \"info\")\n ], \"row\", \"m\", \"center\", \"start\"),\n Separator(\"horizontal\"),\n Stack([\n TextContent(\"Next billing date: Oct 15, 2023\", \"small\"),\n TextContent(\"Payment method: Visa ending in •••• 4242\", \"small\")\n ], \"column\", \"none\", \"start\")\n], \"card\")\n\nActionSection = Stack([\n TextContent(\"What would you like to do next?\", \"large-heavy\"),\n Stack([\n UpgradeCard,\n CancelCard\n ], \"row\", \"m\", \"start\", \"center\", true)\n], \"column\", \"xl\", \"center\", \"center\")\n\nUpgradeCard = Card([\n Stack([\n CardHeader(\"Upgrade to Annual\", \"Save 20%\"),\n TextContent(\"Get 12 months for the price of 10. One single payment per year.\"),\n Separator(\"horizontal\"),\n Button(\"Upgrade to Annual\", Action([@ToAssistant(\"I want to upgrade to the annual plan\")]), \"primary\", \"medium\")\n ], \"column\", \"m\", \"center\", \"center\")\n], \"card\")\n\nCancelCard = Card([\n Stack([\n CardHeader(\"Cancel Subscription\", \"Lose access\"),\n TextContent(\"If you cancel, you will lose access to Pro features at the end of your billing cycle.\"),\n Separator(\"horizontal\"),\n Button(\"Cancel Plan\", Action([@ToAssistant(\"I want to cancel my subscription\")]), \"secondary\", \"destructive\", \"medium\")\n ], \"column\", \"m\", \"center\", \"center\")\n], \"card\")","promptTokens":4571,"completionTokens":509,"latencyMs":30571,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"button/realistic","repeat":1,"output":"root = Stack([\n SubscriptionHeader,\n CurrentPlanCard,\n Separator(\"horizontal\"),\n ActionSection\n], \"column\", \"xl\", \"center\")\n\nSubscriptionHeader = Card([\n CardHeader(\"Manage Subscription\", \"Control your billing and plan settings\"),\n TextContent(\"You are currently on the Monthly Pro Plan. Your next billing date is October 12, 2023.\")\n], \"clear\")\n\nCurrentPlanCard = Card([\n Stack([\n Tag(\"Active\", null, \"sm\", \"success\"),\n TextContent(\"Pro Monthly\", \"large-heavy\"),\n TextContent(\"€19.99 / month\", \"default\")\n ], \"column\", \"xs\", \"start\"),\n Separator(\"horizontal\"),\n TextContent(\"Includes: Unlimited projects, Advanced analytics, Priority support, and 50GB Cloud storage.\", \"small\")\n], \"sunk\")\n\nActionSection = Stack([\n TextContent(\"What would you like to do next?\", \"large\"),\n Stack([\n UpgradeCard,\n CancelCard\n ], \"row\", \"m\", \"start\", \"stretch\", true)\n], \"column\", \"l\", \"center\")\n\nUpgradeCard = Card([\n Stack([\n CardHeader(\"Upgrade to Annual\", \"Save 20% per year\"),\n TextContent(\"Get the best value with our yearly billing cycle.\", \"small\"),\n Separator(\"horizontal\"),\n TextContent(\"€192.00 / year\", \"large-heavy\"),\n Button(\"Switch to Annual\", Action([@ToAssistant(\"I want to upgrade to the annual plan\")]), \"primary\", \"normal\", \"large\")\n ], \"column\", \"m\", \"center\", \"center\")\n], \"card\")\n\nCancelCard = Card([\n Stack([\n CardHeader(\"Cancel Plan\", \"End your subscription\"),\n TextContent(\"You will lose access to Pro features at the end of your current period.\", \"small\"),\n Separator(\"horizontal\"),\n Button(\"Cancel Subscription\", Action([@ToAssistant(\"I want to cancel my subscription\")]), \"secondary\", \"destructive\", \"large\")\n ], \"column\", \"m\", \"center\", \"center\")\n], \"card\")","promptTokens":4571,"completionTokens":488,"latencyMs":45965,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/minimal","repeat":2,"output":"root = Stack([\n headerCard,\n mainContent,\n setupSteps\n], \"column\", \"l\")\n\nheaderCard = Card([\n CardHeader(\"New Laptop Setup Guide\", \"A comprehensive checklist for a smooth transition\"),\n TextContent(\"Follow these steps to ensure your new machine is secure, optimized, and ready for productivity.\")\n], \"card\", \"column\", \"m\")\n\nmainContent = Tabs([\n TabItem(\"essentials\", \"Essentials\", [essentialsCard]),\n TabItem(\"software\", \"Software & Apps\", [softwareCard]),\n TabItem(\"optimization\", \"Optimization\", [optimizationCard])\n])\n\nessentialsCard = Card([\n CardHeader(\"Core Essentials\", \"Security and System Integrity\"),\n Separator(\"horizontal\"),\n Stack([\n TextContent(\"1. Physical Inspection: Check for screen defects or chassis damage.\"),\n TextContent(\"2. OS Updates: Run system updates immediately to patch vulnerabilities.\"),\n TextContent(\"3. Security: Enable FileVault (macOS) or BitLocker (Windows) encryption.\"),\n TextContent(\"4. User Accounts: Create a standard user account for daily use and keep Admin separate.\")\n ], \"column\", \"s\")\n], \"sunk\")\n\nsoftwareCard = Card([\n CardHeader(\"Software & Apps\", \"Getting your tools ready\"),\n Separator(\"horizontal\"),\n Stack([\n TextContent(\"• Browser: Install Chrome, Firefox, or Brave.\"),\n TextContent(\"• Productivity: Set up Microsoft 365, Google Workspace, or Notion.\"),\n TextContent(\"• Communication: Slack, Discord, or Zoom.\"),\n TextContent(\"• Utilities: 7-Zip (Windows), Keka (macOS), or specialized drivers.\")\n ], \"column\", \"s\")\n], \"sunk\")\n\noptimizationCard = Card([\n CardHeader(\"Optimization & Cleanup\", \"Performance Tuning\"),\n Separator(\"horizontal\"),\n Stack([\n TextContent(\"• Bloatware: Uninstall pre-installed trial software (McAfee, etc.).\"),\n TextContent(\"• Startup Apps: Disable unnecessary programs in Task Manager/Activity Monitor.\"),\n TextContent(\"• Battery: Calibrate and set charging limits if supported.\"),\n TextContent(\"• Backup: Configure Time Machine (macOS) or OneDrive/Google Drive.\")\n ], \"column\", \"s\")\n], \"sunk\")\n\nsetupSteps = Card([\n CardHeader(\"Quick Setup Workflow\"),\n Steps([\n StepsItem(\"Unbox & Inspect\", \"Check for physical damage and verify all accessories are present.\"),\n StepsItem(\"System Boot\", \"Complete the initial OS setup (Region, Wi-Fi, Account).\"),\n StepsItem(\"Update Everything\", \"Update OS, then update all drivers and firmware.\"),\n StepsItem(\"Install Essentials\", \"Install your browser and primary security tools.\"),\n StepsItem(\"Data Migration\", \"Transfer files from your old device via cloud or external drive.\"),\n StepsItem(\"Final Polish\", \"Customize UI settings, dark mode, and notification preferences.\")\n ])\n], \"card\", \"column\", \"l\")","promptTokens":4542,"completionTokens":675,"latencyMs":23504,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/minimal","repeat":1,"output":"root = Stack([\n headerCard,\n mainContent\n], \"column\", \"m\")\n\nheaderCard = Card([\n CardHeader(\"New Laptop Setup Guide\", \"Essential checklist for a seamless start\"),\n TextContent(\"Follow these steps to ensure your new device is secure, optimized, and ready for work.\")\n], \"sunk\")\n\nmainContent = Stack([\n Steps([\n StepsItem(\"Unboxing & Physical Inspection\", \"Check for any physical damage, ensure all accessories (charger, cables) are included, and verify the model matches your order.\"),\n StepsItem(\"Initial Boot & OS Setup\", \"Connect to Wi-Fi, complete the OS setup (Windows/macOS/Linux), and sign in to your primary OS account.\"),\n StepsItem(\"System & Driver Updates\", \"Run Windows Update or macOS Software Update immediately. Check manufacturer websites (Dell, Lenovo, Apple, etc.) for specific driver/firmware updates.\"),\n StepsItem(\"Security Configuration\", \"Enable FileVault (macOS) or BitLocker (Windows). Set up biometric login (TouchID/FaceID) and ensure your firewall is active.\"),\n StepsItem(\"Software Installation\", \"Install your essential browser (Chrome/Firefox), communication tools (Slack/Discord), and productivity suites (Office/Adobe).\"),\n StepsItem(\"Cloud & Backup Setup\", \"Sync your files with OneDrive, iCloud, or Google Drive. Configure a local backup solution like Time Machine or an external drive.\")\n ]),\n categoryTabs\n], \"column\", \"l\")\n\ncategoryTabs = Tabs([\n TabItem(\"software\", \"Software & Apps\", [softwareCard]),\n TabItem(\"security\", \"Security Checklist\", [securityCard]),\n TabItem(\"optimization\", \"Optimization Tips\", [optimizationCard])\n])\n\nsoftwareCard = Card([\n CardHeader(\"Essential Software Categories\"),\n Table([\n Col(\"Category\", [\"Browsers\", \"Communication\", \"Productivity\", \"Utilities\", \"Dev Tools\"]),\n Col(\"Examples\", [\"Chrome, Firefox\", \"Slack, Zoom\", \"Office 365, Notion\", \"7-Zip, VLC\", \"VS Code, Git\"])\n ])\n])\n\nsecurityCard = Card([\n CardHeader(\"Security Hardening\"),\n Stack([\n Tag(\"Critical\", null, \"md\", \"danger\"),\n TextContent(\"Don't skip these security steps:\"),\n TagBlock([\"Enable Disk Encryption\", \"Set Up 2FA\", \"Update BIOS/Firmware\", \"Configure Firewall\", \"Install Antivirus/EDR\"]),\n Separator(\"horizontal\"),\n TextCallout(\"warning\", \"Pro Tip\", \"Avoid using the default 'Administrator' account for daily tasks; create a standard user account instead.\")\n ])\n])\n\noptimizationCard = Card([\n CardHeader(\"Performance Tuning\"),\n Stack([\n TextContent(\"Quick wins for a faster experience:\"),\n TagBlock([\"Disable Startup Apps\", \"Adjust Power Plan\", \"Clean Up Bloatware\", \"Set Refresh Rate\", \"Optimize Storage\"]),\n Separator(\"horizontal\"),\n CodeBlock(\"powershell\", \"# Example: Check system health (Windows)\\nGet-ComputerInfo | Select-Object OsName, OsVersion\")\n ])\n])","promptTokens":4542,"completionTokens":696,"latencyMs":31936,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/minimal","repeat":3,"output":"root = Stack([\n header,\n mainContent\n], \"column\", \"m\")\n\nheader = Card([\n CardHeader(\"New Laptop Setup Checklist\", \"Essential steps for a smooth transition and optimal performance\"),\n TextContent(\"Follow this guide to ensure your new device is secure, updated, and ready for work.\")\n], \"sunk\")\n\nmainContent = Stack([\n setupStages,\n proTips\n], \"column\", \"l\")\n\nsetupStages = Accordion([\n AccordionItem(\"initial\", \"Phase 1: Physical & OS Setup\", [\n Steps([\n StepsItem(\"Unboxing & Inspection\", \"Check for physical damage, screen defects, or loose components.\"),\n StepsItem(\"Power Connection\", \"Plug in your charger before the first boot to prevent shutdown during updates.\"),\n StepsItem(\"OS Configuration\", \"Follow the on-screen prompts for region, keyboard layout, and Wi-Fi.\"),\n StepsItem(\"Account Setup\", \"Create or sign in to your primary OS account (Microsoft, Apple, or Linux).\")\n ])\n ]),\n AccordionItem(\"security\", \"Phase 2: Security & Updates\", [\n Steps([\n StepsItem(\"System Updates\", \"Run OS updates immediately to patch critical security vulnerabilities.\"),\n StepsItem(\"Security Software\", \"Enable built-in protection (Windows Defender/FileVault) or install trusted antivirus.\"),\n StepsItem(\"Biometrics\", \"Set up Fingerprint or Face Recognition for faster, secure access.\"),\n StepsItem(\"Disk Encryption\", \"Ensure BitLocker (Windows) or FileVault (macOS) is active.\")\n ])\n ]),\n AccordionItem(\"software\", \"Phase 3: Software & Productivity\", [\n Steps([\n StepsItem(\"Browser Setup\", \"Install your preferred browser (Chrome, Firefox, Brave) and sync bookmarks.\"),\n StepsItem(\"Essential Apps\", \"Install productivity suites (Office/Google Drive), communication (Slack/Discord), and dev tools.\"),\n StepsItem(\"Driver Updates\", \"Check manufacturer websites (Dell, Lenovo, HP, etc.) for specific hardware drivers.\"),\n StepsItem(\"Cloud Sync\", \"Sign in to iCloud, OneDrive, or Dropbox to begin syncing files.\")\n ])\n ]),\n AccordionItem(\"maintenance\", \"Phase 4: Optimization & Cleanup\", [\n Steps([\n StepsItem(\"Bloatware Removal\", \"Uninstall pre-installed trial software and manufacturer 'helper' apps you won't use.\"),\n StepsItem(\"Startup Management\", \"Disable unnecessary apps from launching on boot to improve speed.\"),\n StepsItem(\"Battery Health\", \"Configure battery charging limits if your laptop supports them for long-term health.\"),\n StepsItem(\"Backup Routine\", \"Set up your first automated backup to an external drive or cloud service.\")\n ])\n ])\n])\n\nproTips = Card([\n CardHeader(\"Pro Tips\", \"Expert advice for power users\"),\n Stack([\n TextContent(\"• **Don't skip updates:** The first 30 minutes of a new laptop's life should be spent downloading updates.\", \"small\"),\n TextContent(\"• **Check Warranty:** Register your device on the manufacturer's site immediately.\", \"small\"),\n TextContent(\"• **Night Shift/Blue Light:** Enable blue light filters early to reduce eye strain.\", \"small\")\n ], \"column\", \"xs\")\n], \"clear\")","promptTokens":4542,"completionTokens":730,"latencyMs":44001,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/minimal","repeat":4,"output":"root = Stack([\n header,\n mainContent\n], \"column\", \"none\", \"start\", \"stretch\")\n\nheader = Card([\n CardHeader(\"New Laptop Setup Guide\", \"A comprehensive checklist to ensure your new device is secure, optimized, and ready for work.\"),\n Separator(\"horizontal\")\n], \"sunk\")\n\nmainContent = Stack([\n setupSteps,\n softwareChecklist,\n securityChecklist,\n optimizationTips\n], \"column\", \"m\")\n\nsetupSteps = Card([\n CardHeader(\"Step 1: Physical & OS Setup\"),\n Steps([\n StepsItem(\"Unbox & Inspect\", \"Check for any physical damage or screen defects immediately.\"),\n StepsItem(\"Initial Boot\", \"Follow the OS setup wizard (Region, Keyboard, Wi-Fi).\"),\n StepsItem(\"System Updates\", \"Run Windows Update or macOS Software Update until no more are found.\"),\n StepsItem(\"Account Setup\", \"Configure your user profile and enable biometric login (Fingerprint/Face ID).\")\n ])\n], \"card\")\n\nsoftwareChecklist = Card([\n CardHeader(\"Step 2: Essential Software\"),\n Tabs([\n TabItem(\"browsers\", \"Web Browsers\", [\n TagBlock([\"Chrome\", \"Firefox\", \"Brave\", \"Edge\"])\n ]),\n TabItem(\"productivity\", \"Productivity\", [\n TagBlock([\"Microsoft 365\", \"Google Workspace\", \"Notion\", \"Slack\"])\n ]),\n TabItem(\"utilities\", \"Utilities\", [\n TagBlock([\"7-Zip/Keka\", \"VLC Player\", \"Spotify\", \"Zoom\"])\n ])\n ])\n], \"card\")\n\nsecurityChecklist = Card([\n CardHeader(\"Step 3: Security Hardening\"),\n Stack([\n securityItem1,\n securityItem2,\n securityItem3\n ], \"column\", \"s\")\n], \"card\")\n\nsecurityItem1 = TextCallout(\"security\", \"Enable Disk Encryption\", \"Ensure FileVault (macOS) or BitLocker (Windows) is active.\")\nsecurityItem2 = TextCallout(\"security\", \"Antivirus/Firewall\", \"Verify that the built-in firewall is turned on.\")\nsecurityItem3 = TextCallout(\"security\", \"Backup Strategy\", \"Set up Time Machine or OneDrive/Google Drive sync immediately.\")\n\noptimizationTips = Card([\n CardHeader(\"Pro Tips\"),\n Accordion([\n AccordionItem(\"Battery Health\", [\n MarkDownRenderer(\"Avoid keeping the laptop plugged in at 100% all the time. Use 'Battery Charge Limit' settings if available to extend longevity.\")\n ]),\n AccordionItem(\"Bloatware Removal\", [\n MarkDownRenderer(\"Go to 'Add or Remove Programs' and uninstall pre-installed trial software from the manufacturer.\")\n ]),\n AccordionItem(\"Display Calibration\", [\n MarkDownRenderer(\"Adjust brightness and 'Night Light' / 'Night Shift' to reduce eye strain during long sessions.\")\n ])\n ])\n], \"card\")","promptTokens":4542,"completionTokens":668,"latencyMs":49364,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/realistic","repeat":2,"output":"root = Stack([onboardingHeader, onboardingSteps, onboardingFooter], \"column\", \"xl\")\n\nonboardingHeader = Card([\n CardHeader(\"IT Onboarding Checklist\", \"Welcome to the team! Complete these steps to get fully set up.\"),\n TextContent(\"Please follow the steps below in order. Once completed, you will have full access to all necessary company resources.\", \"small\")\n], \"sunk\")\n\nonboardingSteps = Steps([\n StepsItem(\"Account Activation\", \"Log in to your primary company email using the temporary credentials provided in your welcome packet.\"),\n StepsItem(\"Password Manager\", \"Download and install the company-approved password manager. Set up your master vault and import your initial credentials.\"),\n StepsItem(\"VPN Configuration\", \"Install the VPN client and connect using your corporate identity. This is required for accessing internal tools.\"),\n StepsItem(\"Two-Factor Authentication (2FA)\", \"Enroll your mobile device in the 2FA application to secure all your logins.\"),\n StepsItem(\"Laptop Encryption Check\", \"Verify that FileVault (macOS) or BitLocker (Windows) is active. Report any issues to IT Support immediately.\"),\n StepsItem(\"Software Audit\", \"Ensure all required software (Slack, Zoom, VS Code, etc.) is installed and launching correctly.\"),\n StepsItem(\"Team Channels\", \"Join your specific department channels in Slack and introduce yourself to the team.\"),\n StepsItem(\"IT Confirmation\", \"Once all steps are complete, click the button below to notify the IT team that you are ready.\")\n])\n\nonboardingFooter = Card([\n TextContent(\"Ready to go?\", \"large-heavy\"),\n TextContent(\"Click below to signal that your onboarding is complete.\", \"small\"),\n Buttons([\n Button(\"Notify IT Team\", Action([@ToAssistant(\"I have completed my IT onboarding checklist and am ready for my next tasks.\")]), \"primary\"),\n Button(\"Need Help?\", Action([@OpenUrl(\"https://support.company.com\")]), \"secondary\")\n ])\n], \"clear\")","promptTokens":4580,"completionTokens":437,"latencyMs":29221,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/minimal","repeat":0,"output":"root = Stack([\n Header,\n MainContent\n], \"column\", \"none\", \"start\", \"stretch\")\n\nHeader = Card([\n CardHeader(\"Laptop Setup Guide\", \"A comprehensive checklist for your new device\"),\n TextContent(\"Follow these steps to ensure your new laptop is secure, optimized, and ready for work or play.\", \"small\")\n], \"sunk\")\n\nMainContent = Stack([\n SetupSteps,\n SoftwareChecklist,\n SecurityChecklist\n], \"column\", \"m\")\n\nSetupSteps = Card([\n CardHeader(\"Phase 1: Physical & OS Setup\"),\n Steps([\n StepsItem(\"Unboxing & Inspection\", \"Check for physical damage, screen defects, or loose ports.\"),\n StepsItem(\"Initial Boot & Region\", \"Select language, region, and keyboard layout.\"),\n StepsItem(\"Wi-Fi Connection\", \"Connect to a secure network to begin updates.\"),\n StepsItem(\"OS Updates\", \"Run Windows Update or macOS Software Update until no more are found.\"),\n StepsItem(\"Account Setup\", \"Sign in with your primary OS account (Microsoft/Apple/Google).\")\n ])\n], \"card\")\n\nSoftwareChecklist = Card([\n CardHeader(\"Phase 2: Essential Software\"),\n Tabs([\n TabItem(\"browsers\", \"Browsers\", [\n TextContent(\"Install your preferred web browser and sync extensions.\", \"small\"),\n TagBlock([\"Chrome\", \"Firefox\", \"Brave\", \"Edge\"])\n ]),\n TabItem(\"productivity\", \"Productivity\", [\n TextContent(\"Set up your workflow tools.\", \"small\"),\n TagBlock([\"Office 365\", \"Google Workspace\", \"Slack\", \"Notion\"])\n ]),\n TabItem(\"media\", \"Media\", [\n TextContent(\"Install codecs and players.\", \"small\"),\n TagBlock([\"VLC\", \"Spotify\", \"Adobe Creative Cloud\"])\n ])\n ])\n], \"card\")\n\nSecurityChecklist = Card([\n CardHeader(\"Phase 3: Security & Maintenance\"),\n Stack([\n CheckItem(\"Enable Biometrics\", \"Set up Fingerprint or Face ID.\"),\n CheckItem(\"Disk Encryption\", \"Ensure BitLocker (Windows) or FileVault (macOS) is ON.\"),\n CheckItem(\"Backup Solution\", \"Configure OneDrive, iCloud, or Time Machine.\"),\n CheckItem(\"Antivirus/Firewall\", \"Verify system firewall is active.\")\n ], \"column\", \"s\")\n], \"card\")\n\nCheckItem = (label, desc) => Stack([\n Stack([\n Tag(\"✓\", \"sm\", \"success\"),\n TextContent(label, \"default-heavy\")\n ], \"row\", \"xs\", \"center\"),\n TextContent(desc, \"small\")\n], \"column\", \"xs\")\n\n// Note: In openui-lang, complex conditional logic for UI elements \n// like CheckItem is better handled via component definitions.\n// Since we can't use arrow functions in the language spec, \n// I will define them as standard components below.\n\nCheckItem1 = Stack([\n Stack([Tag(\"✓\", \"sm\", \"success\"), TextContent(\"Enable Biometrics\", \"default-heavy\")], \"row\", \"xs\", \"center\"),\n TextContent(\"Set up Fingerprint or Face ID.\", \"small\")\n], \"column\", \"xs\")\n\nCheckItem2 = Stack([\n Stack([Tag(\"✓\", \"sm\", \"success\"), TextContent(\"Disk Encryption\", \"default-heavy\")], \"row\", \"xs\", \"center\"),\n TextContent(\"Ensure BitLocker (Windows) or FileVault (macOS) is ON.\", \"small\")\n], \"column\", \"xs\")\n\nCheckItem3 = Stack([\n Stack([Tag(\"✓\", \"sm\", \"success\"), TextContent(\"Backup Solution\", \"default-heavy\")], \"row\", \"xs\", \"center\"),\n TextContent(\"Configure OneDrive, iCloud, or Time Machine.\", \"small\")\n], \"column\", \"xs\")\n\nCheckItem4 = Stack([\n Stack([Tag(\"✓\", \"sm\", \"success\"), TextContent(\"Antivirus/Firewall\", \"default-heavy\")], \"row\", \"xs\", \"center\"),\n TextContent(\"Verify system firewall is active.\", \"small\")\n], \"column\", \"xs\")\n\n// Re-defining Security Stack to use the specific components\nSecurityStack = Stack([CheckItem1, CheckItem2, CheckItem3, CheckItem4], \"column\", \"s\")\n\n// Final Assembly adjustments for the actual rendered output\nSecurityChecklist = Card([\n CardHeader(\"Phase 3: Security & Maintenance\"),\n SecurityStack\n], \"card\")","promptTokens":4542,"completionTokens":1025,"latencyMs":64667,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/realistic","repeat":3,"output":"root = Stack([onboardingContainer], \"column\", \"xl\", \"center\")\n\nonboardingContainer = Card([\n header,\n progressSection,\n stepsContainer,\n footer\n], \"card\")\n\nheader = CardHeader(\"IT Onboarding Checklist\", \"Welcome to the team! Please complete these steps to secure your workspace.\")\n\nprogressSection = Card([\n TextContent(\"Your Progress\", \"small\"),\n SingleStackedBarChart([\"Completed\", \"Remaining\"], [\n @Count(@Filter(stepsData.items, \"item\", \"completed\" == true)), \n @Count(@Filter(stepsData.items, \"item\", \"completed\" == false))\n ])\n], \"sunk\")\n\nstepsContainer = Steps(stepsData.items)\n\nstepsData = {\n items: [\n StepsItem(\"Account Activation\", \"Log in to your company email and SSO portal using the temporary credentials provided in your welcome email.\"),\n StepsItem(\"Password Manager\", \"Download 1Password and follow the setup guide to import your initial vault credentials.\"),\n StepsItem(\"VPN Configuration\", \"Install the GlobalProtect VPN client and connect to the 'Remote-Access' gateway.\"),\n StepsItem(\"Two-Factor Authentication (2FA)\", \"Set up Okta Verify on your mobile device to enable secure logins.\"),\n StepsItem(\"Laptop Encryption Check\", \"Verify that FileVault (macOS) or BitLocker (Windows) is active in your system settings.\"),\n StepsItem(\"Security Software Check\", \"Ensure the CrowdStrike sensor is running and showing a green status in your taskbar.\"),\n StepsItem(\"Team Channels\", \"Join the #it-announcements, #team-general, and your specific department Slack channels.\"),\n StepsItem(\"IT Support Setup\", \"Bookmark the IT Helpdesk portal and ensure you can access the ticketing system.\")\n ]\n}\n\nfooter = Stack([\n Separator(\"horizontal\"),\n Stack([\n Button(\"Need Help?\", Action([@ToAssistant(\"I am stuck on my IT onboarding. Can you help me with the next step?\")]), \"secondary\"),\n Button(\"Contact IT Support\", Action([@OpenUrl(\"https://helpdesk.company.com\")]), \"primary\")\n ], \"row\", \"m\", \"center\", \"end\")\n], \"column\", \"s\")","promptTokens":4580,"completionTokens":499,"latencyMs":42542,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/realistic","repeat":1,"output":"root = Stack([header, onboardingSteps, statusCard], \"column\", \"l\")\n\nheader = Card([\n CardHeader(\"IT Onboarding Checklist\", \"Welcome to the team! Please complete these steps to get your workstation ready.\"),\n TextContent(\"Follow the steps below to ensure your access and security settings are correctly configured.\", \"small\")\n], \"sunk\")\n\nonboardingSteps = Card([\n CardHeader(\"Your First Day Tasks\"),\n Steps([\n StepsItem(\"Account Activation\", \"Log in to your corporate email and SSO portal using the temporary credentials provided in your welcome packet.\"),\n StepsItem(\"Password Manager\", \"Install the company-approved password manager and import your initial credentials.\"),\n StepsItem(\"VPN Setup\", \"Install and configure the VPN client. Ensure you can connect to the 'Global-Secure' gateway.\"),\n StepsItem(\"Two-Factor Authentication (2FA)\", \"Set up your authenticator app on your mobile device for all security prompts.\"),\n StepsItem(\"Laptop Encryption Check\", \"Verify that FileVault (macOS) or BitLocker (Windows) is active and reporting 'Encrypted' status.\"),\n StepsItem(\"Security Software Update\", \"Run the 'IT-Check' utility to ensure antivirus and endpoint detection are up to date.\"),\n StepsItem(\"Team Channel Access\", \"Join your specific department channels in Slack/Teams (e.g., #dept-engineering, #it-help).\"),\n StepsItem(\"Hardware Audit\", \"Confirm all peripherals (monitor, dock, keyboard) are functioning and report any issues via the IT portal.\")\n ])\n], \"card\")\n\nstatusCard = Card([\n CardHeader(\"Onboarding Progress\"),\n Stack([\n TextContent(\"Estimated completion time: 45 minutes\", \"small\"),\n Separator(\"horizontal\"),\n Stack([\n TextContent(\"Tasks Completed\", \"small\"),\n TagBlock([\"Account Activation\", \"Password Manager\", \"VPN Setup\"])\n ], \"row\", \"m\", \"start\", \"center\"),\n Stack([\n TextContent(\"Remaining Tasks\", \"small\"),\n TagBlock([\"2FA\", \"Encryption\", \"Channels\", \"Hardware\"])\n ], \"row\", \"m\", \"start\", \"center\")\n ], \"column\", \"s\")\n], \"clear\")","promptTokens":4579,"completionTokens":494,"latencyMs":66354,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/adversarial","repeat":0,"output":"root = Stack([\n Card([\n CardHeader(\"Deployment Lifecycle Guide\", \"A comprehensive walkthrough of production readiness\"),\n Separator(),\n Steps([\n StepsItem(\"Environment Configuration\", \"Define your secrets and environment variables.\"),\n StepsItem(\"Dependency Installation\", \"Resolve and lock your project dependencies.\"),\n StepsItem(\"Build & Optimization\", \"Compile assets and prepare for production.\"),\n StepsItem(\"Staging Deployment\", \"Validate in a production-mirror environment.\"),\n StepsItem(\"Production Rollout\", \"Execute the final cutover to live users.\")\n ]),\n Separator(),\n Accordion([\n AccordionItem(\"1. Environment Configuration\", [\n TextCallout(\"info\", \"The Foundation\", \"Never hardcode credentials. Use environment variables to separate code from secrets.\"),\n CodeBlock(\"bash\", \"# Create a local .env file (DO NOT commit to git)\\necho 'DATABASE_URL=postgres://user:pass@host:5432/db'\\necho 'API_KEY=your_secret_key'\\necho 'NODE_ENV=production' > .env\"),\n TextContent(\"Why it matters: Security and portability. If secrets are in code, anyone with repo access owns your infrastructure.\", \"small\"),\n TextContent(\"If skipped: You risk leaking credentials in version control and will be unable to change configurations without re-deploying code.\", \"small\"),\n TextCallout(\"danger\", \"Critical Risk\", \"Leaked API keys can lead to massive financial loss and data breaches.\")\n ]),\n AccordionItem(\"2. Dependency Installation\", [\n TextCallout(\"info\", \"The Blueprint\", \"Ensure your environment exactly matches your development environment.\"),\n CodeBlock(\"bash\", \"# Use a lockfile to ensure reproducible builds\\nnpm ci\\n# or\\nyarn install --frozen-lockfile\"),\n TextContent(\"Why it matters: Deterministic builds. It ensures that the exact versions of libraries used in testing are the ones used in production.\", \"small\"),\n TextContent(\"If skipped: You might encounter 'it works on my machine' bugs caused by minor version mismatches in transitive dependencies.\", \"small\")\n ]),\n AccordionItem(\"3. Build & Optimization\", [\n TextCallout(\"info\", \"The Performance Phase\", \"Transforming source code into highly optimized production assets.\",\n CodeBlock(\"bash\", \"# Run the build script defined in package.json\\nnpm run build\\n\\n# Verify the build output folder exists\\nls -la ./dist\")),\n TextContent(\"Why it matters: Minification, tree-shaking, and transpilation reduce payload size and increase execution speed.\", \"small\"),\n TextContent(\"If skipped: Your application will be slow, consume excessive memory, and potentially expose unminified source code.\", \"small\")\n ]),\n AccordionItem(\"4. Staging Deployment\", [\n TextCallout(\"success\", \"The Safety Net\", \"Testing in a 'Pre-Production' environment that mimics live settings.\",\n CodeBlock(\"bash\", \"# Deploy to the staging cluster\\ngit checkout staging\\ngit pull origin staging\\n# Trigger CI/CD pipeline via CLI or Webhook\\ncurl -X POST https://ci.example.com/trigger -d 'env=staging'\"),\n TextContent(\"Why it matters: It catches environment-specific bugs (like database connection issues or DNS errors) before users see them.\", \"small\"),\n TextContent(\"If skipped: You are effectively using your real users as your first QA testers.\", \"small\")\n ]),\n AccordionItem(\"5. Production Rollout\", [\n TextCallout(\"warning\", \"The Final Step\", \"The controlled release to the live user base.\",\n CodeBlock(\"bash\", \"# Deploy to production cluster\\ngit checkout main\\ngit pull origin main\\n# Deploy using your provider (e.g., Vercel, AWS, Heroku)\\nvercel --prod\"),\n TextContent(\"Why it matters: Ensures the latest, tested code is served to the global audience.\", \"small\"),\n TextContent(\"If skipped: Your application remains outdated or, if attempted manually without a pipeline, risks downtime and inconsistent states.\", \"small\")\n ])\n ])\n ])\n], \"column\", \"m\")","promptTokens":4561,"completionTokens":922,"latencyMs":37075,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/adversarial","repeat":1,"output":"root = Stack([\n Card([\n CardHeader(\"Deployment Lifecycle Guide\", \"A comprehensive walkthrough for production-ready deployments\"),\n Separator(),\n Steps([\n StepsItem(\"Environment Configuration\", \"Setting up variables and secrets to ensure the app runs in the correct context.\"),\n StepsItem(\"Dependency Installation\", \"Fetching the exact software packages and libraries required for the runtime.\"),\n StepsItem(\"Build & Compilation\", \"Transforming source code into optimized, production-ready artifacts.\"),\n StepsItem(\"Automated Testing\", \"Verifying that the new code meets quality and security standards.\"),\n StepsItem(\"Staging Deployment\", \"Deploying to a mirror of production to catch environment-specific bugs.\"),\n StepsItem(\"Production Rollout\", \"The final release to the live user base.\")\n ]),\n Separator(),\n Tabs([\n TabItem(\"details\", \"Deep Dive\", [\n Accordion([\n AccordionItem(\"1. Environment Configuration\", [\n TextContent(\"Why it matters: Prevents hardcoded secrets from leaking and ensures the app connects to the right database/services.\", \"default\"),\n TextCallout(\"danger\", \"If skipped: You risk leaking API keys in your repo or connecting your local dev DB to your production environment, causing data corruption.\", \"neutral\"),\n CodeBlock(\"bash\", \"export NODE_ENV=production\\nexport DATABASE_URL='postgres://user:password@prod-db:5432/mydb'\\nexport API_KEY='your-secret-key'\"),\n Separator()\n ]),\n AccordionItem(\"2. Dependency Installation\", [\n TextContent(\"Why it matters: Ensures all third-party libraries are present and versions are locked to prevent 'works on my machine' syndrome.\", \"default\"),\n TextCallout(\"warning\", \"If skipped: The application will crash immediately upon startup due to 'Module Not Found' errors.\", \"neutral\"),\n CodeBlock(\"bash\", \"npm ci\\n# OR\\npip install -r requirements.txt\"),\n Separator()\n ]),\n AccordionItem(\"3. Build & Compilation\", [\n TextContent(\"Why it matters: Minifies code, transpiles modern syntax for compatibility, and optimizes assets for speed.\", \"default\"),\n TextCallout(\"info\", \"If skipped: Your application will be significantly slower, and modern JS features might break in older browsers.\", \"neutral\"),\n CodeBlock(\"bash\", \"npm run build\\n# OR\\nmake build\"),\n Separator()\n ]),\n AccordionItem(\"4. Automated Testing\", [\n TextContent(\"Why it matters: Provides a safety net to ensure new changes haven't broken existing functionality (regressions).\", \"default\"),\n TextCallout(\"error\", \"If skipped: You will likely deploy bugs that frustrate users and potentially cause downtime.\", \"neutral\"),\n CodeBlock(\"bash\", \"npm test\\n# OR\\npytest\"),\n Separator()\n ]),\n AccordionItem(\"5. Staging Deployment\", [\n TextContent(\"Why it matters: Validates that the build process and environment variables work in a real cloud environment.\", \"default\"),\n TextCallout(\"warning\", \"If skipped: You lose the last chance to catch 'environment drift' (differences between your PC and the server).\", \"neutral\"),\n CodeBlock(\"bash\", \"git checkout staging\\ngit push origin staging\"),\n Separator()\n ]),\n AccordionItem(\"6. Production Rollout\", [\n TextContent(\"Why it matters: The actual delivery of value to your customers.\", \"default\"),\n TextCallout(\"success\", \"If skipped: You aren't deploying! But if done poorly without a rollback plan, you risk major outages.\", \"neutral\"),\n CodeBlock(\"bash\", \"git checkout main\\ngit merge staging\\ngit push origin main\\n# Followed by your CI/CD trigger (e.g., Vercel, AWS, Heroku)\"),\n ])\n ])\n ]),\n TabItem(\"checklist\", \"Quick Checklist\", [\n Card([\n TextContent(\"Use this to verify your pipeline before hitting 'Merge'.\", \"small\"),\n Separator(),\n Table([\n Col(\"Phase\", [\"Config\", \"Deps\", \"Build\", \"Test\", \"Staging\", \"Prod\"]),\n Col(\"Status\", [\"✅\", \"✅\", \"⏳\", \"❌\", \"❌\", \"❌\"])\n ])\n ], \"sunk\")\n ])\n ])\n ])\n], \"column\", \"m\")","promptTokens":4561,"completionTokens":982,"latencyMs":39786,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/realistic","repeat":0,"output":"root = Stack([onboardingContainer, welcomeHeader, checklistCard, statusCard], \"column\", \"xl\", \"center\")\n\nwelcomeHeader = Card([\n CardHeader(\"IT Onboarding\", \"First Day Checklist\"),\n TextContent(\"Welcome to the team! Please complete these steps to ensure your workstation is secure and ready for work.\", \"small\")\n], \"sunk\")\n\nonboardingContainer = Stack([\n checklistCard,\n Separator(\"horizontal\"),\n statusCard\n], \"column\", \"l\")\n\nchecklistCard = Card([\n CardHeader(\"Your Checklist\", \"Complete all steps to finish setup\"),\n Steps([\n StepsItem(\"Account Activation\", \"Log in to your company email and verify your identity via the welcome link.\"),\n StepsItem(\"Password Manager\", \"Install the company-approved password manager and import your initial credentials.\"),\n StepsItem(\"VPN Setup\", \"Configure the VPN client and test your connection to the secure office network.\"),\n StepsItem(\"2FA Configuration\", \"Link your mobile device to the authentication app for all secure logins.\"),\n StepsItem(\"Laptop Encryption\", \"Run the 'Security Check' utility to verify FileVault/BitLocker is active.\"),\n StepsItem(\"Software Updates\", \"Ensure your OS and security patches are fully up to date.\"),\n StepsItem(\"Team Channels\", \"Join your department's Slack/Teams channels using the provided directory.\"),\n StepsItem(\"IT Support Sync\", \"Send a quick 'All Clear' message to the IT Helpdesk channel.\")\n ])\n])\n\nstatusCard = Card([\n CardHeader(\"Current Progress\", \"Real-time status\"),\n Tabs([\n TabItem(\"summary\", \"Summary\", [progressSummary]),\n TabItem(\"details\", \"Detailed Status\", [statusTable])\n ])\n])\n\nprogressSummary = Stack([\n TextContent(\"Completion Rate\", \"small\"),\n SingleStackedBarChart([\"Progress\"], [Series(\"Completed\", [62])]),\n TextContent(\"6 of 8 steps completed\", \"small-heavy\")\n], \"column\", \"m\", \"center\")\n\nstatusTable = Table([\n Col(\"Task\", onboardingTasks.map(t => t.name)),\n Col(\"Status\", onboardingTasks.map(t => t.status)),\n Col(\"Action\", onboardingTasks.map(t => t.action))\n])\n\nonboardingTasks = [\n {name: \"Account Activation\", status: Tag(\"Success\", null, \"sm\", \"success\"), action: Button(\"Done\", null, \"tertiary\")},\n {name: \"Password Manager\", status: Tag(\"Success\", null, \"sm\", \"success\"), action: Button(\"Done\", null, \"tertiary\")},\n {name: \"VPN Setup\", status: Tag(\"Success\", null, \"sm\", \"success\"), action: Button(\"Done\", null, \"tertiary\")},\n {name: \"2FA Configuration\", status: Tag(\"Success\", null, \"sm\", \"success\"), action: Button(\"Done\", null, \"tertiary\")},\n {name: \"Laptop Encryption\", status: Tag(\"In Progress\", null, \"sm\", \"warning\"), action: Button(\"Verify\", Action([@ToAssistant(\"I am checking encryption status\")]), \"primary\")},\n {name: \"Software Updates\", status: Tag(\"Pending\", null, \"sm\", \"neutral\"), action: Button(\"Start\", null, \"secondary\")},\n {name: \"Team Channels\", status: Tag(\"Pending\", null, \"sm\", \"neutral\"), action: Button(\"Join\", null, \"secondary\")},\n {name: \"IT Support Sync\", status: Tag(\"Pending\", null, \"sm\", \"neutral\"), action: Button(\"Notify\", null, \"secondary\")}\n]","promptTokens":4580,"completionTokens":819,"latencyMs":91876,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/realistic","repeat":4,"output":"root = Stack([header, onboardingSteps, summaryCard], \"column\", \"xl\")\n\nheader = Card([\n CardHeader(\"IT Onboarding Checklist\", \"Welcome to the team! Complete these steps to get fully set up.\"),\n TextContent(\"Please follow the sequence below to ensure your workstation and accounts are secure and ready for use.\", \"small\")\n], \"sunk\")\n\nonboardingSteps = Steps([\n StepsItem(\"Account Activation\", \"Log in to your corporate email and SSO dashboard using the temporary credentials provided in your welcome packet.\"),\n StepsItem(\"Password Manager\", \"Install the company-approved password manager. Set a strong master password and import your initial credentials.\"),\n StepsItem(\"VPN Configuration\", \"Download and install the VPN client. Test your connection from a home network to ensure remote access is working.\"),\n StepsItem(\"Multi-Factor Authentication (2FA)\", \"Set up your authentication app (e.g., Okta, Duo, or Microsoft Authenticator) on your mobile device.\"),\n StepsItem(\"Laptop Encryption Check\", \"Verify that FileVault (macOS) or BitLocker (Windows) is active. Check your system settings to confirm encryption is 'On'.\"),\n StepsItem(\"Security Software Audit\", \"Ensure the endpoint protection (Antivirus/EDR) is running and the dashboard shows 'Protected' status.\"),\n StepsItem(\"Team Channel Access\", \"Join your department's Slack/Teams channels. Verify you can see the #announcements and #it-help channels.\"),\n StepsItem(\"Final IT Sign-off\", \"Once all steps are complete, submit your confirmation via the IT portal or notify your manager.\")\n])\n\nsummaryCard = Card([\n CardHeader(\"Support & Resources\"),\n Stack([\n TextContent(\"Need help? Contact the IT Helpdesk immediately.\", \"default\"),\n Separator(\"horizontal\"),\n Stack([\n Tag(\"Email: help@company.com\", null, \"sm\", \"info\"),\n Tag(\"Slack: #it-support\", null, \"sm\", \"info\"),\n Tag(\"Phone: ext. 555\", null, \"sm\", \"info\")\n ], \"row\", \"s\", \"start\", true)\n ], \"column\", \"s\")\n], \"card\")","promptTokens":4579,"completionTokens":487,"latencyMs":63241,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/adversarial","repeat":2,"output":"root = Stack([\n Card([\n CardHeader(\"Deployment Lifecycle Guide\", \"A comprehensive walkthrough for production-ready deployment\"),\n Steps([\n StepsItem(\"Environment Configuration\", \"Set up your environment variables and secrets.\"),\n StepsItem(\"Dependency Installation\", \"Install necessary packages and build assets.\"),\n StepsItem(\"Automated Testing\", \"Run unit and integration tests.\"),\n StepsItem(\"Build & Containerization\", \"Create a production build and Docker image.\"),\n StepsItem(\"Deployment & Health Check\", \"Push to server and verify service availability.\")\n ]),\n Separator(\"horizontal\"),\n Tabs([\n TabItem(\"details\", \"Deep Dive\", [\n Accordion([\n AccordionItem(\"1. Environment Configuration\", [\n TextContent(\"### Why it matters\", \"Production environments require different credentials (DB URLs, API keys) than local development. This ensures security and correct connectivity.\"),\n TextCallout(\"warning\", \"Risk of Skipping\", \"If skipped, your application will attempt to connect to local or development databases, leading to data corruption or immediate runtime crashes.\"),\n CodeBlock(\"bash\", \"cp .env.example .env.production\\nnvim .env.production # Edit with your real secrets\"),\n Separator()\n ]),\n AccordionItem(\"2. Dependency Installation\", [\n TextContent(\"### Why it matters\", \"Ensures that the exact versions of libraries used in development are present. It prepares the runtime environment.\"),\n TextCallout(\"warning\", \"Risk of Skipping\", \"Missing dependencies lead to 'Module Not Found' errors or inconsistent behavior due to version mismatches.\"),\n CodeBlock(\"bash\", \"npm install --frozen-lockfile\\n# or\\npip install -r requirements.txt\"),\n Separator()\n ]),\n AccordionItem(\"3. Automated Testing\", [\n TextContent(\"### Why it matters\", \"Verifies that new changes haven't broken existing functionality (regression) and that the code meets quality standards.\"),\n TextCallout(\"error\", \"Risk of Skipping\", \"You risk deploying bugs that could cause downtime, data loss, or security vulnerabilities to your end users.\"),\n CodeBlock(\"bash\", \"npm test\\n# or\\npytest tests/\"),\n Separator()\n ]),\n AccordionItem(\"4. Build & Containerization\", [\n TextContent(\"### Why it matters\", \"Compiles code (like TypeScript to JS) and bundles assets into a portable, immutable artifact (Docker image) that runs the same everywhere.\"),\n TextCallout(\"info\", \"Risk of Skipping\", \"Without a build, you may deploy unoptimized source code or encounter 'it works on my machine' issues due to OS differences.\"),\n CodeBlock(\"bash\", \"npm run build\\ndocker build -t my-app:latest .\\n# or\\ngo build -o app main.go\"),\n Separator()\n ]),\n AccordionItem(\"5. Deployment & Health Check\", [\n TextContent(\"### Why it matters\", \"The actual transfer of code to the host. The health check confirms the application is actually responding to requests after the push.\"),\n TextCallout(\"success\", \"Risk of Skipping\", \"Without a health check, you might assume a deployment succeeded while the service is actually in a 'CrashLoopBackOff' state.\"),\n CodeBlock(\"bash\", \"docker push my-registry/my-app:latest\\nkubectl rollout status deployment/my-app\\n# Verify with:\\ncurl -I https://api.myapp.com/health\"),\n ])\n ])\n ]),\n TabItem(\"checklist\", \"Quick Checklist\", [\n Card([\n TextContent(\"Ensure all items are completed before hitting 'Deploy' in your CI/CD pipeline.\", \"small\"),\n Separator(),\n Stack([\n TextContent(\"☐ Environment Variables Set\"),\n TextContent(\"☐ Dependencies Installed\"),\n TextContent(\"☐ Tests Passed (100%)\"),\n TextContent(\"☐ Build Artifact Created\"),\n TextContent(\"☐ Container Image Pushed\"),\n TextContent(\"☐ Health Check Endpoint 200 OK\"),\n ], \"column\", \"xs\")\n ], \"sunk\")\n ])\n ])\n ])\n])","promptTokens":4561,"completionTokens":911,"latencyMs":32459,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/adversarial","repeat":4,"output":"root = Stack([\n Card([\n CardHeader(\"Deployment Roadmap\", \"A comprehensive guide to production-ready deployment\"),\n Steps([\n StepsItem(\"Environment Configuration\", \"Set up your environment variables and secrets before code execution.\"),\n StepsItem(\"Dependency Installation\", \"Install all necessary packages and build tools.\"),\n StepsItem(\"Build & Artifact Creation\", \"Compile the source code into a production-ready bundle.\"),\n StepsItem(\"Staging Deployment\", \"Deploy to a non-production environment for final validation.\"),\n StepsItem(\"Production Rollout\", \"Execute the final deployment to the live environment.\")\n ]),\n Separator(\"horizontal\"),\n Accordion([\n AccordionItem(\"1. Environment Configuration\", [\n TextContent(\"### Why it matters\", \"Production environments require different credentials (API keys, DB URLs) than local development. This ensures security and prevents accidental data corruption in your local DB.\"),\n TextContent(\"### If you skip it\", \"Your application will likely crash on startup due to missing credentials, or worse, it will attempt to connect to your development database, causing data leaks or corruption.\"),\n CodeBlock(\"bash\", \"cp .env.example .env.production\\nnano .env.production\\n# Add your production secrets here\"),\n Tag(\"Security\", null, \"sm\", \"critical\")\n ]),\n AccordionItem(\"2. Dependency Installation\", [\n TextContent(\"### Why it matters\", \"Modern applications rely on a specific tree of libraries. Ensuring a clean, locked installation prevents 'it works on my machine' syndrome.\"),\n TextContent(\"### If you skip it\", \"You will encounter 'Module Not Found' errors or, more dangerously, runtime errors caused by version mismatches between your local machine and the server.\"),\n CodeBlock(\"bash\", \"npm ci\\n# or for python\\npip install -r requirements.txt\"),\n Tag(\"Stability\", null, \"sm\", \"important\")\n ]),\n AccordionItem(\"3. Build & Artifact Creation\", [\n TextContent(\"### Why it matters\", \"Minification, tree-shaking, and transpilation reduce file size and improve performance. It converts human-readable code into machine-optimized code.\"),\n TextContent(\"### If you skip it\", \"The application will be significantly slower, use more memory, and might fail to run if the production environment doesn't support modern syntax (like ESNext).\"),\n CodeBlock(\"bash\", \"npm run build\\n# This creates the /dist or /build folder\"),\n Tag(\"Performance\", null, \"sm\", \"high\")\n ]),\n AccordionItem(\"4. Staging Deployment\", [\n TextContent(\"### Why it matters\", \"Staging is a 'pre-flight' check. It mimics production perfectly, allowing you to catch infrastructure-level bugs (like CORS issues or permission errors) without impacting users.\"),\n TextContent(\"### If you skip it\", \"The first time you encounter an infrastructure bug will be during a live rollout, leading to immediate downtime and user frustration.\"),\n CodeBlock(\"bash\", \"terraform apply -var-file=staging.tfvars\\n# or\\naws s3 sync ./dist s3://staging-bucket\"),\n Tag(\"Quality\", null, \"sm\", \"essential\")\n ]),\n AccordionItem(\"5. Production Rollout\", [\n TextContent(\"### Why it matters\", \"This is the final step where the validated artifacts are pushed to the live traffic. Using a blue-green or canary deployment ensures zero downtime.\"),\n TextContent(\"### If you skip it\", \"N/A - This is the goal of the entire process.\"),\n CodeBlock(\"bash\", \"npm run deploy\\n# or using a CI/CD tool\\ngit push origin main\"),\n Tag(\"Release\", null, \"sm\", \"final\")\n ])\n ])\n ])\n], \"column\", \"m\")","promptTokens":4561,"completionTokens":842,"latencyMs":33694,"finishReason":"stop"} +{"model":"google/gemma-4-26b-a4b-it","format":"openui-v2","scenario":"tasklist/adversarial","repeat":3,"output":"root = Stack([\n Card([\n CardHeader(\"Deployment Pipeline Guide\", \"Follow these steps to ensure a production-ready release\"),\n Steps([\n StepsItem(\"Environment Configuration\", \"Set up your environment variables and secrets to ensure the application has the necessary credentials and configurations to run in the target environment.\"),\n StepsItem(\"Dependency Installation\", \"Install all required packages and dependencies to ensure the application has the necessary libraries and modules to function correctly.\"),\n StepsItem(\"Build Process\", \"Compile and bundle your application into a production-ready state, optimizing for performance and security.\"),\n StepsItem(\"Automated Testing\", \"Run your test suite to verify that the new deployment does not introduce regressions or break existing functionality.\"),\n StepsItem(\"Deployment to Production\", \"Push the built and tested artifacts to your production servers or cloud provider to make the changes live.\")\n ]),\n Separator(\"horizontal\"),\n Accordion([\n AccordionItem(\"1. Environment Configuration\", [\n TextCallout(\"info\", \"Critical Step\", \"Configuration defines how your app talks to databases, APIs, and services.\"),\n MarkDownRenderer(\"### Why it matters\\nEnvironment variables separate code from configuration. This allows the same code to run in 'Development', 'Staging', and 'Production' without modification.\\n\\n### If you skip it\\nThe application will likely crash on startup due to missing connection strings, or worse, attempt to connect to a development database while in production, leading to data corruption.\\n\\n### Commands\\n```bash\\n# Create a local .env file (do NOT commit this to Git)\\necho \\\"DATABASE_URL=postgres://user:pass@host:5432/db\\\" > .env\\necho \\\"API_KEY=your_secret_key\\\" >> .env\\n```\"),\n TagBlock([\"Security\", \"Configuration\"])\n ]),\n AccordionItem(\"2. Dependency Installation\", [\n TextCallout(\"warning\", \"Prerequisite\", \"Missing dependencies are the #1 cause of 'it works on my machine' errors.\"),\n MarkDownRenderer(\"### Why it matters\\nModern applications rely on hundreds of external libraries. Ensuring these are installed exactly as specified in your lockfile ensures consistency across all environments.\\n\\n### If you skip it\\nYou will encounter 'Module Not Found' errors during the build or runtime, causing the deployment to fail immediately.\\n\\n### Commands\\n```bash\\n# For Node.js environments\\nnpm ci\\n\\n# For Python environments\\npip install -r requirements.txt\\n```\"),\n TagBlock([\"Stability\", \"Environment\"])\n ]),\n AccordionItem(\"3. Build Process\", [\n TextCallout(\"info\", \"Optimization\", \"The build step transforms source code into efficient, minified production assets.\"),\n MarkDownRenderer(\"### Why it matters\\nBuilding minifies code, removes comments, and optimizes images. This significantly reduces the payload size sent to users, improving load times and SEO.\\n\\n### If you skip it\\nYour application will be much slower, consume more bandwidth, and might expose your original source code structure to the public.\\n\\n### Commands\\n```bash\\n# For React/Vue/Next.js\\nnpm run build\\n\\n# For Go\\ngo build -o myapp main.go\\n```\"),\n TagBlock([\"Performance\", \"Security\"])\n ]),\n AccordionItem(\"4. Automated Testing\", [\n TextCallout(\"danger\", \"Quality Assurance\", \"Testing is your safety net against breaking existing features.\"),\n MarkDownRenderer(\"### Why it matters\\nAutomated tests (Unit, Integration, and E2E) provide immediate feedback on whether your changes are safe. It is the only way to scale development without fear.\\n\\n### If you skip it\\nYou risk deploying 'silent failures'—bugs that don't crash the app but corrupt data or break user workflows, which are much harder and more expensive to fix later.\\n\\n### Commands\\n```bash\\n# For Jest/Vitest\\nnpm test\\n\\n# For Pytest\\npytest tests/\\n```\"),\n TagBlock([\"Quality\", \"Risk-Mitigation\"])\n ]),\n AccordionItem(\"5. Deployment to Production\", [\n TextCallout(\"success\", \"Final Step\", \"The moment of truth where your code meets the real world.\"),\n MarkDownRenderer(\"### Why it matters\\nThis step moves your verified artifacts to the live infrastructure. Using a controlled deployment method (like Blue/Green or Rolling updates) ensures zero downtime.\\n\\n### If you skip it\\nYour changes simply aren't live. If you do it manually (e.g., dragging files via FTP), you lose the ability to roll back easily if something goes wrong.\\n\\n### Commands\\n```bash\\n# Using Docker\\ndocker compose up -d --build\\n\\n# Using AWS CLI\\naws s3 sync ./dist s3://my-production-bucket\\n```\"),\n TagBlock([\"Release\", \"Operations\"])\n ])\n ])\n ], \"card\")\n], \"column\", \"m\")","promptTokens":4561,"completionTokens":1115,"latencyMs":40305,"finishReason":"stop"} diff --git a/benchmarks/genui-formats/results/results.json b/benchmarks/genui-formats/results/results.json index f2e5697..7cf2b95 100644 --- a/benchmarks/genui-formats/results/results.json +++ b/benchmarks/genui-formats/results/results.json @@ -1,5 +1,5 @@ { - "generatedAt": "2026-07-28T07:26:52.650Z", + "generatedAt": "2026-08-07T14:19:12.249Z", "cells": [ { "model": "google/gemma-4-26b-a4b-it", @@ -5166,6 +5166,1040 @@ "parse-error", "parse-error" ] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "contact-form/minimal", + "family": "contact-form", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 478.8, + "avgPromptTokens": 6942, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "contact-form/realistic", + "family": "contact-form", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 3, + "avgOutputTokens": 1044.2, + "avgPromptTokens": 7028, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "contact-form/adversarial", + "family": "contact-form", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 934, + "avgPromptTokens": 7009, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "data-table/minimal", + "family": "data-table", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 473, + "avgPromptTokens": 6943, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "data-table/realistic", + "family": "data-table", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 2831, + "avgPromptTokens": 6999, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "chart/minimal", + "family": "chart", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 831.6, + "avgPromptTokens": 6952, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "chart/realistic", + "family": "chart", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 1862.6, + "avgPromptTokens": 6979, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "data-table/adversarial", + "family": "data-table", + "variant": "adversarial", + "repeats": 5, + "truncated": 5, + "scored": 0, + "renderable": 0, + "everyTime": false, + "stableShape": false, + "distinctShapes": 0, + "avgOutputTokens": 8192, + "avgPromptTokens": 7016, + "failures": [ + "truncated", + "truncated", + "truncated", + "truncated", + "truncated" + ] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "chart/adversarial", + "family": "chart", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 1408.4, + "avgPromptTokens": 7008, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "callout/minimal", + "family": "callout", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 611.8, + "avgPromptTokens": 6945, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "callout/realistic", + "family": "callout", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 2524, + "avgPromptTokens": 7000, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "callout/adversarial", + "family": "callout", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 1932.6, + "avgPromptTokens": 6971, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "button/minimal", + "family": "button", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 2, + "avgOutputTokens": 533.2, + "avgPromptTokens": 6943, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "button/adversarial", + "family": "button", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": true, + "distinctShapes": 1, + "avgOutputTokens": 24.4, + "avgPromptTokens": 6967, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "button/realistic", + "family": "button", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 1053, + "avgPromptTokens": 6985, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "tasklist/minimal", + "family": "tasklist", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 3833.2, + "avgPromptTokens": 6945, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "tasklist/realistic", + "family": "tasklist", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 3005.6, + "avgPromptTokens": 7002, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "contact-form/minimal", + "family": "contact-form", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": true, + "distinctShapes": 1, + "avgOutputTokens": 184.6, + "avgPromptTokens": 4389, + "failures": [] + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "scenario": "tasklist/adversarial", + "family": "tasklist", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 6676.6, + "avgPromptTokens": 6969, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "contact-form/realistic", + "family": "contact-form", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 4, + "avgOutputTokens": 457, + "avgPromptTokens": 4449, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "contact-form/adversarial", + "family": "contact-form", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 4, + "avgOutputTokens": 455.2, + "avgPromptTokens": 4433, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "data-table/minimal", + "family": "data-table", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 4, + "avgOutputTokens": 161, + "avgPromptTokens": 4389, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "data-table/realistic", + "family": "data-table", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 4, + "avgOutputTokens": 482.2, + "avgPromptTokens": 4425, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "chart/minimal", + "family": "chart", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 2, + "avgOutputTokens": 150.8, + "avgPromptTokens": 4392, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "data-table/adversarial", + "family": "data-table", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 4, + "everyTime": false, + "stableShape": false, + "distinctShapes": 4, + "avgOutputTokens": 4228.8, + "avgPromptTokens": 4438, + "failures": [ + "broken-reference" + ] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "chart/realistic", + "family": "chart", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 4, + "avgOutputTokens": 255.6, + "avgPromptTokens": 4413, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "chart/adversarial", + "family": "chart", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 3, + "avgOutputTokens": 306.6, + "avgPromptTokens": 4433, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "callout/minimal", + "family": "callout", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": true, + "distinctShapes": 1, + "avgOutputTokens": 40.6, + "avgPromptTokens": 4389, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "callout/realistic", + "family": "callout", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 257.2, + "avgPromptTokens": 4421, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "callout/adversarial", + "family": "callout", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 154, + "avgPromptTokens": 4403, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "button/minimal", + "family": "button", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": true, + "distinctShapes": 1, + "avgOutputTokens": 89.6, + "avgPromptTokens": 4388, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "button/realistic", + "family": "button", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 2, + "avgOutputTokens": 172, + "avgPromptTokens": 4417, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "button/adversarial", + "family": "button", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": true, + "distinctShapes": 1, + "avgOutputTokens": 16.2, + "avgPromptTokens": 4404, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "tasklist/minimal", + "family": "tasklist", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 911.2, + "avgPromptTokens": 4388, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "tasklist/realistic", + "family": "tasklist", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 691.6, + "avgPromptTokens": 4425, + "failures": [] + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "scenario": "tasklist/adversarial", + "family": "tasklist", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 4, + "avgOutputTokens": 2213.4, + "avgPromptTokens": 4407, + "failures": [] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "contact-form/minimal", + "family": "contact-form", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 3, + "avgOutputTokens": 340.8, + "avgPromptTokens": 4542.8, + "failures": [] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "contact-form/realistic", + "family": "contact-form", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 4, + "everyTime": false, + "stableShape": false, + "distinctShapes": 2, + "avgOutputTokens": 500.6, + "avgPromptTokens": 4603, + "failures": [ + "broken-reference" + ] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "contact-form/adversarial", + "family": "contact-form", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 2, + "everyTime": false, + "stableShape": false, + "distinctShapes": 2, + "avgOutputTokens": 411.6, + "avgPromptTokens": 4586.4, + "failures": [ + "broken-reference", + "broken-reference", + "broken-reference", + "truncated", + "broken-reference", + "broken-reference", + "broken-reference", + "schema-error", + "schema-error", + "schema-error", + "broken-reference", + "schema-error" + ] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "data-table/minimal", + "family": "data-table", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 3, + "avgOutputTokens": 184.4, + "avgPromptTokens": 4542.8, + "failures": [] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "data-table/realistic", + "family": "data-table", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 3, + "everyTime": false, + "stableShape": false, + "distinctShapes": 3, + "avgOutputTokens": 863.6, + "avgPromptTokens": 4578, + "failures": [ + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error" + ] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "data-table/adversarial", + "family": "data-table", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 1112.8, + "avgPromptTokens": 4590.8, + "failures": [] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "chart/minimal", + "family": "chart", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 3, + "avgOutputTokens": 458.6, + "avgPromptTokens": 4546, + "failures": [] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "chart/realistic", + "family": "chart", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 2, + "everyTime": false, + "stableShape": false, + "distinctShapes": 2, + "avgOutputTokens": 659.2, + "avgPromptTokens": 4567, + "failures": [ + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "broken-reference", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error" + ] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "chart/adversarial", + "family": "chart", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 4, + "everyTime": false, + "stableShape": false, + "distinctShapes": 4, + "avgOutputTokens": 494.8, + "avgPromptTokens": 4600.4, + "failures": [ + "unknown-component" + ] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "callout/minimal", + "family": "callout", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 270.8, + "avgPromptTokens": 4543, + "failures": [] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "callout/realistic", + "family": "callout", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 783.4, + "avgPromptTokens": 4578.6, + "failures": [] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "callout/adversarial", + "family": "callout", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 4, + "everyTime": false, + "stableShape": false, + "distinctShapes": 4, + "avgOutputTokens": 285.2, + "avgPromptTokens": 4556.6, + "failures": [ + "unknown-component", + "schema-error" + ] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "button/minimal", + "family": "button", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 4, + "everyTime": false, + "stableShape": false, + "distinctShapes": 4, + "avgOutputTokens": 338.4, + "avgPromptTokens": 4541.6, + "failures": [ + "broken-reference", + "unknown-component", + "unknown-component", + "unknown-component" + ] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "button/realistic", + "family": "button", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 480.4, + "avgPromptTokens": 4570.8, + "failures": [] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "button/adversarial", + "family": "button", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": true, + "distinctShapes": 1, + "avgOutputTokens": 8, + "avgPromptTokens": 4557.8, + "failures": [] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "tasklist/minimal", + "family": "tasklist", + "variant": "minimal", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 3, + "everyTime": false, + "stableShape": false, + "distinctShapes": 3, + "avgOutputTokens": 758.8, + "avgPromptTokens": 4542, + "failures": [ + "schema-error", + "schema-error", + "schema-error", + "broken-reference" + ] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "tasklist/realistic", + "family": "tasklist", + "variant": "realistic", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 5, + "everyTime": true, + "stableShape": false, + "distinctShapes": 5, + "avgOutputTokens": 547.2, + "avgPromptTokens": 4579.6, + "failures": [] + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "scenario": "tasklist/adversarial", + "family": "tasklist", + "variant": "adversarial", + "repeats": 5, + "truncated": 0, + "scored": 5, + "renderable": 0, + "everyTime": false, + "stableShape": false, + "distinctShapes": 0, + "avgOutputTokens": 954.4, + "avgPromptTokens": 4561, + "failures": [ + "truncated", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error", + "schema-error" + ] } ], "aggregates": [ @@ -5434,6 +6468,57 @@ "failureCounts": { "parse-error": 17 } + }, + { + "model": "anthropic/claude-opus-5", + "format": "openui-v2", + "generations": 90, + "truncationRate": 0.05555555555555555, + "scored": 85, + "renderableRate": 1, + "everyTimeRate": 0.9444444444444444, + "shapeStability": 0.05555555555555555, + "avgOutputTokens": 2125, + "avgPromptTokens": 6977.944444444444, + "efficiency": 0.47058823529411764, + "failureCounts": { + "truncated": 5 + } + }, + { + "model": "openai/gpt-5.6-terra", + "format": "openui-v2", + "generations": 90, + "truncationRate": 0, + "scored": 90, + "renderableRate": 0.9888888888888889, + "everyTimeRate": 0.9444444444444444, + "shapeStability": 0.2222222222222222, + "avgOutputTokens": 623.7555555555556, + "avgPromptTokens": 4411.277777777777, + "efficiency": 1.585378887740924, + "failureCounts": { + "broken-reference": 1 + } + }, + { + "model": "google/gemma-4-26b-a4b-it", + "format": "openui-v2", + "generations": 90, + "truncationRate": 0, + "scored": 90, + "renderableRate": 0.7888888888888889, + "everyTimeRate": 0.5, + "shapeStability": 0.05555555555555555, + "avgOutputTokens": 525.1666666666666, + "avgPromptTokens": 4566.011111111111, + "efficiency": 1.5021686237173386, + "failureCounts": { + "broken-reference": 37, + "truncated": 2, + "schema-error": 55, + "unknown-component": 5 + } } ], "promptSizes": { @@ -5441,10 +6526,11 @@ "openui": 5172, "json-render": 8466, "agenui": 19689, - "a2ui": 16085 + "a2ui": 16085, + "openui-v2": 6978 }, "totals": { - "generations": 1377, + "generations": 1647, "errored": 0 } } \ No newline at end of file diff --git a/benchmarks/genui-formats/src/adapters/a2ui-transport.ts b/benchmarks/genui-formats/src/adapters/a2ui-transport.ts index 12e82cf..db0a731 100644 --- a/benchmarks/genui-formats/src/adapters/a2ui-transport.ts +++ b/benchmarks/genui-formats/src/adapters/a2ui-transport.ts @@ -34,7 +34,12 @@ const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..', '..'); const PROMPT_FILE = join(ROOT, 'vendor', 'a2ui-transport-prompt.txt'); const TOOL = join(ROOT, 'a2ui-python', 'a2ui_tool.py'); -/** Interpreter with the A2UI SDK installed. See a2ui-python/README.md. */ +/** + * Interpreter with the A2UI SDK installed. The default lives under /tmp, which + * the OS eventually cleans, so a missing interpreter is expected rather than + * exceptional. Run `./a2ui-python/setup.sh` to rebuild it, or point + * A2UI_PYTHON at a durable venv. + */ const PYTHON = process.env.A2UI_PYTHON ?? '/tmp/a2ui-venv/bin/python'; const SYSTEM_PROMPT = existsSync(PROMPT_FILE) ? readFileSync(PROMPT_FILE, 'utf8') : ''; @@ -57,10 +62,23 @@ export function validateBatch(outputs: string[]): Verdict[] { const file = join(dir, 'batch.jsonl'); writeFileSync(file, outputs.map((o) => JSON.stringify({ output: o })).join('\n'), 'utf8'); - const stdout = execFileSync(PYTHON, [TOOL, 'validate', file], { - encoding: 'utf8', - maxBuffer: 64 * 1024 * 1024, - }); + let stdout: string; + try { + stdout = execFileSync(PYTHON, [TOOL, 'validate', file], { + encoding: 'utf8', + maxBuffer: 64 * 1024 * 1024, + }); + } catch (err) { + const e = err as { code?: string }; + if (e.code === 'ENOENT') { + throw new Error( + `A2UI python environment missing at ${PYTHON}.\n` + + `Rebuild it with: ./a2ui-python/setup.sh\n` + + `Or set A2UI_PYTHON to an interpreter with the a2ui SDK installed.`, + ); + } + throw err; + } return stdout .split('\n') .filter(Boolean) diff --git a/benchmarks/genui-formats/src/adapters/index.ts b/benchmarks/genui-formats/src/adapters/index.ts index deb684f..0ce68fe 100644 --- a/benchmarks/genui-formats/src/adapters/index.ts +++ b/benchmarks/genui-formats/src/adapters/index.ts @@ -2,15 +2,17 @@ import { a2uiAdapter } from './a2ui.js'; import { a2uiTransportAdapter } from './a2ui-transport.js'; import { jsonRenderAdapter } from './json-render.js'; import { mdmaAdapter } from './mdma.js'; +import { openuiV2Adapter } from './openui-v2.js'; import { openuiAdapter } from './openui.js'; import type { FormatAdapter } from './types.js'; export const ADAPTERS: FormatAdapter[] = [ mdmaAdapter, - openuiAdapter, + openuiV2Adapter, jsonRenderAdapter, a2uiTransportAdapter, a2uiAdapter, + openuiAdapter, ]; export const ADAPTER_BY_ID = Object.fromEntries(ADAPTERS.map((a) => [a.id, a])) as Record< diff --git a/benchmarks/genui-formats/src/adapters/openui-v2.ts b/benchmarks/genui-formats/src/adapters/openui-v2.ts new file mode 100644 index 0000000..57c0422 --- /dev/null +++ b/benchmarks/genui-formats/src/adapters/openui-v2.ts @@ -0,0 +1,65 @@ +/** + * OpenUI Lang, generated from the SHIPPED library rather than their pinned + * benchmark artifact. + * + * Why this exists. The `openui` arm uses `vendor/openui-system-prompt.txt`, + * the file thesysdev publish in their own `benchmarks/` directory. It is + * byte-identical to their repo today, but it is stale relative to the library + * they ship on npm: `openuiLibrary.prompt()` from `@openuidev/react-ui@0.13.5` + * produces a materially different prompt. + * + * pinned benchmarks/system-prompt.txt 13,080 chars, has `## Examples` + * openuiLibrary.prompt() @ 0.13.5 17,431 chars, no `## Examples`, + * adds `## Action - Button Behavior`, + * `## Final Verification`, and the + * Modal and Action components + * + * An integrator who installs the package today gets the second one, so this + * arm measures what they would actually experience. + * + * The schema is regenerated from the SAME library version via + * `toJSONSchema()`. That pairing matters: the pinned schema has no Modal or + * Action, so validating 0.13.5 output against it would report unknown-component + * failures that are an artifact of our mismatch, not of the model. + * + * Validation logic is otherwise identical to the `openui` arm and is reused + * from it, so any difference in score comes from the prompt and schema alone. + * + * Prompt and schema are generated, not committed. Regenerate with: + * node -e "import('@openuidev/react-ui').then(m=>{require('fs') + * .writeFileSync('vendor/openui-prompt-v0.13.5.txt', m.openuiLibrary.prompt())})" + */ + +import { existsSync, readFileSync } from 'node:fs'; +import { dirname, join } from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { createParser } from '@openuidev/lang-core'; +import { validateWithParser } from './openui.js'; +import type { FormatAdapter, ValidationResult } from './types.js'; + +const VENDOR = join(dirname(fileURLToPath(import.meta.url)), '..', '..', 'vendor'); +const PROMPT_FILE = join(VENDOR, 'openui-prompt-v0.13.5.txt'); +const SCHEMA_FILE = join(VENDOR, 'openui-schema-0.13.5.json'); + +const SYSTEM_PROMPT = existsSync(PROMPT_FILE) ? readFileSync(PROMPT_FILE, 'utf8') : ''; +const parser = existsSync(SCHEMA_FILE) + ? createParser(JSON.parse(readFileSync(SCHEMA_FILE, 'utf8'))) + : null; + +export const openuiV2Adapter: FormatAdapter = { + id: 'openui-v2', + label: 'OpenUI Lang', + promptSource: 'openuiLibrary.prompt() from @openuidev/react-ui@0.13.5', + + async systemPrompt(): Promise { + if (!SYSTEM_PROMPT) { + throw new Error(`missing ${PROMPT_FILE} - regenerate from @openuidev/react-ui`); + } + return SYSTEM_PROMPT; + }, + + validate(output: string): ValidationResult { + if (!parser) throw new Error(`missing ${SCHEMA_FILE} - regenerate from @openuidev/react-ui`); + return validateWithParser(output, parser); + }, +}; diff --git a/benchmarks/genui-formats/src/adapters/openui.ts b/benchmarks/genui-formats/src/adapters/openui.ts index b87b84f..a1920e4 100644 --- a/benchmarks/genui-formats/src/adapters/openui.ts +++ b/benchmarks/genui-formats/src/adapters/openui.ts @@ -54,9 +54,22 @@ function shapeOf(node: unknown, acc: string[] = [], depth = 0): string[] { return acc; } +/** + * The validation logic, parameterised by parser so the `openui-v2` arm can + * reuse it verbatim against a different schema. Keeping one implementation + * guarantees any score difference between the two arms comes from the prompt + * and schema, never from a divergence in how we judge them. + */ +export function validateWithParser( + output: string, + activeParser: ReturnType, +): ValidationResult { + return validateImpl(output, activeParser); +} + export const openuiAdapter: FormatAdapter = { id: 'openui', - label: 'OpenUI Lang', + label: 'OpenUI Lang (pinned artifact)', promptSource: 'vendor/openui-system-prompt.txt (thesysdev/openui @ 65b5f93)', async systemPrompt(): Promise { @@ -64,98 +77,102 @@ export const openuiAdapter: FormatAdapter = { }, validate(output: string): ValidationResult { - const issues: ValidationIssue[] = []; - const { text, fenced } = unfence(output); - - if (!text.trim()) { - issues.push({ kind: 'no-structured-output', message: 'empty response' }); - return { ok: false, issues, componentCount: 0 }; - } - - if (fenced) { - issues.push({ - kind: 'prose-leakage', - message: 'output was wrapped in a code fence (prompt says "just openui-lang")', - }); - } - - let result: ReturnType; - try { - result = parser.parse(text); - } catch (err) { - issues.push({ kind: 'parse-error', message: (err as Error).message }); - return { ok: false, issues, componentCount: 0 }; - } - - if (!result.root) { - issues.push({ - kind: result.meta.statementCount === 0 ? 'no-structured-output' : 'parse-error', - message: - result.meta.statementCount === 0 - ? 'no openui-lang statements found in the response' - : `parsed ${result.meta.statementCount} statement(s) but produced no root`, - }); - return { ok: false, issues, componentCount: result.meta.statementCount }; - } - - if (result.meta.incomplete) { - issues.push({ kind: 'truncated', message: 'parser reported incomplete/truncated input' }); - } - - for (const name of result.meta.unresolved) { - // Two very different things land in `unresolved`, and conflating them - // would misreport this format badly. - // - // 1. The model wrote a NAMED argument — `Stack([...], direction="column")`. - // OpenUI Lang is positional-only, so `direction` is read as a bare - // reference. Verified against the parser: the surrounding component - // and all its children still render, only that one prop is lost and - // the component falls back to its default. Degraded, not broken. - // - // 2. A genuine dangling reference — a child listed in a children array - // that was never defined. That leaves a hole in the rendered output. - // - // Distinguish by how the name is used in the source: `name=` inline - // (after a comma or open paren) is the named-argument mistake; anything - // else is a real missing definition. - const namedArg = new RegExp(`[(,]\\s*${escapeRegex(name)}\\s*=`).test(text); - issues.push({ - kind: 'broken-reference', - message: namedArg - ? `named argument "${name}=" — openui-lang is positional-only, so the value is dropped and the component renders with its default` - : `reference "${name}" is used but never defined (renders as null)`, - degraded: namedArg, - }); - } - - for (const name of result.meta.orphaned) { - issues.push({ - kind: 'broken-reference', - message: `statement "${name}" is unreachable from root — silently dropped, will not render`, - }); - } - - for (const err of result.meta.errors.slice(0, 8)) { - const e = err as { code?: string; message?: string }; - issues.push({ - kind: e.code === 'unknown-component' ? 'unknown-component' : 'schema-error', - message: `${e.code}: ${e.message ?? ''}`, - // `excess-args` drops a surplus argument but the component still - // renders — degraded, not broken. Everything else (`missing-required`, - // `null-required`, `unknown-component`) causes the parser to redact the - // component entirely, which is a render failure. - degraded: e.code === 'excess-args', - }); - } - - const shape = shapeOf(result.root); - const fatal = issues.filter((i) => !i.degraded && i.kind !== 'prose-leakage'); - - return { - ok: fatal.length === 0, - issues, - shape: shape.join(','), - componentCount: shape.length, - }; + return validateImpl(output, parser); }, }; + +function validateImpl(output: string, parser: ReturnType): ValidationResult { + const issues: ValidationIssue[] = []; + const { text, fenced } = unfence(output); + + if (!text.trim()) { + issues.push({ kind: 'no-structured-output', message: 'empty response' }); + return { ok: false, issues, componentCount: 0 }; + } + + if (fenced) { + issues.push({ + kind: 'prose-leakage', + message: 'output was wrapped in a code fence (prompt says "just openui-lang")', + }); + } + + let result: ReturnType; + try { + result = parser.parse(text); + } catch (err) { + issues.push({ kind: 'parse-error', message: (err as Error).message }); + return { ok: false, issues, componentCount: 0 }; + } + + if (!result.root) { + issues.push({ + kind: result.meta.statementCount === 0 ? 'no-structured-output' : 'parse-error', + message: + result.meta.statementCount === 0 + ? 'no openui-lang statements found in the response' + : `parsed ${result.meta.statementCount} statement(s) but produced no root`, + }); + return { ok: false, issues, componentCount: result.meta.statementCount }; + } + + if (result.meta.incomplete) { + issues.push({ kind: 'truncated', message: 'parser reported incomplete/truncated input' }); + } + + for (const name of result.meta.unresolved) { + // Two very different things land in `unresolved`, and conflating them + // would misreport this format badly. + // + // 1. The model wrote a NAMED argument — `Stack([...], direction="column")`. + // OpenUI Lang is positional-only, so `direction` is read as a bare + // reference. Verified against the parser: the surrounding component + // and all its children still render, only that one prop is lost and + // the component falls back to its default. Degraded, not broken. + // + // 2. A genuine dangling reference — a child listed in a children array + // that was never defined. That leaves a hole in the rendered output. + // + // Distinguish by how the name is used in the source: `name=` inline + // (after a comma or open paren) is the named-argument mistake; anything + // else is a real missing definition. + const namedArg = new RegExp(`[(,]\\s*${escapeRegex(name)}\\s*=`).test(text); + issues.push({ + kind: 'broken-reference', + message: namedArg + ? `named argument "${name}=" — openui-lang is positional-only, so the value is dropped and the component renders with its default` + : `reference "${name}" is used but never defined (renders as null)`, + degraded: namedArg, + }); + } + + for (const name of result.meta.orphaned) { + issues.push({ + kind: 'broken-reference', + message: `statement "${name}" is unreachable from root — silently dropped, will not render`, + }); + } + + for (const err of result.meta.errors.slice(0, 8)) { + const e = err as { code?: string; message?: string }; + issues.push({ + kind: e.code === 'unknown-component' ? 'unknown-component' : 'schema-error', + message: `${e.code}: ${e.message ?? ''}`, + // `excess-args` drops a surplus argument but the component still + // renders — degraded, not broken. Everything else (`missing-required`, + // `null-required`, `unknown-component`) causes the parser to redact the + // component entirely, which is a render failure. + degraded: e.code === 'excess-args', + }); + } + + const shape = shapeOf(result.root); + const fatal = issues.filter((i) => !i.degraded && i.kind !== 'prose-leakage'); + + return { + ok: fatal.length === 0, + issues, + shape: shape.join(','), + componentCount: shape.length, + }; +} diff --git a/benchmarks/genui-formats/src/adapters/types.ts b/benchmarks/genui-formats/src/adapters/types.ts index f46089e..9cd0e0b 100644 --- a/benchmarks/genui-formats/src/adapters/types.ts +++ b/benchmarks/genui-formats/src/adapters/types.ts @@ -60,7 +60,7 @@ export interface ValidationResult { } export interface FormatAdapter { - id: 'mdma' | 'openui' | 'json-render' | 'agenui' | 'a2ui'; + id: 'mdma' | 'openui' | 'json-render' | 'agenui' | 'a2ui' | 'openui-v2'; /** Display name used in report tables. */ label: string; /** Where the system prompt came from — printed in the report for auditability. */ diff --git a/benchmarks/genui-formats/src/fixtures.ts b/benchmarks/genui-formats/src/fixtures.ts index 2343453..9f209bb 100644 --- a/benchmarks/genui-formats/src/fixtures.ts +++ b/benchmarks/genui-formats/src/fixtures.ts @@ -53,7 +53,7 @@ fields: corruption: 'broken YAML indentation inside the mdma block', }, - openui: { + 'openui-v2': { valid: `root = Stack([title, form]) title = TextContent("Contact us", "large-heavy") form = Form("contact", [nameField, emailField]) diff --git a/benchmarks/genui-formats/src/report.ts b/benchmarks/genui-formats/src/report.ts index f8d1186..5f9ce53 100644 --- a/benchmarks/genui-formats/src/report.ts +++ b/benchmarks/genui-formats/src/report.ts @@ -46,8 +46,11 @@ function matrix(results: Results, pick: (agg: Aggregate) => string, formats: str function main(): void { const results = JSON.parse(readFileSync(RESULTS, 'utf8')) as Results; - const formats = ADAPTERS.map((a) => a.id).filter((id) => - results.aggregates.some((a) => a.format === id), + // `openui` is the superseded pinned-artifact run (see section 8), kept as a + // probe rather than a peer column so the comparison shows one OpenUI result. + const PROBE_FORMATS = new Set(['openui']); + const formats = ADAPTERS.map((a) => a.id).filter( + (id) => !PROBE_FORMATS.has(id) && results.aggregates.some((a) => a.format === id), ); // Only models with a COMPLETE sweep of every format are reported. A model @@ -553,6 +556,82 @@ function main(): void { '', ); + const G = 'google/gemma-4-26b-a4b-it'; + const O = 'anthropic/claude-opus-5'; + const T = 'openai/gpt-5.6-terra'; + const pin = (m: string) => results.aggregates.find((a) => a.model === m && a.format === 'openui'); + const shp = (m: string) => + results.aggregates.find((a) => a.model === m && a.format === 'openui-v2'); + + if (pin(O) && shp(O)) { + md.push("## 8. OpenUI's published prompt is stale against the library it ships"); + md.push(''); + md.push( + 'thesysdev publish `benchmarks/system-prompt.txt` and use it in their own benchmark. It is', + 'byte-identical to their repository today, but it no longer matches the library they ship on', + 'npm: `openuiLibrary.prompt()` from `@openuidev/react-ui@0.13.5` produces a different prompt.', + '', + ); + md.push( + table( + ['', 'Published artifact', 'Shipped library v0.13.5'], + [ + ['Size', '13,080 chars', '17,431 chars'], + ['`## Examples`', 'present', '**removed**'], + ['`## Action - Button Behavior`', 'absent', '**added**'], + ['`## Final Verification`', 'absent', '**added**'], + ['Components', 'base set', '**+Modal, +Action**'], + ], + ), + ); + md.push(''); + md.push( + 'Anyone installing the package today gets the second one, so **the scored OpenUI arm uses the', + 'shipped library prompt**, with the schema regenerated from the same version so parser and', + 'prompt agree. We re-ran all 270 generations rather than adjusting anything after the fact.', + '', + 'The superseded run is kept here because the comparison is informative:', + '', + ); + md.push( + table( + ['Model', 'Published artifact', 'Shipped library', 'Delta'], + [ + [ + 'Opus 5', + pct(pin(O)?.everyTimeRate ?? 0), + pct(shp(O)?.everyTimeRate ?? 0), + `+${(((shp(O)?.everyTimeRate ?? 0) - (pin(O)?.everyTimeRate ?? 0)) * 100).toFixed(1)}pp`, + ], + [ + 'GPT-5.6-terra', + pct(pin(T)?.everyTimeRate ?? 0), + pct(shp(T)?.everyTimeRate ?? 0), + `${(((shp(T)?.everyTimeRate ?? 0) - (pin(T)?.everyTimeRate ?? 0)) * 100).toFixed(1)}pp`, + ], + [ + 'Gemma-4-26B', + pct(pin(G)?.everyTimeRate ?? 0), + pct(shp(G)?.everyTimeRate ?? 0), + `${(((shp(G)?.everyTimeRate ?? 0) - (pin(G)?.everyTimeRate ?? 0)) * 100).toFixed(1)}pp`, + ], + ], + ), + ); + md.push(''); + md.push( + '"Every time" rate, five repeats per scenario. The correction gained 2 scenarios on the', + 'flagship and lost 1 on each cheaper model, averaging out to no change across the ladder.', + 'Correcting it was therefore about accuracy, not about advantage: the published artifact was', + 'not systematically unfair to OpenUI, and scored better on two rungs of three.', + '', + 'The direction is consistent with section 7. The shipped prompt trades worked examples for', + 'prose rules; a strong model applies rules, a weak one needs the examples. On Gemma the', + 'largest failure category became `missing-required`, which is what losing examples predicts.', + '', + ); + } + md.push('## Method and fairness'); md.push(''); md.push( diff --git a/benchmarks/genui-formats/src/verify-adapters.ts b/benchmarks/genui-formats/src/verify-adapters.ts index c3f4ea0..9b8d242 100644 --- a/benchmarks/genui-formats/src/verify-adapters.ts +++ b/benchmarks/genui-formats/src/verify-adapters.ts @@ -26,6 +26,16 @@ const RED = '\x1b[31m'; const DIM = '\x1b[2m'; const RESET = '\x1b[0m'; +/** + * Probe adapters differ from a peer arm only in their prompt (and matching + * schema) and deliberately reuse that arm's validator. They have no fixture of + * their own, and "rejects " is expected to + * fail because it IS that validator. Both checks are skipped; the + * prompt-builds check still applies. + */ +const PROBES = new Set(['openui']); +const SHARES_VALIDATOR_WITH: Record = { openui: 'openui-v2' }; + let failures = 0; function check(label: string, pass: boolean, detail = ''): void { @@ -52,6 +62,7 @@ async function main(): Promise { console.log('\n=== 2. Valid passes, corrupted fails ===\n'); for (const adapter of ADAPTERS) { + if (PROBES.has(adapter.id)) continue; const fixture = FIXTURES[adapter.id]; if (!fixture) { check(`${adapter.label} has a fixture`, false); @@ -82,6 +93,7 @@ async function main(): Promise { for (const adapter of ADAPTERS) { for (const [otherId, fixture] of Object.entries(FIXTURES)) { if (otherId === adapter.id) continue; + if (SHARES_VALIDATOR_WITH[adapter.id] === otherId) continue; const result = adapter.validate(fixture.valid); check( `${adapter.label.padEnd(14)} rejects ${otherId}`, diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 3835267..2abbdfa 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -44,6 +44,9 @@ importers: '@openuidev/lang-core': specifier: 0.2.10 version: 0.2.10(@modelcontextprotocol/sdk@1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76))(zod@3.25.76) + '@openuidev/react-ui': + specifier: 0.13.5 + version: 0.13.5(@openuidev/react-headless@0.9.6(ai@6.0.175(zod@3.25.76))(react@19.2.4)(zustand@4.5.7(@types/react@19.2.14)(immer@11.1.4)(react@19.2.4)))(@openuidev/react-lang@0.2.11(@modelcontextprotocol/sdk@1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76))(react-dom@19.2.4(react@19.2.4))(react@19.2.4)(zod@3.25.76))(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4)(zod@3.25.76)(zustand@4.5.7(@types/react@19.2.14)(immer@11.1.4)(react@19.2.4)) ajv: specifier: ^8.17.1 version: 8.18.0 @@ -722,6 +725,9 @@ packages: '@ag-ui/client@0.0.57': resolution: {integrity: sha512-Xap2alG9Z0/j5kb3x4D7oTpe2sw1dfrC9rgJJr2NZu5vKcm8dzIPNd31mF2B4zS3BKqYIu245yxKPhEtT30MHw==} + '@ag-ui/core@0.0.53': + resolution: {integrity: sha512-11UocR7fFdMWw503bWCX2IOK15vbWfxT11Mn9xOiPBVO/UVcn57ywGrlLL4UaBlPgmUTvuzr2yYR2ElSqiN2wQ==} + '@ag-ui/core@0.0.57': resolution: {integrity: sha512-gho1OWjNE6E3Rl7ZEZ1wr2CEpUHjLFU0FqzCZZk439TicLu+BfLCMkMokB07bMGlRmbJ60hM6LW60iOVauCx+Q==} @@ -2104,6 +2110,9 @@ packages: '@dabh/diagnostics@2.0.8': resolution: {integrity: sha512-R4MSXTVnuMzGD7bzHdW2ZhhdPC/igELENcq5IjEverBvq5hn1SXCWcsi6eSsdWP0/Ur+SItRRjAktmdoX/8R/Q==} + '@date-fns/tz@1.5.0': + resolution: {integrity: sha512-lwYN/vDPeNRULcepoE/LO2Pgx+7/RV+S9ARfbc9lr2DtGkOD7pAiruHvbR1RX3Qyf6ja47EWJDMsNK5vK08DJg==} + '@dimforge/rapier3d-compat@0.12.0': resolution: {integrity: sha512-uekIGetywIgopfD97oDL5PfeezkFpNhwlzlaEYNOA0N6ghdsOvh/HYjSMek5Q2O1PYvRSDFcqFVJl4r4ZBwOow==} @@ -3197,6 +3206,13 @@ packages: resolution: {integrity: sha512-cifvXDhcqMwwTlTK04GBNeIe7yyo28Mfby85QXFe1Yk8nmi36Ab/5UQwptOx84SsoGNRg+EVSjwzfSZMy6pmlw==} engines: {node: '>=14'} + '@openuidev/devtools@0.0.4': + resolution: {integrity: sha512-DVf5Bu6gpKEfAnETyl/ZOQMyZCi1vapGohNSxaIyZ03jVZ+e5F9fkQJp+5T1ryg0v3vYi3AC0B4NRwEuCbZmmQ==} + peerDependencies: + '@openuidev/observability': ^0.0.2 + react: ^18.3.1 || ^19.0.0 + react-dom: ^18.0.0 || ^19.0.0 + '@openuidev/lang-core@0.2.10': resolution: {integrity: sha512-GeLoP4S+1e7IHAz2a4Fl9+Q+boiyTXpXtdUPI+PN3ml+834q35TCQasm/+s6NW+79FRmyiR8F5xJtoLsH0kGjQ==} peerDependencies: @@ -3206,6 +3222,51 @@ packages: '@modelcontextprotocol/sdk': optional: true + '@openuidev/lang-core@0.2.11': + resolution: {integrity: sha512-30tTX7bnTLMQMs1kPhIv/tVY8hlW5xtimf6fBBS16ctidceuMZNWEe0VTpuz3lEmdyThp2lc7vy+lSfTqIIwlg==} + peerDependencies: + '@modelcontextprotocol/sdk': '>=1.0.0' + zod: ^3.25.0 || ^4.0.0 + peerDependenciesMeta: + '@modelcontextprotocol/sdk': + optional: true + + '@openuidev/observability@0.0.2': + resolution: {integrity: sha512-uNtO4VGGwcL8Lg1L+9a8JCpee0dOm0djIjItxtGpUPkZNUwzpMQX6mQ4YCtA9Hho4VuqZgHBo1ON8FjsLbNMHg==} + + '@openuidev/react-headless@0.9.6': + resolution: {integrity: sha512-EJO+WkkG9Rk8lTl2QSNK4cTl7D82o/z2saVIOo0CJMDn9Y5g/uqCoUe+v8J56c8EYkoLrrOce2SFsq7w+aWu9A==} + peerDependencies: + ai: ^6.0.0 + react: ^18.3.1 || ^19.0.0 + zustand: ^4.5.5 + peerDependenciesMeta: + ai: + optional: true + + '@openuidev/react-lang@0.2.11': + resolution: {integrity: sha512-RXyRmFLxRBQ/pNnCq4CYFLl/1SnUvv3O5cnyTaaG6RqtXQ/CLmRiooqCtNY8xqi/cDVt3HJTgrfQOCN2ZhAQbg==} + peerDependencies: + '@modelcontextprotocol/sdk': '>=1.0.0' + react: '*' + react-dom: '*' + zod: ^3.25.0 || ^4.0.0 + peerDependenciesMeta: + '@modelcontextprotocol/sdk': + optional: true + react-dom: + optional: true + + '@openuidev/react-ui@0.13.5': + resolution: {integrity: sha512-6mpCpLKouva4RK+nhBUcl4gMaE/1NOU5yW0bXnSvMFoC5UdFD2vtQOkmmNYG4t3giMHI5lHRMZ6u5RRKGe6ggg==} + peerDependencies: + '@openuidev/react-headless': ^0.9.6 + '@openuidev/react-lang': ^0.2.11 + react: ^18.3.1 || ^19.0.0 + react-dom: ^18.0.0 || ^19.0.0 + zod: ^3.25.0 || ^4.0.0 + zustand: ^4.5.5 + '@pkgjs/parseargs@0.11.0': resolution: {integrity: sha512-+1VkjdD0QBLPodGrJUeqarH8VAIvQODIbwh9XpP5Syisf7YoQgsJKPNFoqqLQlu+VQ/tVSshMR6loPMn8U+dPg==} engines: {node: '>=14'} @@ -4293,6 +4354,10 @@ packages: '@sinclair/typebox@0.27.10': resolution: {integrity: sha512-MTBk/3jGLNB2tVxv6uLlFh1iu64iYOQ2PbdOSK3NW8JZsmlaOh2q6sdtKowBhfw8QFLmYNzTW4/oK4uATIi6ZA==} + '@sindresorhus/is@4.6.0': + resolution: {integrity: sha512-t09vSN3MdfsyCHoFcTRCH/iUtG7OJ0CsjzB8cjAmKc/va/kIgeDI/TxsigdncE/4be734m0cvIYwNaV4i2XqAw==} + engines: {node: '>=10'} + '@sindresorhus/merge-streams@2.3.0': resolution: {integrity: sha512-LtoMMhxAlorcGhmFYI+LhPgbPZCkgP6ra1YL604EeF6U98pLlQ3iWIGMdWSC+vWmPBWBNgmDBAhnAobLROJmwg==} engines: {node: '>=18'} @@ -4630,6 +4695,10 @@ packages: '@swc/types@0.1.26': resolution: {integrity: sha512-lyMwd7WGgG79RS7EERZV3T8wMdmPq3xwyg+1nmAM64kIhx5yl+juO2PYIHb7vTiPgPCj8LYjsNV2T5wiQHUEaw==} + '@tabby_ai/hijri-converter@1.0.5': + resolution: {integrity: sha512-r5bClKrcIusDoo049dSL8CawnHR6mRdDwhlQuIgZRNty68q0x8k3Lf1BtPAMxRf/GgnHBnIO4ujd3+GQdLWzxQ==} + engines: {node: '>=16.0.0'} + '@tailwindcss/node@4.2.1': resolution: {integrity: sha512-jlx6sLk4EOwO6hHe1oCGm1Q4AN/s0rSrTTPBGPM0/RQ6Uylwq17FuU8IeJJKEjtc6K6O07zsvP+gDO6MMWo7pg==} @@ -4793,12 +4862,18 @@ packages: '@types/deep-eql@4.0.2': resolution: {integrity: sha512-c9h9dVVMigMPc4bwTvC5dxqtqJZwQPePsWjPlpSOnojbor6pGqdk541lfA7AqFQr5pB1BRdq0juY9db81BwyFw==} + '@types/estree-jsx@1.0.5': + resolution: {integrity: sha512-52CcUVNFyfb1A2ALocQw/Dd1BQFNmSdkuC3BkZ6iqhdMfQz7JWOFRuJFloOzjk+6WijU56m9oKXFAXc7o3Towg==} + '@types/estree@1.0.8': resolution: {integrity: sha512-dWHzHa2WqEXI/O1E9OjrocMTKJl2mSrEolh1Iomrv6U+JuNwaHXsXx9bLu5gG7BUWFIN0skIQJQ/L1rIex4X6w==} '@types/graceful-fs@4.1.9': resolution: {integrity: sha512-olP3sd1qOEe5dXTSaFvQG+02VdRXcdytWLAZsAq1PecU8uqQAhkrnbli7DagjtXKW/Bl7YJbUsa8MPcuc8LHEQ==} + '@types/hast@3.0.5': + resolution: {integrity: sha512-rp/ezSWaD1m44dPKICGhiskI13nVr7qTloFwDa/IYkhhf5nzwP+zIQcIJh3WIFSBOy/H1PzB40jPjMDksN4F+g==} + '@types/istanbul-lib-coverage@2.0.6': resolution: {integrity: sha512-2QF/t/auWm0lsy8XtKVPG19v3sSOQlJe/YHZgfjb/KBBHOGSV+J2q/S671rcq9uTBrLAXmZpqJiaQbMT+zNU1w==} @@ -4811,6 +4886,9 @@ packages: '@types/json-schema@7.0.15': resolution: {integrity: sha512-5+fP8P8MFNC+AyZCDxrB2pkZFPGzqQWUzpSeuuVLvm8VMcorNYavBqoFcxK8bQz4Qsbn4oUEEem4wDLfcysGHA==} + '@types/katex@0.16.8': + resolution: {integrity: sha512-trgaNyfU+Xh2Tc+ABIb44a5AYUpicB3uwirOioeOkNPPbmgRNtcWyDeeFRzjPZENO9Vq8gvVqfhaaXWLlevVwg==} + '@types/mdast@4.0.4': resolution: {integrity: sha512-kGaNbPh1k7AFzgpud/gMdvIm5xuECykRR+JnWKQno9TAXVa6WIVCGTPvYGekIDL4uwCZQSYbUxNBSb1aUo79oA==} @@ -4838,6 +4916,9 @@ packages: '@types/pegjs@0.10.6': resolution: {integrity: sha512-eLYXDbZWXh2uxf+w8sXS8d6KSoXTswfps6fvCUuVAGN8eRpfe7h9eSRydxiSJvo9Bf+GzifsDOr9TMQlmJdmkw==} + '@types/prismjs@1.26.6': + resolution: {integrity: sha512-vqlvI7qlMvcCBbVe0AKAb4f97//Hy0EBTaiW8AalRnG/xAN5zOiWWyrNqNXeq8+KAuvRewjCVY1+IPxk4RdNYw==} + '@types/react-dom@19.2.3': resolution: {integrity: sha512-jp2L/eY6fn+KgVVQAOqYItbF0VY/YApe5Mz2F0aykSO8gx31bYCZyvSeYxCHKvzHG5eZjc+zyaS5BrBWya2+kQ==} peerDependencies: @@ -4867,6 +4948,9 @@ packages: '@types/triple-beam@1.3.5': resolution: {integrity: sha512-6WaYesThRMCl19iryMYP7/x2OVgCtbIVflDGFpWnb9irXI3UjYE4AzmYuiUKY1AJstGijoY+MgUszMgRxIYTYw==} + '@types/unist@2.0.11': + resolution: {integrity: sha512-CmBKiL6NNo/OqgmMn95Fk9Whlp2mtvIv+KNpQKN2F4SjvrEesubTRWGYSg+BnWZOnlCaSTU1sMpsBOzgbYhnsA==} + '@types/unist@3.0.3': resolution: {integrity: sha512-ko/gIFJRv177XgZsZcBwnqJN5x/Gien8qNOn0D5bQU/zAzVf9Zt3BlcUiLqhV9y4ARk0GbT3tnUiPNgnTXzc/Q==} @@ -5484,9 +5568,22 @@ packages: resolution: {integrity: sha512-7NzBL0rN6fMUW+f7A6Io4h40qQlG+xGmtMxfbnH/K7TAtt8JQWVQK+6g0UXKMeVJoyV5EkkNsErQ8pVD3bLHbA==} engines: {node: ^12.17.0 || ^14.13 || >=16.0.0} + char-regex@1.0.2: + resolution: {integrity: sha512-kWWXztvZ5SBQV+eRgKFeh8q5sLuZY2+8WUIzlxWVTg+oGwY14qylx1KbKzHd8P6ZYkAg0xyIDU9JMHhyJMZ1jw==} + engines: {node: '>=10'} + + character-entities-html4@2.1.0: + resolution: {integrity: sha512-1v7fgQRj6hnSwFpq1Eu0ynr/CDEw0rXo2B61qXrLNdHZmPKgb7fqS1a2JwF0rISo9q77jDI8VMEHoApn8qDoZA==} + + character-entities-legacy@3.0.0: + resolution: {integrity: sha512-RpPp0asT/6ufRm//AJVwpViZbGM/MkjQFxJccQRHmISF/22NBtsHqAWmL+/pmkPWoIUJdWyeVleTl1wydHATVQ==} + character-entities@2.0.2: resolution: {integrity: sha512-shx7oQ0Awen/BRIdkjkvz54PnEEI/EjwXDSIZp86/KKdbafHh1Df/RYGBhn4hbe2+uKC9FnT5UCEdyPz3ai9hQ==} + character-reference-invalid@2.0.1: + resolution: {integrity: sha512-iBZ4F4wRbyORVsu0jPV7gXkOsGYjGHPmAyv+HiHG8gi5PtC9KI2j1+v8/tlibRvjoWX027ypmG/n0HtO5t7unw==} + chardet@2.1.1: resolution: {integrity: sha512-PsezH1rqdV9VvyNhxxOW32/d75r01NY7TQCmOqomRo15ZSOKbpTFVsfjghxo6JloQUCGnH4k1LGu0R4yCLlWQQ==} @@ -5623,6 +5720,9 @@ packages: resolution: {integrity: sha512-FQN4MRfuJeHf7cBbBMJFXhKSDq+2kAArBlmRBvcvFE5BB1HZKXtSFASDhdlz9zOYwxh8lDdnvmMOe/+5cdoEdg==} engines: {node: '>= 0.8'} + comma-separated-tokens@2.0.3: + resolution: {integrity: sha512-Fu4hJdvzeylCfQPp9SGWidpzrMs7tTrlu6Vb8XGaRGck8QSNZJJp538Wrb60Lax4fPwR64ViY468OIUTbRlGZg==} + commander@10.0.1: resolution: {integrity: sha512-y4Mg2tXshplEbSGzx7amzPwKKOCGuoSRP/CjEdwwk0FOGlUbq6lKuoyDZTNZkmxHdJtp54hdfY/JUrdL7Xfdug==} engines: {node: '>=14'} @@ -5650,6 +5750,10 @@ packages: resolution: {integrity: sha512-QrWXB+ZQSVPmIWIhtEO9H+gwHaMGYiF5ChvoJ+K9ZGHG/sVsa6yiesAD1GC/x46sET00Xlwo1u49RVVVzvcSkw==} engines: {node: '>= 10'} + commander@8.3.0: + resolution: {integrity: sha512-OkTL9umf+He2DZkUq8f8J9of7yL6RJKI24dVITBmNfZBmri9zYZQrKkuXiKhyfPSu8tUhnVBB1iKXevvnlR4Ww==} + engines: {node: '>= 12'} + commondir@1.0.1: resolution: {integrity: sha512-W9pAhw0ja1Edb5GVdIF1mjZw/ASI0AlShXM83UUGe2DVr5TdAPEA1OA8m/g8zWp9x6On7gqufY+FatDbC3MDQg==} @@ -5809,6 +5913,12 @@ packages: resolution: {integrity: sha512-23XHcCF+coGYevirZceTVD7NdJOqVn+49IHyxgszm+JIiHLoB2TkmPtsYkNWT1pvRSGkc35L6NHs0yHkN2SumA==} engines: {node: ^20.19.0 || ^22.12.0 || >=24.0.0} + date-fns-jalali@4.1.0-0: + resolution: {integrity: sha512-hTIP/z+t+qKwBDcmmsnmjWTduxCg+5KfdqWQvb2X/8C9+knYY6epN/pfxdDuyVlSVeFz0sM5eEfwIUQ70U4ckg==} + + date-fns@4.4.0: + resolution: {integrity: sha512-+1UMbeh68lH1SegH83CGWwpb6OHHbpSgr3+s5Eww5M4CAgswBpoWS0AjTOfEJ33HiYKz1hdj/KTFprzXHmq/6w==} + de-indent@1.0.2: resolution: {integrity: sha512-e/1zu3xH5MQryN2zdVaF0OrdNLUbvWxzMbi+iNA6Bky7l1RoP8a2fIbRocyHclXt/arDrrR6lL3TqFD9pMQTsg==} @@ -5961,6 +6071,9 @@ packages: resolution: {integrity: sha512-WkrWp9GR4KXfKGYzOLmTuGVi1UWFfws377n9cc55/tb6DuqyF6pcQ5AbiHEshaDpY9v6oaSr2XCDidGmMwdzIA==} engines: {node: '>=8'} + dom-helpers@5.2.1: + resolution: {integrity: sha512-nRCa7CK3VTrM2NmGkIy4cbK7IZlgBE/PYMn55rrXefr5xXDP0LdtfPnblFDoVdcAfslJ7or6iqAUnx0CCGIWQA==} + dotenv-expand@11.0.7: resolution: {integrity: sha512-zIHwmZPRshsCdpMDyVsqGmgyP0yT8GAgXUnkdAoJisxvf33k7yO6OuoKmcTGuXPWSsm8Oh88nZicRLA9Y0rUeA==} engines: {node: '>=12'} @@ -6115,6 +6228,12 @@ packages: emoji-regex@9.2.2: resolution: {integrity: sha512-L18DaJsXSUk2+42pv8mLs5jJT2hqFkFE4j21wOmgbUqsZ2hL72NsUU785g9RXgo3s0ZNgVl42TiHp3ZtOv/Vyg==} + emojilib@2.4.0: + resolution: {integrity: sha512-5U0rVMU5Y2n2+ykNLQqMoqklN9ICBT/KsvC1Gz6vqHbz2AXXGkG+Pm5rMWk/8Vjrr/mY9985Hi8DYzn1F09Nyw==} + + emoticon@4.1.0: + resolution: {integrity: sha512-VWZfnxqwNcc51hIy/sbOdEem6D+cVtpPzEEtVAFdaas30+1dgkyaOQ4sQ6Bp0tOMqWO1v+HQfYaoodOkdhK6SQ==} + enabled@2.0.0: resolution: {integrity: sha512-AKrN98kuwOzMIdAizXGI86UFBoo26CL21UM763y1h/GMSJ4/OHU9k2YlsmBpyScFo/wbLzWQJBMCW4+IO3/+OQ==} @@ -6249,6 +6368,9 @@ packages: resolution: {integrity: sha512-MMdARuVEQziNTeJD8DgMqmhwR11BRQ/cBP+pLtYdSTnf3MIO8fFeiINEbX36ZdNlfU/7A9f3gUw49B3oQsvwBA==} engines: {node: '>=4.0'} + estree-util-is-identifier-name@3.0.0: + resolution: {integrity: sha512-hFtqIDZTIUZ9BXLb8y4pYGyk6+wekIivNVTcmvk8NoOh+VeRn5y6cEHzbURrWbfp1fIqdVipilzj+lfaadNZmg==} + estree-walker@2.0.2: resolution: {integrity: sha512-Rfkk/Mp/DL7JVje3u18FxFujQlTNR2q6QfMSMB7AvCBx91NGj/ba3kCfza0f6dVDbw7YlRf/nDrn7pQrCCyQ/w==} @@ -6403,6 +6525,10 @@ packages: fast-deep-equal@3.1.3: resolution: {integrity: sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q==} + fast-equals@5.4.1: + resolution: {integrity: sha512-DjlFSM5Pk9cGcL0q5QXl66eGzx0N6szNgaswwc5ZphlBohjTVJSnGgI+rJVOgOi65qUoQnDZN4nDqi33udtydQ==} + engines: {node: '>=6.0.0'} + fast-glob@3.3.3: resolution: {integrity: sha512-7MptL8U0cqcFdzIzwOTHoilX9x5BrNqye7Z/LuC7kCMRio1EMSyqRK3BEAUD7sXRq4iT4AzTVuZdhgQ2TCvYLg==} engines: {node: '>=8.6.0'} @@ -6445,6 +6571,9 @@ packages: fastq@1.20.1: resolution: {integrity: sha512-GGToxJ/w1x32s/D2EKND7kTil4n8OVk/9mycTc4VDza13lOvpUZTGX3mFSCtV9ksdGBVzvsyAVLM6mHFThxXxw==} + fault@1.0.4: + resolution: {integrity: sha512-CJ0HCB5tL5fYTEA7ToAq5+kTwd++Borf1/bifxd9iT70QcXr4MRrO3Llf8Ifs70q+SJcGHFtnIE/Nw6giCtECA==} + fb-watchman@2.0.2: resolution: {integrity: sha512-p5161BqbuCaSnB8jIbzQHOlpgsPmK5rJVDfDKO91Axs5NC1uu3HRQm6wt9cd9/+GtQQIO53JdGXXoyDpTAsgYA==} @@ -6564,6 +6693,10 @@ packages: resolution: {integrity: sha512-8RipRLol37bNs2bhoV67fiTEvdTrbMUYcFTiy3+wuuOnUog2QBHCZWXDRijWQfAkhBj2Uf5UnVaiWwA5vdd82w==} engines: {node: '>= 6'} + format@0.2.2: + resolution: {integrity: sha512-wzsgA6WOq+09wrU1tsJ09udeR/YZRaeArL9e1wPbFg3GG2yDnC2ldKpxs4xunpFF9DgqCqOIra3bc1HWrJ37Ww==} + engines: {node: '>=0.4.x'} + formdata-node@4.4.1: resolution: {integrity: sha512-0iirZp3uVDjVGt9p49aTaqjk84TrglENEDuqfdlZQ1roC9CWlPk6Avf8EEnZNcAqPonwkG35x4n3ww/1THYAeQ==} engines: {node: '>= 12.20'} @@ -6778,6 +6911,36 @@ packages: resolution: {integrity: sha512-T2UbfbBEF32wiepXIsMlTW9+dDYC6wMh/t/vYA4tuOMKqWz/n3vr1NFSxQiyP+zk2mXsoMA/i/7qV6LKut1t1A==} engines: {node: '>= 0.4'} + hast-util-from-dom@5.0.1: + resolution: {integrity: sha512-N+LqofjR2zuzTjCPzyDUdSshy4Ma6li7p/c3pA78uTwzFgENbgbUrm2ugwsOdcjI1muO+o6Dgzp9p8WHtn/39Q==} + + hast-util-from-html-isomorphic@2.0.0: + resolution: {integrity: sha512-zJfpXq44yff2hmE0XmwEOzdWin5xwH+QIhMLOScpX91e/NSGPsAzNCvLQDIEPyO2TXi+lBmU6hjLIhV8MwP2kw==} + + hast-util-from-html@2.0.3: + resolution: {integrity: sha512-CUSRHXyKjzHov8yKsQjGOElXy/3EKpyX56ELnkHH34vDVw1N1XSQ1ZcAvTyAPtGqLTuKP/uxM+aLkSPqF/EtMw==} + + hast-util-from-parse5@8.0.3: + resolution: {integrity: sha512-3kxEVkEKt0zvcZ3hCRYI8rqrgwtlIOFMWkbclACvjlDw8Li9S2hk/d51OI0nr/gIpdMHNepwgOKqZ/sy0Clpyg==} + + hast-util-is-element@3.0.0: + resolution: {integrity: sha512-Val9mnv2IWpLbNPqc/pUem+a7Ipj2aHacCwgNfTiK0vJKl0LF+4Ba4+v1oPHFpf3bLYmreq0/l3Gud9S5OH42g==} + + hast-util-parse-selector@4.0.0: + resolution: {integrity: sha512-wkQCkSYoOGCRKERFWcxMVMOcYE2K1AaNLU8DXS9arxnLOUEWbOXKXiJUNzEpqZ3JOKpnha3jkFrumEjVliDe7A==} + + hast-util-to-jsx-runtime@2.3.6: + resolution: {integrity: sha512-zl6s8LwNyo1P9uw+XJGvZtdFF1GdAkOg8ujOw+4Pyb76874fLps4ueHXDhXWdk6YHQ6OgUtinliG7RsYvCbbBg==} + + hast-util-to-text@4.0.2: + resolution: {integrity: sha512-KK6y/BN8lbaq654j7JgBydev7wuNMcID54lkRav1P0CaE1e47P72AWWPiGKXTJU271ooYzcvTAn/Zt0REnvc7A==} + + hast-util-whitespace@3.0.0: + resolution: {integrity: sha512-88JUN06ipLwsnv+dVn+OIYOvAuvBMy/Qoi6O7mQHxdPXpjy+Cd6xRkWwux7DKO+4sYILtLBRIKgsdpS2gQc7qw==} + + hastscript@9.0.1: + resolution: {integrity: sha512-g7df9rMFX/SPi34tyGCyUBREQoKkapwdY/T04Qn9TDWfHhAYt4/I0gMVirzK5wEzeUqIjEB+LXC/ypb7Aqno5w==} + he@1.2.0: resolution: {integrity: sha512-F/1DnUGPopORZi0ni+CvrCgHQ5FyEAHRLSApuYWMmrbSwoN2Mn/7k+Gl38gJnR7yyDZk6WLXwiGod1JOWNDKGw==} hasBin: true @@ -6818,10 +6981,16 @@ packages: hermes-parser@0.36.0: resolution: {integrity: sha512-GdpwMmH5x6IpC1cijvcvYnlPB60Mh6kTSF/NFdYV/j56gYdi+0RIakYs+eqOV+bbO0SW7mgVVGSsTJxyPQfo3w==} + highlight.js@10.7.3: + resolution: {integrity: sha512-tzcUFauisWKNHaRkN4Wjl/ZA07gENAjFl3J/c480dprkGTg5EQstgaNFqBfUqCq54kZRIEcreTsAgF/m2quD7A==} + highlight.js@11.11.1: resolution: {integrity: sha512-Xwwo44whKBVCYoliBQwaPvtd/2tYFkRQtXDWj1nackaV2JPXx3L0+Jvd8/qCJ2p+ML0/XVkJ2q+Mr+UVdpJK5w==} engines: {node: '>=12.0.0'} + highlightjs-vue@1.0.0: + resolution: {integrity: sha512-PDEfEF102G23vHmPhLyPboFCD+BkMGu+GuJe2d9/eH4FsCwvgBpnc9n0pGE+ffKdph38s6foEZiEjdgHdzp+IA==} + hono@4.12.12: resolution: {integrity: sha512-p1JfQMKaceuCbpJKAPKVqyqviZdS0eUxH9v82oWo1kb9xjQ5wA6iP3FNVAPDFlz5/p7d45lO+BpSk1tuSZMF4Q==} engines: {node: '>=16.9.0'} @@ -6837,6 +7006,9 @@ packages: resolution: {integrity: sha512-CV9TW3Y3f8/wT0BRFc1/KAVQ3TUHiXmaAb6VW9vtiMFf7SLoMd1PdAc4W3KFOFETBJUb90KatHqlsZMWV+R9Gg==} engines: {node: ^20.19.0 || ^22.12.0 || >=24.0.0} + html-url-attributes@3.0.1: + resolution: {integrity: sha512-ol6UPyBWqsrO6EJySPz2O7ZSr856WDrEzM5zMqp+FJJLGMW35cLYmmZnl0vztAZxRUoNZJFTCohfjuIJ8I4QBQ==} + http-errors@2.0.1: resolution: {integrity: sha512-4FbRdAX+bSdmo4AUFuS0WNiPz8NgFt+r8ThgNWmlrjQjt1Q7ZR9+zTlce2859x4KSXrwIsaeTqDoKQmtP8pLmQ==} engines: {node: '>= 0.8'} @@ -6947,6 +7119,9 @@ packages: react-devtools-core: optional: true + inline-style-parser@0.2.7: + resolution: {integrity: sha512-Nb2ctOyNR8DqQoR0OwRG95uNWIC0C1lCgf5Naz5H6Ji72KZ8OcFZLz2P5sNgwlyoJ8Yif11oMuYs5pBQa86csA==} + inline-style-prefixer@6.0.4: resolution: {integrity: sha512-FwXmZC2zbeeS7NzGjJ6pAiqRhXR0ugUShSNb6GApMl6da0/XGc4MOJsoWAywia52EEWbXNSy0pzkwz/+Y+swSg==} @@ -6965,6 +7140,12 @@ packages: resolution: {integrity: sha512-0KI/607xoxSToH7GjN1FfSbLoU0+btTicjsQSWQlh/hZykN8KpmMf7uYwPW3R+akZ6R/w18ZlXSHBYXiYUPO3g==} engines: {node: '>= 0.10'} + is-alphabetical@2.0.1: + resolution: {integrity: sha512-FWyyY60MeTNyeSRpkM2Iry0G9hpr7/9kD40mD/cGQEuilcZYS4okz8SN2Q6rLCJ8gbCt6fN+rC+6tMGS99LaxQ==} + + is-alphanumerical@2.0.1: + resolution: {integrity: sha512-hmbYhX/9MUMF5uh7tOXyK/n0ZvWpad5caBA17GsC6vyuCqaWliRG5K1qS9inmUhEMaOBIW7/whAnSwveW/LtZw==} + is-arrayish@0.2.1: resolution: {integrity: sha512-zz06S8t0ozoDXMG+ube26zeCTNXcKIPJZJi8hBrF4idCLms4CG9QtK7qBl1boi5ODzFpjswb5JPmHCbMpjaYzg==} @@ -6975,6 +7156,9 @@ packages: resolution: {integrity: sha512-evOr8xfXKxE6qSR0hSXL2r3sd7ALj8+7jQEUvPYcm5sgZFdJ+AYzT6yNmJenvIYQBgIGwfwz08sL8zoL7yq2BA==} engines: {node: '>= 0.4'} + is-decimal@2.0.1: + resolution: {integrity: sha512-AAB9hiomQs5DXWcRB1rqsxGUstbRroFOPPVAomNk/3XHR5JyEZChOyTWe2oayKnsSsr/kcGqF+z6yuH6HHpN0A==} + is-directory@0.3.1: resolution: {integrity: sha512-yVChGzahRFvbkscn2MlwGismPO12i9+znNruC5gVEntG3qu0xQMzsGg/JFbrsqDOHtHFPci+V5aP5T9I+yeKqw==} engines: {node: '>=0.10.0'} @@ -7008,6 +7192,9 @@ packages: resolution: {integrity: sha512-xelSayHH36ZgE7ZWhli7pW34hNbNl8Ojv5KVmkJD4hBdD3th8Tfk9vYasLM+mXWOZhFkgZfxhLSnrwRr4elSSg==} engines: {node: '>=0.10.0'} + is-hexadecimal@2.0.1: + resolution: {integrity: sha512-DgZQp241c8oO6cA1SbTEWiXeoxV42vlcJxgH+B3hi1AiqqKruZR3ZGF8In3fj4+/y/7rHvlOZLZtgJ/4ttYGZg==} + is-in-ci@2.0.0: resolution: {integrity: sha512-cFeerHriAnhrQSbpAxL37W1wcJKUUX07HyLWZCW1URJT/ra3GyUTzBgUnh24TMVfNTV2Hij2HLxkPHFZfOZy5w==} engines: {node: '>=20'} @@ -7250,6 +7437,10 @@ packages: resolution: {integrity: sha512-kpSuLD3/7RenBnjnJdOHXCKC8dTd1JzeOiJhN0necWWci6cC+qX+VuwPnMVgb+a4+KNJSfgqahpnfWaeDXCimw==} engines: {node: '>=18.0.0'} + katex@0.16.47: + resolution: {integrity: sha512-Eeo8Ys1doU1z+x8AZsPpQu+p/QcZBI5PeOo7QGQdy2x2m0MU/hYagBbGOmXwr5KVbEfVuWv9LpnQWeehogurjg==} + hasBin: true + keyv-file@5.3.3: resolution: {integrity: sha512-uCFUhiVYf+BcA6DP4smhnRLOR4yzUUA15yJSk4/rP5oAPnF3MpfajejwvSV8l+okm+EGiW4IHP3gn+xahpvZcQ==} @@ -7388,6 +7579,9 @@ packages: resolution: {integrity: sha512-t7hw9pI+WvuwNJXwk5zVHpyhIqzg2qTlklJOf0mVxGSbe3Fp2VieZcduNYjaLDoy6p9uGpQEGWG87WpMKlNq8g==} engines: {node: '>=8'} + lodash-es@4.18.1: + resolution: {integrity: sha512-J8xewKD/Gk22OZbhpOVSwcs60zhd95ESDwezOFuA3/099925PdHJ7OFHNTGtajL3AlZkykD32HykiMo+BIBI8A==} + lodash.debounce@4.0.8: resolution: {integrity: sha512-FT1yDzDYEoYWhnSGnpE/4Kj1fLZkDFyqRb7fNt6FdYOSxlUWAtp42Eh6Wb0rGIv/m9Bgo7x4GhQbm5Ys4SG5ow==} @@ -7418,6 +7612,9 @@ packages: lodash.throttle@4.1.1: resolution: {integrity: sha512-wIkUCfVKpVsWo3JSZlc+8MB5it+2AN5W8J7YVMST30UrvcQNZ1Okbj+rbVniijTWE6FGYy4XJq/rHkas8qJMLQ==} + lodash@4.18.1: + resolution: {integrity: sha512-dMInicTPVE8d1e5otfwmmjlxkZoUpiVLwyeTdUsi/Caj/gfzzblBcCE5sRHV/AsjuCmxWrte2TNGSYuCeCq+0Q==} + log-symbols@2.2.0: resolution: {integrity: sha512-VeIAFslyIerEJLXHziedo2basKbMKtTw3vfn5IzG0XTjhAVEJyNHnL2p7vc+wBDSdQuUpNw3M2u6xb9QsAY5Eg==} engines: {node: '>=4'} @@ -7443,6 +7640,9 @@ packages: loupe@3.2.1: resolution: {integrity: sha512-CdzqowRJCeLU72bHvWqwRBBlLcMEtIvGrlvef74kMnV2AolS9Y8xUv1I0U/MNAWMhBlKIoyuEgoJ0t/bbwHbLQ==} + lowlight@1.20.0: + resolution: {integrity: sha512-8Ktj+prEb1RoCPkEOrPMYUN/nCggB7qAWe3a7OpMjWQkh3l2RD5wKRQ+o8Q8YuI9RG/xs95waaI/E6ym/7NsTw==} + lru-cache@10.4.3: resolution: {integrity: sha512-JNAzZcXrCt42VGLuYz0zfAzDfAvJWW6AfYlDBQyDV5DClI2m5sAmK+OIO7s59XfsRsWHp02jAJrRadPRGTt6SQ==} @@ -7457,11 +7657,21 @@ packages: resolution: {integrity: sha512-jumlc0BIUrS3qJGgIkWZsyfAM7NCWiBcCDhnd+3NNM5KbBmLTgHVfWBcg6W+rLUsIpzpERPsvwUP7CckAQSOoA==} engines: {node: '>=12'} + lucide-react@0.562.0: + resolution: {integrity: sha512-82hOAu7y0dbVuFfmO4bYF1XEwYk/mEbM5E+b1jgci/udUBEE/R7LF5Ip0CCEmXe8AybRM8L+04eP+LGZeDvkiw==} + peerDependencies: + react: ^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0 + lucide-react@0.564.0: resolution: {integrity: sha512-JJ8GVTQqFwuliifD48U6+h7DXEHdkhJ/E87kksGByII3qHxtPciVb8T8woQONHBQgHVOl7rSMrrip3SeVNy7Fg==} peerDependencies: react: ^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0 + lucide-react@0.575.0: + resolution: {integrity: sha512-VuXgKZrk0uiDlWjGGXmKV6MSk9Yy4l10qgVvzGn2AWBx1Ylt0iBexKOAoA6I7JO3m+M9oeovJd3yYENfkUbOeg==} + peerDependencies: + react: ^16.5.1 || ^17.0.0 || ^18.0.0 || ^19.0.0 + magic-string@0.30.21: resolution: {integrity: sha512-vd2F4YUyEXKGcLHoq+TEyCjxueSeHnFxyyjNp80yg0XV4vUhnDer/lvvlqM/arB5bXQN5K2/3oinyCRyx8T2CQ==} @@ -7521,9 +7731,27 @@ packages: mdast-util-gfm@3.1.0: resolution: {integrity: sha512-0ulfdQOM3ysHhCJ1p06l0b0VKlhU0wuQs3thxZQagjcjPrlFRqY215uZGHHJan9GEAXd9MbfPjFJz+qMkVR6zQ==} + mdast-util-math@3.0.0: + resolution: {integrity: sha512-Tl9GBNeG/AhJnQM221bJR2HPvLOSnLE/T9cJI9tlc6zwQk2nPk/4f0cHkOdEixQPC/j8UtKDdITswvLAy1OZ1w==} + + mdast-util-mdx-expression@2.0.1: + resolution: {integrity: sha512-J6f+9hUp+ldTZqKRSg7Vw5V6MqjATc+3E4gf3CFNcuZNWD8XdyI6zQ8GqH7f8169MM6P7hMBRDVGnn7oHB9kXQ==} + + mdast-util-mdx-jsx@3.2.0: + resolution: {integrity: sha512-lj/z8v0r6ZtsN/cGNNtemmmfoLAFZnjMbNyLzBafjzikOM+glrjNHPlf6lQDOTccj9n5b0PPihEBbhneMyGs1Q==} + + mdast-util-mdxjs-esm@2.0.1: + resolution: {integrity: sha512-EcmOpxsZ96CvlP03NghtH1EsLtr0n9Tm4lPUJUBccV9RwUOneqSycg19n5HGzCf+10LozMRSObtVr3ee1WoHtg==} + + mdast-util-newline-to-break@2.0.0: + resolution: {integrity: sha512-MbgeFca0hLYIEx/2zGsszCSEJJ1JSCdiY5xQxRcLDDGa8EPvlLPupJ4DSajbMPAnC0je8jfb9TiUATnxxrHUog==} + mdast-util-phrasing@4.1.0: resolution: {integrity: sha512-TqICwyvJJpBwvGAMZjj4J2n0X8QWp21b9l0o7eXyVJ25YNWYbJDVIyD1bZXE6WtV6RmKJVYmQAKWa0zWOABz2w==} + mdast-util-to-hast@13.2.1: + resolution: {integrity: sha512-cctsq2wp5vTsLIcaymblUriiTcZd0CwWtCbLvrOzYCDZoWyMNV8sZ7krj09FSnsiJi3WVsHLM4k6Dq/yaPyCXA==} + mdast-util-to-markdown@2.1.2: resolution: {integrity: sha512-xj68wMTvGXVOKonmog6LwyJKrYXZPvlwabaryTjLh9LuvovB/KAH+kvi8Gjj+7rJjsFi23nkUxRQv1KqSroMqA==} @@ -7820,6 +8048,9 @@ packages: micromark-extension-gfm@3.0.0: resolution: {integrity: sha512-vsKArQsicm7t0z2GugkCKtZehqUm31oeGBV/KVSorWSy8ZlNAv7ytjFhvaryUiCUJYqs+NoE6AFhpQvBTM6Q4w==} + micromark-extension-math@3.1.0: + resolution: {integrity: sha512-lvEqd+fHjATVs+2v/8kg9i5Q0AP2k85H0WUOwpIVvUML8BapsMvh1XAogmQjOCsLpoKRCVQqEkQBB3NhVBcsOg==} + micromark-factory-destination@2.0.1: resolution: {integrity: sha512-Xe6rDdJlkmbFRExpTOmRj9N3MaWmbAgdpSrBQvCFqhezUn4AHqJHbaEnfbVYYiexVSs//tqOdY/DxhjdCiJnIA==} @@ -8082,6 +8313,10 @@ packages: engines: {node: '>=10.5.0'} deprecated: Use your platform's native DOMException instead + node-emoji@2.2.0: + resolution: {integrity: sha512-Z3lTE9pLaJF47NyMhd4ww1yFTAP8YhYI8SleJiHzM46Fgpm5cnNzSl9XfzFNqbaz+VlJrIj3fXQ4DeN1Rjm6cw==} + engines: {node: '>=18'} + node-fetch@2.7.0: resolution: {integrity: sha512-c4FRfUm/dbcWZ7U+1Wq0AwCyFL+3nt2bEw05wfxSz+DWpWsitgmSgYmy2dQdWyKC1694ELPqMs/YzUSNozLt8A==} engines: {node: 4.x || >=6.0.0} @@ -8346,6 +8581,9 @@ packages: package-manager-detector@0.2.11: resolution: {integrity: sha512-BEnLolu+yuz22S56CU1SUKq3XC3PkwD5wv4ikR4MfGvnRVcmzXR9DwSlW2fEamyTPyXHomBJRzgapeuBvRNzJQ==} + parse-entities@4.0.2: + resolution: {integrity: sha512-GG2AQYWoLgL877gQIKeRPGO1xF9+eG1ujIb5soS5gPvLQ1y2o8FL90w2QWNdf9I361Mpp7726c+lj3U0qK1uGw==} + parse-json@4.0.0: resolution: {integrity: sha512-aOIos8bujGN93/8Ox/jPLh7RwVnPEysynVFE+fQZyg6jKELEHwzgKdLRFHUgXJL6kylijVSBC4BvN9OmsB48Rw==} engines: {node: '>=4'} @@ -8358,6 +8596,9 @@ packages: resolution: {integrity: sha512-Nt/a5SfCLiTnQAjx3fHlqp8hRgTL3z7kTQZzvIMS9uCAepnCyjpdEc6M/sz69WqMBdaDBw9sF1F1UaHROYzGkQ==} engines: {node: '>=10'} + parse5@7.3.0: + resolution: {integrity: sha512-IInvU7fabl34qmi9gY8XOVxhYyMyuH2xUNpb2q8/Y+7552KlejkRvqvD19nMoUW/uQGGbqNpA6Tufu5FL5BZgw==} + parse5@8.0.0: resolution: {integrity: sha512-9m4m5GSgXjL4AjumKzq1Fgfp3Z8rsvjRNbnkVwfu2ImRqE5D0LnY2QfDen18FSY9C573YU5XxSapdHZTZ2WolA==} @@ -8594,6 +8835,10 @@ packages: resolution: {integrity: sha512-gjVS5hOP+M3wMm5nmNOucbIrqudzs9v/57bWRHQWLYklXqoXKrVfYW2W9+glfGsqtPgpiz5WwyEEB+ksXIx3gQ==} engines: {node: '>=18'} + prismjs@1.30.0: + resolution: {integrity: sha512-DEvV2ZF2r2/63V+tK8hQvrR2ZGn10srHbXviTlcv7Kpzw8jWiNTqbVgjO3IY8RxrrOUF8VPMQQFysYYYv0YZxw==} + engines: {node: '>=6'} + proc-log@4.2.0: resolution: {integrity: sha512-g8+OnU/L2v+wyiVK+D5fA34J7EH8jZ8DDlvwhRCMxmMj7UCBvxiO1mGeN+36JXIKF4zevU4kRBd8lVgG9vLelA==} engines: {node: ^14.17.0 || ^16.13.0 || >=18.0.0} @@ -8620,6 +8865,12 @@ packages: resolution: {integrity: sha512-NxNv/kLguCA7p3jE8oL2aEBsrJWgAakBpgmgK6lpPWV+WuOmY6r2/zbAVnP+T8bQlA0nzHXSJSJW0Hq7ylaD2Q==} engines: {node: '>= 6'} + prop-types@15.8.1: + resolution: {integrity: sha512-oj87CgZICdulUohogVAR7AjlC0327U4el4L6eAvOqCeudMDVU0NThNaV+b9Df4dXgSP1gXMTnPdhfe/2qDH5cg==} + + property-information@7.2.0: + resolution: {integrity: sha512-IAtzIB6sUiWaJYrX9smp3V46pBGbBeLFRGdh25kg1334VcBlD8HzhPeNIWQH9zhGmo2itIe25EHt9dQP7G5hmg==} + proto-list@1.2.4: resolution: {integrity: sha512-vtK/94akxsTMhe0/cbfpR+syPuszcuwhqVjJq26CuNDgFGj682oRBXOP5MJpv2r7JtE8MsiepGIqvvOTBwn2vA==} @@ -8705,6 +8956,12 @@ packages: resolution: {integrity: sha512-y3bGgqKj3QBdxLbLkomlohkvsA8gdAiUQlSBJnBhfn+BPxg4bc62d8TcBW15wavDfgexCgccckhcZvywyQYPOw==} hasBin: true + react-day-picker@9.14.0: + resolution: {integrity: sha512-tBaoDWjPwe0M5pGrum4H0SR6Lyk+BO9oHnp9JbKpGKW2mlraNPgP9BMfsg5pWpwrssARmeqk7YBl2oXutZTaHA==} + engines: {node: '>=18'} + peerDependencies: + react: '>=16.8.0' + react-devtools-core@5.3.2: resolution: {integrity: sha512-crr9HkVrDiJ0A4zot89oS0Cgv0Oa4OG1Em4jit3P3ZxZSKPMYyMjfwMqgcJna9o625g8oN87rBm8SWWrSTBZxg==} @@ -8716,12 +8973,21 @@ packages: peerDependencies: react: ^19.2.4 + react-is@16.13.1: + resolution: {integrity: sha512-24e6ynE2H+OKt4kqsOvNd8kBpV65zoxbA4BVsEOB3ARVWQki/DHzaUoC5KuON/BiccDaCCTZBuOcfZs70kR8bQ==} + react-is@18.3.1: resolution: {integrity: sha512-/LLMVyas0ljjAtoYiPqYiL8VWXzUUdThrmU5+n20DZv+a+ClRoevUzw5JxU+Ieh5/c87ytoTBV9G1FiKfNJdmg==} react-is@19.2.4: resolution: {integrity: sha512-W+EWGn2v0ApPKgKKCy/7s7WHXkboGcsrXE+2joLyVxkbyVQfO3MUEaUQDHoSmb8TFFrSKYa9mw64WZHNHSDzYA==} + react-markdown@10.1.0: + resolution: {integrity: sha512-qKxVopLT/TyA6BX3Ue5NwabOsAzm0Q7kAPwq6L+wWDwisYs7R8vZ0nRXqq6rkueboxpkjvLGU9fWifiX/ZZFxQ==} + peerDependencies: + '@types/react': '>=18' + react: '>=18' + react-native-is-edge-to-edge@1.3.1: resolution: {integrity: sha512-NIXU/iT5+ORyCc7p0z2nnlkouYKX425vuU1OEm6bMMtWWR9yvb+Xg5AZmImTKoF9abxCPqrKC3rOZsKzUYgYZA==} peerDependencies: @@ -8802,6 +9068,12 @@ packages: '@types/react': optional: true + react-smooth@4.0.4: + resolution: {integrity: sha512-gnGKTpYwqL0Iii09gHobNolvX4Kiq4PKx6eWBCYYix+8cdw+cGo3do906l1NBPKkSWx1DghC1dlWG9L2uGd61Q==} + peerDependencies: + react: ^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 + react-dom: ^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 + react-style-singleton@2.2.3: resolution: {integrity: sha512-b6jSvxvVnyptAiLjbkWLE/lOnR4lfTtDAl+eUC7RZy+QQWc6wRzIV2CE6xBuMmDxc2qIihtDCZD5NPOFl7fRBQ==} engines: {node: '>=10'} @@ -8812,6 +9084,18 @@ packages: '@types/react': optional: true + react-syntax-highlighter@16.1.1: + resolution: {integrity: sha512-PjVawBGy80C6YbC5DDZJeUjBmC7skaoEUdvfFQediQHgCL7aKyVHe57SaJGfQsloGDac+gCpTfRdtxzWWKmCXA==} + engines: {node: '>= 16.20.2'} + peerDependencies: + react: '>= 0.14.0' + + react-transition-group@4.4.5: + resolution: {integrity: sha512-pZcd1MCJoiKiBR2NRxeCRg13uCXbydPnmB4EOeRrY7480qNWO8IIgQG6zlDkm6uRMsURXPuKq0GWtiM59a5Q6g==} + peerDependencies: + react: '>=16.6.0' + react-dom: '>=16.6.0' + react@19.1.0: resolution: {integrity: sha512-FS+XFBNvn3GTAWq26joslQgWNoFu08F4kl0J4CgdNKADkdSGXQyTCnKteIAJy96Br6YbpEU1LSzV5dYtjMkMDg==} engines: {node: '>=0.10.0'} @@ -8846,6 +9130,17 @@ packages: resolution: {integrity: sha512-hjMmLaUXAm1hIuTqOdeYObMslq/q+Xff6QE3Y2P+uoHAg2nmVlLBps2hzh1UJDdMtDTMXOFewK6ky51JQIeECg==} engines: {node: '>= 4'} + recharts-scale@0.4.5: + resolution: {integrity: sha512-kivNFO+0OcUNu7jQquLXAxz1FIwZj8nrj+YkOKc5694NbjCvcT6aSZiIzNzd2Kul4o4rTto8QVR9lMNtxD4G1w==} + + recharts@2.15.4: + resolution: {integrity: sha512-UT/q6fwS3c1dHbXv2uFgYJ9BMFHu3fwnd7AYZaEQhXuYQ4hgsxLvsUXzGdKeZrW5xopzDCvuA2N41WJ88I7zIw==} + engines: {node: '>=14'} + deprecated: 1.x and 2.x branches are no longer active. Bump to Recharts v3 to receive latest features and bugfixes. See https://github.com/recharts/recharts/wiki/3.0-migration-guide + peerDependencies: + react: ^16.0.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 + react-dom: ^16.0.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 + recharts@3.7.0: resolution: {integrity: sha512-l2VCsy3XXeraxIID9fx23eCb6iCBsxUQDnE8tWm6DFdszVAO7WVY/ChAD9wVit01y6B2PMupYiMmQwhgPHc9Ew==} engines: {node: '>=18'} @@ -8866,6 +9161,9 @@ packages: redux@5.0.1: resolution: {integrity: sha512-M9/ELqF6fy8FwmkpnF0S3YKOqMyoWJ4+CS5Efg2ct3oY9daQvd/Pc71FpGZsVsbl3Cpb+IIcjBDUnnyBdQbq4w==} + refractor@5.0.0: + resolution: {integrity: sha512-QXOrHQF5jOpjjLfiNk5GFnWhRXvxjUVnlFxkeDmewR5sXkr3iM46Zo+CnRR8B+MDVqkULW4EcLVcRBNOPXHosw==} + regenerate-unicode-properties@10.2.2: resolution: {integrity: sha512-m03P+zhBeQd1RGnYxrGyDAPpWX/epKirLrp8e3qevZdVkKtnCrjjWczIbYc8+xd6vcTStVlqfycTx1KR4LOr0g==} engines: {node: '>=4'} @@ -8887,12 +9185,28 @@ packages: resolution: {integrity: sha512-NgRBy2Nx/bE+9F27nVHnqcN5HjyLmecqsqx2PJHu3/IEtADD4WuxuXIVExD5PoSDFVrl78dOonfcOe5O+5nbzQ==} hasBin: true + rehype-katex@7.0.1: + resolution: {integrity: sha512-OiM2wrZ/wuhKkigASodFoo8wimG3H12LWQaH8qSPVJn9apWKFSH3YOCtbKpBorTVw/eI7cuT21XBbvwEswbIOA==} + + remark-breaks@4.0.0: + resolution: {integrity: sha512-IjEjJOkH4FuJvHZVIW0QCDWxcG96kCq7An/KVH2NfJe6rKZU2AsHeB3OEjPNRxi4QC34Xdx7I2KGYn6IpT7gxQ==} + + remark-emoji@5.0.2: + resolution: {integrity: sha512-IyIqGELcyK5AVdLFafoiNww+Eaw/F+rGrNSXoKucjo95uL267zrddgxGM83GN1wFIb68pyDuAsY3m5t2Cav1pQ==} + engines: {node: '>=18'} + remark-gfm@4.0.1: resolution: {integrity: sha512-1quofZ2RQ9EWdeN34S79+KExV1764+wCUGop5CPL1WGdD0ocPpu91lzPGbwWMECpEpd42kJGQwzRfyov9j4yNg==} + remark-math@6.0.0: + resolution: {integrity: sha512-MMqgnP74Igy+S3WwnhQ7kqGlEerTETXMvJhrUzDikVZ2/uogJCb+WHUg97hK9/jcfc0dkD73s3LN8zU49cTEtA==} + remark-parse@11.0.0: resolution: {integrity: sha512-FCxlKLNGknS5ba/1lmpYijMUzX2esxW5xQqjWxw2eHFfS2MSdaHVINFmhjo+qN1WhZhNimq0dZATN9pH0IDrpA==} + remark-rehype@11.1.2: + resolution: {integrity: sha512-Dh7l57ianaEoIpzbp0PC9UKAdCSVklD8E5Rpw7ETfbTl3FqcOOgq5q2LVDhgGCkaBv7p24JXikPdvhhmHvKMsw==} + remark-stringify@11.0.0: resolution: {integrity: sha512-1OSmLd3awB/t8qdoEOMazZkNsfVTeY4fTsgzcQFdXNq8ToTN4ZGwrMnlda4K6smTFKD+GRV6O48i6Z4iKgPPpw==} @@ -9165,6 +9479,10 @@ packages: sisteransi@1.0.5: resolution: {integrity: sha512-bLGGlR1QxBcynn2d5YmDX4MGjlZvy2MRBDRNHLJ8VI6l6+9FUiyTFNJ0IveOSP0bcXgVDPRcfGqA0pjaqUpfVg==} + skin-tone@2.0.0: + resolution: {integrity: sha512-kUMbT1oBJCpgrnKoSr0o6wPtvRWT9W9UKvGLwfJYO2WuahZRHOpEyL1ckyMGgMWh0UdpmaoFqKKD29WTomNEGA==} + engines: {node: '>=8'} + slash@3.0.0: resolution: {integrity: sha512-g9Q1haeby36OSStwb4ntCGGGaKsaVSjQ68fBxoQcutl5fS1vuY18H3wSt3jFyFtrkx+Kz0V1G85A4MyAdDMi2Q==} engines: {node: '>=8'} @@ -9223,6 +9541,9 @@ packages: resolution: {integrity: sha512-UjgapumWlbMhkBgzT7Ykc5YXUT46F0iKu8SGXq0bcwP5dz/h0Plj6enJqjz1Zbq2l5WaqYnrVbwWOWMyF3F47g==} engines: {node: '>=0.10.0'} + space-separated-tokens@2.0.2: + resolution: {integrity: sha512-PEGlAwrG8yXGXRjW32fGbg66JAlOAwbObuqVoJpv/mRgoWDQfgH1wDPvtzWyUSNAXBGSk8h755YDbbcEy3SH2Q==} + sparse-bitfield@3.0.3: resolution: {integrity: sha512-kvzhi7vqKTfkh0PZU+2D2PIllw2ymqJKujUcyPMd9Y75Nv4nPbGJZXNhxsgdQab2BmlDct1YnfQCguEvHr7VsQ==} @@ -9301,6 +9622,9 @@ packages: string_decoder@1.3.0: resolution: {integrity: sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA==} + stringify-entities@4.0.4: + resolution: {integrity: sha512-IwfBptatlO+QCJUo19AqvrPNqlVMpW9YEL2LIVY+Rpv2qsjCGxaDLNRgeGsQWJhfItebuJhsGSLjaBbNSQ+ieg==} + strip-ansi@5.2.0: resolution: {integrity: sha512-DuRs1gKbBqsMKIZlrffwlug8MHkcnpjs5VPmL1PAh+mA30U0DTotfDZ0d2UUsXpPmPmMMJ6W773MaA3J+lbiWA==} engines: {node: '>=6'} @@ -9342,6 +9666,12 @@ packages: structured-headers@0.4.1: resolution: {integrity: sha512-0MP/Cxx5SzeeZ10p/bZI0S6MpgD+yxAhi1BOQ34jgnMXsCq3j1t6tQnZu+KdlL7dvJTLT3g9xN8tl10TqgFMcg==} + style-to-js@1.1.21: + resolution: {integrity: sha512-RjQetxJrrUJLQPHbLku6U/ocGtzyjbJMP9lCNK7Ag0CNh690nSH8woqWH9u16nMjYBAok+i7JO1NP2pOy8IsPQ==} + + style-to-object@1.0.14: + resolution: {integrity: sha512-LIN7rULI0jBscWQYaSswptyderlarFkjQ+t79nzty8tcIAceVomEVlLzH5VP4Cmsv6MtKhs7qaAiwlcp+Mgaxw==} + styleq@0.1.3: resolution: {integrity: sha512-3ZUifmCDCQanjeej1f6kyl/BeP/Vae5EYkQ9iJfUm/QwZvlgnZzyflqAsAWYURdtea8Vkvswu2GrC57h3qffcA==} @@ -9520,6 +9850,9 @@ packages: resolution: {integrity: sha512-bLVMLPtstlZ4iMQHpFHTR7GAGj2jxi8Dg0s2h2MafAE4uSWF98FC/3MomU51iQAMf8/qDUbKWf5GxuvvVcXEhw==} engines: {node: '>=20'} + trim-lines@3.0.1: + resolution: {integrity: sha512-kRj8B+YHZCc9kQYdWfJB2/oUl9rA99qbowYYBtr4ui4mZyAQ2JpvVBd/6U2YloATfqBhBTSMhTpgBHtU0Mf3Rg==} + triple-beam@1.4.1: resolution: {integrity: sha512-aZbgViZrg1QNcG+LULa7nhZpJTZSLm/mXnHXnbAbjmN5aSa0y7V+wvv6+4WaBtpISJzThKy+PIPxc1Nq1EJ9mg==} engines: {node: '>= 14.0.0'} @@ -9662,6 +9995,10 @@ packages: resolution: {integrity: sha512-dA8WbNeb2a6oQzAQ55YlT5vQAWGV9WXOsi3SskE3bcCdM0P4SDd+24zS/OCacdRq5BkdsRj9q3Pg6YyQoxIGqg==} engines: {node: '>=4'} + unicode-emoji-modifier-base@1.0.0: + resolution: {integrity: sha512-yLSH4py7oFH3oG/9K+XWrz1pSi3dfUrWEnInbxMfArOfc1+33BlGPQtLsOYwvdMy11AwUBetYuaRxSPqgkq+8g==} + engines: {node: '>=4'} + unicode-match-property-ecmascript@2.0.0: resolution: {integrity: sha512-5kaZCrbp5mmbz5ulBkDkbY0SsPOjKqVS35VpL9ulMPfSl0J0Xsm+9Evphv9CoIZFwre7aJoa94AY6seMKGVN5Q==} engines: {node: '>=4'} @@ -9681,9 +10018,18 @@ packages: unified@11.0.5: resolution: {integrity: sha512-xKvGhPWw3k84Qjh8bI3ZeJjqnyadK+GEFtazSfZv/rKeTkTjOJho6mFqh2SM96iIcZokxiOpg78GazTSg8+KHA==} + unist-util-find-after@5.0.0: + resolution: {integrity: sha512-amQa0Ep2m6hE2g72AugUItjbuM8X8cGQnFoHk0pGfrFeT9GZhzN5SW8nRsiGKK7Aif4CrACPENkA6P/Lw6fHGQ==} + unist-util-is@6.0.1: resolution: {integrity: sha512-LsiILbtBETkDz8I9p1dQ0uyRUWuaQzd/cuEeS1hoRSyW5E5XGmTzlwY1OrNzzakGowI9Dr/I8HVaw4hTtnxy8g==} + unist-util-position@5.0.0: + resolution: {integrity: sha512-fucsC7HjXvkB5R3kTCO7kUjRdrS0BJt3M/FPxmHMBOm8JQi2BsHAHFsy27E0EolP8rp0NzXsJ+jNPyDWvOJZPA==} + + unist-util-remove-position@5.0.0: + resolution: {integrity: sha512-Hp5Kh3wLxv0PHj9m2yZhhLt58KzPtEYKQQ4yxfYFEO7EvHwzyDYnduhHnY1mDxoqr7VUwVuHXk9RXKIiYS1N8Q==} + unist-util-stringify-position@4.0.0: resolution: {integrity: sha512-0ASV06AAoKCDkS2+xw5RXJywruurpbC4JZSm7nr7MOt1ojAzvyyaO+UxZf18j8FCF6kmzCZKcAgN/yu2gm2XgQ==} @@ -9803,12 +10149,18 @@ packages: resolution: {integrity: sha512-Ck0EJbAGxHwprkzFO966t4/5QkRuzh+/I1RxhLgUKKwEn+Cd8NwM60mE3AqBZg5gYODoXW0EFsQvbZjRlvdqbg==} engines: {node: '>=4'} + vfile-location@5.0.3: + resolution: {integrity: sha512-5yXvWDEgqeiYiBe1lbxYF7UMAIm/IcopxMHrMQDq3nvKcjPKIhZklUKL+AE7J7uApI4kwe2snsK+eI6UTj9EHg==} + vfile-message@4.0.3: resolution: {integrity: sha512-QTHzsGd1EhbZs4AsQ20JX1rC3cOlt/IWJruk893DfLRr57lcnOeMaWG4K0JrRta4mIJZKth2Au3mM3u03/JWKw==} vfile@6.0.3: resolution: {integrity: sha512-KzIbH/9tXat2u30jf+smMwFCsno4wHVdNmzFyL+T/L3UGqqk6JKfVqOFOZEpZSHADH1k40ab6NUIXZq422ov3Q==} + victory-vendor@36.9.2: + resolution: {integrity: sha512-PnpQQMuxlwYdocC8fIJqVXvkeViHYzotI+NJrCuav0ZYFoq912ZHBk3mCeuj+5/VpodOjPe1z0Fk2ihgzlXqjQ==} + victory-vendor@37.3.6: resolution: {integrity: sha512-SbPDPdDBYp+5MJHhBCAyI7wKM3d5ivekigc2Dk2s7pgbZ9wIgIBYGVw4zGHBml/qTFbexrofXW6Gu4noGxrOwQ==} @@ -9918,6 +10270,9 @@ packages: wcwidth@1.0.1: resolution: {integrity: sha512-XHPEwS0q6TaxcvG85+8EYkbiCux2XtWG2mkc47Ng2A77BQu9+DqIOJldST4HgPkuea7dvKSj5VgX3P1d4rW8Tg==} + web-namespaces@2.0.1: + resolution: {integrity: sha512-bKr1DkiNa2krS7qxNtdrtHAmzuYGFQLiQ13TsorsdT6ULTkPLKuu5+GsFpDlg6JFjUTwX2DyhMPG2be8uPrqsQ==} + web-streams-polyfill@3.3.3: resolution: {integrity: sha512-d2JWLCivmZYTSIoge9MsgFCZrt571BikcWGYkjC1khllbTeDlGqZ2D8vD8E/lJa8WGWbb7Plm8/XJYV7IJHZZw==} engines: {node: '>= 8'} @@ -10162,6 +10517,21 @@ packages: zod@4.4.3: resolution: {integrity: sha512-ytENFjIJFl2UwYglde2jchW2Hwm4GJFLDiSXWdTrJQBIN9Fcyp7n4DhxJEiWNAJMV1/BqWfW/kkg71UDcHJyTQ==} + zustand@4.5.7: + resolution: {integrity: sha512-CHOUy7mu3lbD6o6LJLfllpjkzhHXSBlX8B9+qPddUsIfeF5S/UZ5q0kmCsnRqT1UHFQZchNFDDzMbQsuesHWlw==} + engines: {node: '>=12.7.0'} + peerDependencies: + '@types/react': '>=16.8' + immer: '>=9.0.6' + react: '>=16.8' + peerDependenciesMeta: + '@types/react': + optional: true + immer: + optional: true + react: + optional: true + zwitch@2.0.4: resolution: {integrity: sha512-bXE4cR/kVZhKZX/RjPEflHaKVhUVl85noU3v6b8apfQEc1x4A+zBxjZ4lN8LqGd6WZ3dl98pY4o717VFmoPp+A==} @@ -10185,6 +10555,10 @@ snapshots: uuid: 11.1.1 zod: 3.25.76 + '@ag-ui/core@0.0.53': + dependencies: + zod: 3.25.76 + '@ag-ui/core@0.0.57': dependencies: zod: 3.25.76 @@ -10206,6 +10580,14 @@ snapshots: '@ai-sdk/provider-utils': 5.0.5(zod@4.4.3) zod: 4.4.3 + '@ai-sdk/gateway@3.0.110(zod@3.25.76)': + dependencies: + '@ai-sdk/provider': 3.0.10 + '@ai-sdk/provider-utils': 4.0.26(zod@3.25.76) + '@vercel/oidc': 3.2.0 + zod: 3.25.76 + optional: true + '@ai-sdk/gateway@3.0.110(zod@4.4.3)': dependencies: '@ai-sdk/provider': 3.0.10 @@ -10226,6 +10608,14 @@ snapshots: '@ai-sdk/provider-utils': 5.0.5(zod@4.4.3) zod: 4.4.3 + '@ai-sdk/provider-utils@4.0.26(zod@3.25.76)': + dependencies: + '@ai-sdk/provider': 3.0.10 + '@standard-schema/spec': 1.1.0 + eventsource-parser: 3.0.8 + zod: 3.25.76 + optional: true + '@ai-sdk/provider-utils@4.0.26(zod@4.4.3)': dependencies: '@ai-sdk/provider': 3.0.10 @@ -12493,6 +12883,8 @@ snapshots: enabled: 2.0.0 kuler: 2.0.0 + '@date-fns/tz@1.5.0': {} + '@dimforge/rapier3d-compat@0.12.0': {} '@emnapi/runtime@1.8.1': @@ -13703,12 +14095,90 @@ snapshots: '@opentelemetry/semantic-conventions@1.40.0': {} + '@openuidev/devtools@0.0.4(@openuidev/observability@0.0.2)(react-dom@19.2.4(react@19.2.4))(react@19.2.4)': + dependencies: + '@openuidev/observability': 0.0.2 + lucide-react: 0.575.0(react@19.2.4) + react: 19.2.4 + react-dom: 19.2.4(react@19.2.4) + '@openuidev/lang-core@0.2.10(@modelcontextprotocol/sdk@1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76))(zod@3.25.76)': dependencies: zod: 3.25.76 optionalDependencies: '@modelcontextprotocol/sdk': 1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76) + '@openuidev/lang-core@0.2.11(@modelcontextprotocol/sdk@1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76))(zod@3.25.76)': + dependencies: + zod: 3.25.76 + optionalDependencies: + '@modelcontextprotocol/sdk': 1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76) + + '@openuidev/observability@0.0.2': {} + + '@openuidev/react-headless@0.9.6(ai@6.0.175(zod@3.25.76))(react@19.2.4)(zustand@4.5.7(@types/react@19.2.14)(immer@11.1.4)(react@19.2.4))': + dependencies: + '@ag-ui/core': 0.0.53 + '@openuidev/observability': 0.0.2 + react: 19.2.4 + zustand: 4.5.7(@types/react@19.2.14)(immer@11.1.4)(react@19.2.4) + optionalDependencies: + ai: 6.0.175(zod@3.25.76) + + '@openuidev/react-lang@0.2.11(@modelcontextprotocol/sdk@1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76))(react-dom@19.2.4(react@19.2.4))(react@19.2.4)(zod@3.25.76)': + dependencies: + '@openuidev/devtools': 0.0.4(@openuidev/observability@0.0.2)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@openuidev/lang-core': 0.2.11(@modelcontextprotocol/sdk@1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76))(zod@3.25.76) + '@openuidev/observability': 0.0.2 + react: 19.2.4 + zod: 3.25.76 + optionalDependencies: + '@modelcontextprotocol/sdk': 1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76) + react-dom: 19.2.4(react@19.2.4) + + '@openuidev/react-ui@0.13.5(@openuidev/react-headless@0.9.6(ai@6.0.175(zod@3.25.76))(react@19.2.4)(zustand@4.5.7(@types/react@19.2.14)(immer@11.1.4)(react@19.2.4)))(@openuidev/react-lang@0.2.11(@modelcontextprotocol/sdk@1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76))(react-dom@19.2.4(react@19.2.4))(react@19.2.4)(zod@3.25.76))(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4)(zod@3.25.76)(zustand@4.5.7(@types/react@19.2.14)(immer@11.1.4)(react@19.2.4))': + dependencies: + '@floating-ui/react-dom': 2.1.9(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@openuidev/react-headless': 0.9.6(ai@6.0.175(zod@3.25.76))(react@19.2.4)(zustand@4.5.7(@types/react@19.2.14)(immer@11.1.4)(react@19.2.4)) + '@openuidev/react-lang': 0.2.11(@modelcontextprotocol/sdk@1.29.0(@cfworker/json-schema@4.1.1)(zod@3.25.76))(react-dom@19.2.4(react@19.2.4))(react@19.2.4)(zod@3.25.76) + '@radix-ui/react-accordion': 1.2.20(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-aspect-ratio': 1.1.15(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-checkbox': 1.3.11(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-dialog': 1.1.23(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-dropdown-menu': 2.1.24(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-label': 2.1.15(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-radio-group': 1.4.7(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-select': 2.3.7(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-separator': 1.1.15(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-slider': 1.4.7(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-slot': 1.3.3(@types/react@19.2.14)(react@19.2.4) + '@radix-ui/react-switch': 1.3.7(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-tabs': 1.1.21(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-toggle-group': 1.1.19(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + '@radix-ui/react-tooltip': 1.2.16(@types/react-dom@19.2.3(@types/react@19.2.14))(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + clsx: 2.1.1 + date-fns: 4.4.0 + lodash-es: 4.18.1 + lucide-react: 0.562.0(react@19.2.4) + react: 19.2.4 + react-day-picker: 9.14.0(react@19.2.4) + react-dom: 19.2.4(react@19.2.4) + react-markdown: 10.1.0(@types/react@19.2.14)(react@19.2.4) + react-syntax-highlighter: 16.1.1(react@19.2.4) + recharts: 2.15.4(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + rehype-katex: 7.0.1 + remark-breaks: 4.0.0 + remark-emoji: 5.0.2 + remark-gfm: 4.0.1 + remark-math: 6.0.0 + tiny-invariant: 1.3.3 + zod: 3.25.76 + zustand: 4.5.7(@types/react@19.2.14)(immer@11.1.4)(react@19.2.4) + transitivePeerDependencies: + - '@types/react' + - '@types/react-dom' + - supports-color + '@pkgjs/parseargs@0.11.0': optional: true @@ -14981,6 +15451,8 @@ snapshots: '@sinclair/typebox@0.27.10': {} + '@sindresorhus/is@4.6.0': {} + '@sindresorhus/merge-streams@2.3.0': {} '@sindresorhus/merge-streams@4.0.0': {} @@ -15479,6 +15951,8 @@ snapshots: '@swc/counter': 0.1.3 optional: true + '@tabby_ai/hijri-converter@1.0.5': {} + '@tailwindcss/node@4.2.1': dependencies: '@jridgewell/remapping': 2.3.5 @@ -15632,12 +16106,20 @@ snapshots: '@types/deep-eql@4.0.2': {} + '@types/estree-jsx@1.0.5': + dependencies: + '@types/estree': 1.0.8 + '@types/estree@1.0.8': {} '@types/graceful-fs@4.1.9': dependencies: '@types/node': 22.19.11 + '@types/hast@3.0.5': + dependencies: + '@types/unist': 3.0.3 + '@types/istanbul-lib-coverage@2.0.6': {} '@types/istanbul-lib-report@3.0.3': @@ -15650,6 +16132,8 @@ snapshots: '@types/json-schema@7.0.15': {} + '@types/katex@0.16.8': {} + '@types/mdast@4.0.4': dependencies: '@types/unist': 3.0.3 @@ -15683,6 +16167,8 @@ snapshots: '@types/pegjs@0.10.6': optional: true + '@types/prismjs@1.26.6': {} + '@types/react-dom@19.2.3(@types/react@19.1.17)': dependencies: '@types/react': 19.1.17 @@ -15720,6 +16206,8 @@ snapshots: '@types/triple-beam@1.3.5': {} + '@types/unist@2.0.11': {} + '@types/unist@3.0.3': {} '@types/use-sync-external-store@0.0.6': {} @@ -15973,6 +16461,15 @@ snapshots: dependencies: humanize-ms: 1.2.1 + ai@6.0.175(zod@3.25.76): + dependencies: + '@ai-sdk/gateway': 3.0.110(zod@3.25.76) + '@ai-sdk/provider': 3.0.10 + '@ai-sdk/provider-utils': 4.0.26(zod@3.25.76) + '@opentelemetry/api': 1.9.0 + zod: 3.25.76 + optional: true + ai@6.0.175(zod@4.4.3): dependencies: '@ai-sdk/gateway': 3.0.110(zod@4.4.3) @@ -16444,8 +16941,16 @@ snapshots: chalk@5.6.2: {} + char-regex@1.0.2: {} + + character-entities-html4@2.1.0: {} + + character-entities-legacy@3.0.0: {} + character-entities@2.0.2: {} + character-reference-invalid@2.0.1: {} + chardet@2.1.1: {} charenc@0.0.2: {} @@ -16577,6 +17082,8 @@ snapshots: dependencies: delayed-stream: 1.0.0 + comma-separated-tokens@2.0.3: {} + commander@10.0.1: {} commander@12.1.0: {} @@ -16591,6 +17098,8 @@ snapshots: commander@7.2.0: {} + commander@8.3.0: {} + commondir@1.0.1: {} compare-versions@6.1.1: {} @@ -16763,6 +17272,10 @@ snapshots: - '@noble/hashes' optional: true + date-fns-jalali@4.1.0-0: {} + + date-fns@4.4.0: {} + de-indent@1.0.2: {} debounce@3.0.0: {} @@ -16871,6 +17384,11 @@ snapshots: dependencies: path-type: 4.0.0 + dom-helpers@5.2.1: + dependencies: + '@babel/runtime': 7.28.6 + csstype: 3.2.3 + dotenv-expand@11.0.7: dependencies: dotenv: 16.4.5 @@ -16935,6 +17453,10 @@ snapshots: emoji-regex@9.2.2: {} + emojilib@2.4.0: {} + + emoticon@4.1.0: {} + enabled@2.0.0: {} encodeurl@1.0.2: {} @@ -17109,6 +17631,8 @@ snapshots: estraverse@5.3.0: {} + estree-util-is-identifier-name@3.0.0: {} + estree-walker@2.0.2: {} estree-walker@3.0.3: @@ -17121,8 +17645,7 @@ snapshots: event-target-shim@5.0.1: {} - eventemitter3@4.0.7: - optional: true + eventemitter3@4.0.7: {} eventemitter3@5.0.4: {} @@ -17314,6 +17837,8 @@ snapshots: fast-deep-equal@3.1.3: {} + fast-equals@5.4.1: {} + fast-glob@3.3.3: dependencies: '@nodelib/fs.stat': 2.0.5 @@ -17359,6 +17884,10 @@ snapshots: dependencies: reusify: 1.1.0 + fault@1.0.4: + dependencies: + format: 0.2.2 + fb-watchman@2.0.2: dependencies: bser: 2.1.1 @@ -17502,6 +18031,8 @@ snapshots: hasown: 2.0.2 mime-types: 2.1.35 + format@0.2.2: {} + formdata-node@4.4.1: dependencies: node-domexception: 1.0.0 @@ -17762,6 +18293,86 @@ snapshots: dependencies: function-bind: 1.1.2 + hast-util-from-dom@5.0.1: + dependencies: + '@types/hast': 3.0.5 + hastscript: 9.0.1 + web-namespaces: 2.0.1 + + hast-util-from-html-isomorphic@2.0.0: + dependencies: + '@types/hast': 3.0.5 + hast-util-from-dom: 5.0.1 + hast-util-from-html: 2.0.3 + unist-util-remove-position: 5.0.0 + + hast-util-from-html@2.0.3: + dependencies: + '@types/hast': 3.0.5 + devlop: 1.1.0 + hast-util-from-parse5: 8.0.3 + parse5: 7.3.0 + vfile: 6.0.3 + vfile-message: 4.0.3 + + hast-util-from-parse5@8.0.3: + dependencies: + '@types/hast': 3.0.5 + '@types/unist': 3.0.3 + devlop: 1.1.0 + hastscript: 9.0.1 + property-information: 7.2.0 + vfile: 6.0.3 + vfile-location: 5.0.3 + web-namespaces: 2.0.1 + + hast-util-is-element@3.0.0: + dependencies: + '@types/hast': 3.0.5 + + hast-util-parse-selector@4.0.0: + dependencies: + '@types/hast': 3.0.5 + + hast-util-to-jsx-runtime@2.3.6: + dependencies: + '@types/estree': 1.0.8 + '@types/hast': 3.0.5 + '@types/unist': 3.0.3 + comma-separated-tokens: 2.0.3 + devlop: 1.1.0 + estree-util-is-identifier-name: 3.0.0 + hast-util-whitespace: 3.0.0 + mdast-util-mdx-expression: 2.0.1 + mdast-util-mdx-jsx: 3.2.0 + mdast-util-mdxjs-esm: 2.0.1 + property-information: 7.2.0 + space-separated-tokens: 2.0.2 + style-to-js: 1.1.21 + unist-util-position: 5.0.0 + vfile-message: 4.0.3 + transitivePeerDependencies: + - supports-color + + hast-util-to-text@4.0.2: + dependencies: + '@types/hast': 3.0.5 + '@types/unist': 3.0.3 + hast-util-is-element: 3.0.0 + unist-util-find-after: 5.0.0 + + hast-util-whitespace@3.0.0: + dependencies: + '@types/hast': 3.0.5 + + hastscript@9.0.1: + dependencies: + '@types/hast': 3.0.5 + comma-separated-tokens: 2.0.3 + hast-util-parse-selector: 4.0.0 + property-information: 7.2.0 + space-separated-tokens: 2.0.2 + he@1.2.0: {} hermes-estree@0.23.1: {} @@ -17802,8 +18413,12 @@ snapshots: hermes-estree: 0.36.0 optional: true + highlight.js@10.7.3: {} + highlight.js@11.11.1: {} + highlightjs-vue@1.0.0: {} + hono@4.12.12: {} hookified@1.15.1: {} @@ -17819,6 +18434,8 @@ snapshots: - '@noble/hashes' optional: true + html-url-attributes@3.0.1: {} + http-errors@2.0.1: dependencies: depd: 2.0.0 @@ -17966,6 +18583,8 @@ snapshots: - bufferutil - utf-8-validate + inline-style-parser@0.2.7: {} + inline-style-prefixer@6.0.4: dependencies: css-in-js-utils: 3.1.0 @@ -17981,6 +18600,13 @@ snapshots: ipaddr.js@1.9.1: {} + is-alphabetical@2.0.1: {} + + is-alphanumerical@2.0.1: + dependencies: + is-alphabetical: 2.0.1 + is-decimal: 2.0.1 + is-arrayish@0.2.1: {} is-buffer@1.1.6: {} @@ -17989,6 +18615,8 @@ snapshots: dependencies: hasown: 2.0.4 + is-decimal@2.0.1: {} + is-directory@0.3.1: {} is-docker@2.2.1: {} @@ -18010,6 +18638,8 @@ snapshots: dependencies: is-extglob: 2.1.1 + is-hexadecimal@2.0.1: {} + is-in-ci@2.0.0: {} is-inside-container@1.0.0: @@ -18321,6 +18951,10 @@ snapshots: kareem@3.3.0: optional: true + katex@0.16.47: + dependencies: + commander: 8.3.0 + keyv-file@5.3.3: dependencies: '@keyv/serialize': 1.1.1 @@ -18434,6 +19068,8 @@ snapshots: dependencies: p-locate: 4.1.0 + lodash-es@4.18.1: {} + lodash.debounce@4.0.8: {} lodash.includes@4.3.0: @@ -18461,6 +19097,8 @@ snapshots: lodash.throttle@4.1.1: {} + lodash@4.18.1: {} + log-symbols@2.2.0: dependencies: chalk: 2.4.2 @@ -18489,6 +19127,11 @@ snapshots: loupe@3.2.1: {} + lowlight@1.20.0: + dependencies: + fault: 1.0.4 + highlight.js: 10.7.3 + lru-cache@10.4.3: {} lru-cache@11.3.6: {} @@ -18499,10 +19142,18 @@ snapshots: lru-cache@7.18.3: {} + lucide-react@0.562.0(react@19.2.4): + dependencies: + react: 19.2.4 + lucide-react@0.564.0(react@19.2.4): dependencies: react: 19.2.4 + lucide-react@0.575.0(react@19.2.4): + dependencies: + react: 19.2.4 + magic-string@0.30.21: dependencies: '@jridgewell/sourcemap-codec': 1.5.5 @@ -18629,11 +19280,79 @@ snapshots: transitivePeerDependencies: - supports-color + mdast-util-math@3.0.0: + dependencies: + '@types/hast': 3.0.5 + '@types/mdast': 4.0.4 + devlop: 1.1.0 + longest-streak: 3.1.0 + mdast-util-from-markdown: 2.0.2 + mdast-util-to-markdown: 2.1.2 + unist-util-remove-position: 5.0.0 + transitivePeerDependencies: + - supports-color + + mdast-util-mdx-expression@2.0.1: + dependencies: + '@types/estree-jsx': 1.0.5 + '@types/hast': 3.0.5 + '@types/mdast': 4.0.4 + devlop: 1.1.0 + mdast-util-from-markdown: 2.0.2 + mdast-util-to-markdown: 2.1.2 + transitivePeerDependencies: + - supports-color + + mdast-util-mdx-jsx@3.2.0: + dependencies: + '@types/estree-jsx': 1.0.5 + '@types/hast': 3.0.5 + '@types/mdast': 4.0.4 + '@types/unist': 3.0.3 + ccount: 2.0.1 + devlop: 1.1.0 + mdast-util-from-markdown: 2.0.2 + mdast-util-to-markdown: 2.1.2 + parse-entities: 4.0.2 + stringify-entities: 4.0.4 + unist-util-stringify-position: 4.0.0 + vfile-message: 4.0.3 + transitivePeerDependencies: + - supports-color + + mdast-util-mdxjs-esm@2.0.1: + dependencies: + '@types/estree-jsx': 1.0.5 + '@types/hast': 3.0.5 + '@types/mdast': 4.0.4 + devlop: 1.1.0 + mdast-util-from-markdown: 2.0.2 + mdast-util-to-markdown: 2.1.2 + transitivePeerDependencies: + - supports-color + + mdast-util-newline-to-break@2.0.0: + dependencies: + '@types/mdast': 4.0.4 + mdast-util-find-and-replace: 3.0.2 + mdast-util-phrasing@4.1.0: dependencies: '@types/mdast': 4.0.4 unist-util-is: 6.0.1 + mdast-util-to-hast@13.2.1: + dependencies: + '@types/hast': 3.0.5 + '@types/mdast': 4.0.4 + '@ungap/structured-clone': 1.3.2 + devlop: 1.1.0 + micromark-util-sanitize-uri: 2.0.1 + trim-lines: 3.0.1 + unist-util-position: 5.0.0 + unist-util-visit: 5.1.0 + vfile: 6.0.3 + mdast-util-to-markdown@2.1.2: dependencies: '@types/mdast': 4.0.4 @@ -19459,6 +20178,16 @@ snapshots: micromark-util-combine-extensions: 2.0.1 micromark-util-types: 2.0.2 + micromark-extension-math@3.1.0: + dependencies: + '@types/katex': 0.16.8 + devlop: 1.1.0 + katex: 0.16.47 + micromark-factory-space: 2.0.1 + micromark-util-character: 2.1.1 + micromark-util-symbol: 2.0.1 + micromark-util-types: 2.0.2 + micromark-factory-destination@2.0.1: dependencies: micromark-util-character: 2.1.1 @@ -19749,6 +20478,13 @@ snapshots: node-domexception@1.0.0: {} + node-emoji@2.2.0: + dependencies: + '@sindresorhus/is': 4.6.0 + char-regex: 1.0.2 + emojilib: 2.4.0 + skin-tone: 2.0.0 + node-fetch@2.7.0: dependencies: whatwg-url: 5.0.0 @@ -20039,6 +20775,16 @@ snapshots: dependencies: quansync: 0.2.11 + parse-entities@4.0.2: + dependencies: + '@types/unist': 2.0.11 + character-entities-legacy: 3.0.0 + character-reference-invalid: 2.0.1 + decode-named-character-reference: 1.3.0 + is-alphanumerical: 2.0.1 + is-decimal: 2.0.1 + is-hexadecimal: 2.0.1 + parse-json@4.0.0: dependencies: error-ex: 1.3.4 @@ -20050,6 +20796,10 @@ snapshots: dependencies: pngjs: 3.4.0 + parse5@7.3.0: + dependencies: + entities: 6.0.1 + parse5@8.0.0: dependencies: entities: 6.0.1 @@ -20270,6 +21020,8 @@ snapshots: dependencies: parse-ms: 4.0.0 + prismjs@1.30.0: {} + proc-log@4.2.0: {} process-nextick-args@2.0.1: @@ -20468,6 +21220,14 @@ snapshots: kleur: 3.0.3 sisteransi: 1.0.5 + prop-types@15.8.1: + dependencies: + loose-envify: 1.4.0 + object-assign: 4.1.1 + react-is: 16.13.1 + + property-information@7.2.0: {} + proto-list@1.2.4: {} protobufjs@7.5.6: @@ -20623,6 +21383,14 @@ snapshots: minimist: 1.2.8 strip-json-comments: 2.0.1 + react-day-picker@9.14.0(react@19.2.4): + dependencies: + '@date-fns/tz': 1.5.0 + '@tabby_ai/hijri-converter': 1.0.5 + date-fns: 4.4.0 + date-fns-jalali: 4.1.0-0 + react: 19.2.4 + react-devtools-core@5.3.2: dependencies: shell-quote: 1.9.0 @@ -20644,10 +21412,30 @@ snapshots: react: 19.2.4 scheduler: 0.27.0 + react-is@16.13.1: {} + react-is@18.3.1: {} react-is@19.2.4: {} + react-markdown@10.1.0(@types/react@19.2.14)(react@19.2.4): + dependencies: + '@types/hast': 3.0.5 + '@types/mdast': 4.0.4 + '@types/react': 19.2.14 + devlop: 1.1.0 + hast-util-to-jsx-runtime: 2.3.6 + html-url-attributes: 3.0.1 + mdast-util-to-hast: 13.2.1 + react: 19.2.4 + remark-parse: 11.0.0 + remark-rehype: 11.1.2 + unified: 11.0.5 + unist-util-visit: 5.1.0 + vfile: 6.0.3 + transitivePeerDependencies: + - supports-color + react-native-is-edge-to-edge@1.3.1(react-native@0.81.5(@babel/core@7.29.0)(@react-native/metro-config@0.86.0(@babel/core@7.29.0))(@types/react@19.1.17)(react@19.1.0))(react@19.1.0): dependencies: react: 19.1.0 @@ -20804,6 +21592,14 @@ snapshots: optionalDependencies: '@types/react': 19.2.14 + react-smooth@4.0.4(react-dom@19.2.4(react@19.2.4))(react@19.2.4): + dependencies: + fast-equals: 5.4.1 + prop-types: 15.8.1 + react: 19.2.4 + react-dom: 19.2.4(react@19.2.4) + react-transition-group: 4.4.5(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + react-style-singleton@2.2.3(@types/react@19.2.14)(react@19.2.4): dependencies: get-nonce: 1.0.1 @@ -20812,6 +21608,25 @@ snapshots: optionalDependencies: '@types/react': 19.2.14 + react-syntax-highlighter@16.1.1(react@19.2.4): + dependencies: + '@babel/runtime': 7.28.6 + highlight.js: 10.7.3 + highlightjs-vue: 1.0.0 + lowlight: 1.20.0 + prismjs: 1.30.0 + react: 19.2.4 + refractor: 5.0.0 + + react-transition-group@4.4.5(react-dom@19.2.4(react@19.2.4))(react@19.2.4): + dependencies: + '@babel/runtime': 7.28.6 + dom-helpers: 5.2.1 + loose-envify: 1.4.0 + prop-types: 15.8.1 + react: 19.2.4 + react-dom: 19.2.4(react@19.2.4) + react@19.1.0: {} react@19.2.4: {} @@ -20858,6 +21673,23 @@ snapshots: source-map: 0.6.1 tslib: 2.8.1 + recharts-scale@0.4.5: + dependencies: + decimal.js-light: 2.5.1 + + recharts@2.15.4(react-dom@19.2.4(react@19.2.4))(react@19.2.4): + dependencies: + clsx: 2.1.1 + eventemitter3: 4.0.7 + lodash: 4.18.1 + react: 19.2.4 + react-dom: 19.2.4(react@19.2.4) + react-is: 18.3.1 + react-smooth: 4.0.4(react-dom@19.2.4(react@19.2.4))(react@19.2.4) + recharts-scale: 0.4.5 + tiny-invariant: 1.3.3 + victory-vendor: 36.9.2 + recharts@3.7.0(@types/react@19.2.14)(react-dom@19.2.4(react@19.2.4))(react-is@19.2.4)(react@19.2.4)(redux@5.0.1): dependencies: '@reduxjs/toolkit': 2.11.2(react-redux@9.2.0(@types/react@19.2.14)(react@19.2.4)(redux@5.0.1))(react@19.2.4) @@ -20896,6 +21728,13 @@ snapshots: redux@5.0.1: {} + refractor@5.0.0: + dependencies: + '@types/hast': 3.0.5 + '@types/prismjs': 1.26.6 + hastscript: 9.0.1 + parse-entities: 4.0.2 + regenerate-unicode-properties@10.2.2: dependencies: regenerate: 1.4.2 @@ -20919,6 +21758,30 @@ snapshots: dependencies: jsesc: 3.1.0 + rehype-katex@7.0.1: + dependencies: + '@types/hast': 3.0.5 + '@types/katex': 0.16.8 + hast-util-from-html-isomorphic: 2.0.0 + hast-util-to-text: 4.0.2 + katex: 0.16.47 + unist-util-visit-parents: 6.0.2 + vfile: 6.0.3 + + remark-breaks@4.0.0: + dependencies: + '@types/mdast': 4.0.4 + mdast-util-newline-to-break: 2.0.0 + unified: 11.0.5 + + remark-emoji@5.0.2: + dependencies: + '@types/mdast': 4.0.4 + emoticon: 4.1.0 + mdast-util-find-and-replace: 3.0.2 + node-emoji: 2.2.0 + unified: 11.0.5 + remark-gfm@4.0.1: dependencies: '@types/mdast': 4.0.4 @@ -20930,6 +21793,15 @@ snapshots: transitivePeerDependencies: - supports-color + remark-math@6.0.0: + dependencies: + '@types/mdast': 4.0.4 + mdast-util-math: 3.0.0 + micromark-extension-math: 3.1.0 + unified: 11.0.5 + transitivePeerDependencies: + - supports-color + remark-parse@11.0.0: dependencies: '@types/mdast': 4.0.4 @@ -20939,6 +21811,14 @@ snapshots: transitivePeerDependencies: - supports-color + remark-rehype@11.1.2: + dependencies: + '@types/hast': 3.0.5 + '@types/mdast': 4.0.4 + mdast-util-to-hast: 13.2.1 + unified: 11.0.5 + vfile: 6.0.3 + remark-stringify@11.0.0: dependencies: '@types/mdast': 4.0.4 @@ -21310,6 +22190,10 @@ snapshots: sisteransi@1.0.5: {} + skin-tone@2.0.0: + dependencies: + unicode-emoji-modifier-base: 1.0.0 + slash@3.0.0: {} slash@5.1.0: {} @@ -21388,6 +22272,8 @@ snapshots: source-map@0.6.1: {} + space-separated-tokens@2.0.2: {} + sparse-bitfield@3.0.3: dependencies: memory-pager: 1.5.0 @@ -21465,6 +22351,11 @@ snapshots: dependencies: safe-buffer: 5.2.1 + stringify-entities@4.0.4: + dependencies: + character-entities-html4: 2.1.0 + character-entities-legacy: 3.0.0 + strip-ansi@5.2.0: dependencies: ansi-regex: 4.1.1 @@ -21498,6 +22389,14 @@ snapshots: structured-headers@0.4.1: {} + style-to-js@1.1.21: + dependencies: + style-to-object: 1.0.14 + + style-to-object@1.0.14: + dependencies: + inline-style-parser: 0.2.7 + styleq@0.1.3: {} sucrase@3.35.1: @@ -21679,6 +22578,8 @@ snapshots: punycode: 2.3.1 optional: true + trim-lines@3.0.1: {} + triple-beam@1.4.1: {} trough@2.2.0: {} @@ -21794,6 +22695,8 @@ snapshots: unicode-canonical-property-names-ecmascript@2.0.1: {} + unicode-emoji-modifier-base@1.0.0: {} + unicode-match-property-ecmascript@2.0.0: dependencies: unicode-canonical-property-names-ecmascript: 2.0.1 @@ -21815,10 +22718,24 @@ snapshots: trough: 2.2.0 vfile: 6.0.3 + unist-util-find-after@5.0.0: + dependencies: + '@types/unist': 3.0.3 + unist-util-is: 6.0.1 + unist-util-is@6.0.1: dependencies: '@types/unist': 3.0.3 + unist-util-position@5.0.0: + dependencies: + '@types/unist': 3.0.3 + + unist-util-remove-position@5.0.0: + dependencies: + '@types/unist': 3.0.3 + unist-util-visit: 5.1.0 + unist-util-stringify-position@4.0.0: dependencies: '@types/unist': 3.0.3 @@ -21923,6 +22840,11 @@ snapshots: version-range@4.15.0: {} + vfile-location@5.0.3: + dependencies: + '@types/unist': 3.0.3 + vfile: 6.0.3 + vfile-message@4.0.3: dependencies: '@types/unist': 3.0.3 @@ -21933,6 +22855,23 @@ snapshots: '@types/unist': 3.0.3 vfile-message: 4.0.3 + victory-vendor@36.9.2: + dependencies: + '@types/d3-array': 3.2.2 + '@types/d3-ease': 3.0.2 + '@types/d3-interpolate': 3.0.4 + '@types/d3-scale': 4.0.9 + '@types/d3-shape': 3.1.8 + '@types/d3-time': 3.0.4 + '@types/d3-timer': 3.0.2 + d3-array: 3.2.4 + d3-ease: 3.0.1 + d3-interpolate: 3.0.1 + d3-scale: 4.0.2 + d3-shape: 3.2.0 + d3-time: 3.1.0 + d3-timer: 3.0.1 + victory-vendor@37.3.6: dependencies: '@types/d3-array': 3.2.2 @@ -22067,6 +23006,8 @@ snapshots: dependencies: defaults: 1.0.4 + web-namespaces@2.0.1: {} + web-streams-polyfill@3.3.3: {} web-streams-polyfill@4.0.0-beta.3: {} @@ -22265,4 +23206,12 @@ snapshots: zod@4.4.3: {} + zustand@4.5.7(@types/react@19.2.14)(immer@11.1.4)(react@19.2.4): + dependencies: + use-sync-external-store: 1.6.0(react@19.2.4) + optionalDependencies: + '@types/react': 19.2.14 + immer: 11.1.4 + react: 19.2.4 + zwitch@2.0.4: {}