diff --git a/.gitignore b/.gitignore index 8ba15f7..b4aac33 100644 --- a/.gitignore +++ b/.gitignore @@ -9,3 +9,6 @@ build/ .DS_Store .pytest_cache/ PUBLISHING.md + +# Benchmark receipts from local runs (the committed examples under samples/ are kept) +/bench-receipts/ diff --git a/CLAUDE.md b/CLAUDE.md index 9c53555..ece838b 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -65,8 +65,21 @@ src/av/ │ ├── ffmpeg.py # ffmpeg/ffprobe wrappers │ ├── chunker.py # Text chunking for embeddings │ └── dense_caption.py # Structured event export +├── bench/ # Cost/accuracy frontier measurement +│ ├── cost.py # Per-hour vs per-token cost models (never conflated) +│ ├── datasets.py # Adapters for public benchmarks (no data vendored) +│ ├── fixtures.py # Deterministic ffmpeg fixtures for the ordering gate +│ ├── frames.py # Pinned frame extraction (command recorded in the receipt) +│ ├── receipts.py # Labelled evidence + endpoint redaction +│ ├── runner.py # Sweep axes, noise floor, collapse point +│ ├── vlm.py # Provider-agnostic multi-image calls with usage accounting +│ └── tasks/ +│ ├── ordering.py # Temporal-ordering capability gate +│ ├── videoqa.py # Dense vs agentic arms +│ └── events.py # Event recall vs sampling interval ├── providers/ │ ├── base.py # Abstract interfaces +│ ├── deepseek.py # DeepSeek-V4.1-Flash (SGLang) config + image-token model │ └── openai.py # OpenAI-compatible client (works for all providers) ├── search/ │ ├── semantic.py # FTS5 + cosine reranking @@ -117,6 +130,38 @@ On partial failure: `{"status": "complete_with_warnings", ..., "warnings": ["Tra ### `av list` / `av info ` / `av transcript ` / `av export` / `av open ` See `av --help` for details. +### `av bench` +Measures the cost/accuracy frontier. Headline axes are **tokens per query** and +**accuracy**, plus **dollars per query on hardware you own** — the axis a per-token +API vendor cannot report. + +```bash +av bench probe # is tokens-per-frame tunable on this endpoint? +av bench gate --sizes 2,4,8 # can the model order frames at all? +av bench plan --budgets 200,400,800 # predicted token cost per resolution (offline) +av bench prepare minerva ann.json -o task.jsonl +av bench run task.jsonl --arms dense,agentic --cost hourly:25.0:20000 +av bench sweep captions.jsonl videos/ --intervals 1,2,5,10,30 +av bench noise --repeats 5 +av bench cost --tokens-per-frame 1024 --prefill-tok-s 20000 --hourly-usd 25 +``` + +Every subcommand writes a JSON receipt to `./bench-receipts/`. + +**Rules that are not optional here:** +- **Run the gate before quoting any score.** A model that cannot order eight frames + is not being measured on temporal understanding, and its throughput is irrelevant. +- **Label every claim.** `measured` / `derived` / `documented` / `community-reported` + / `untested`. Non-measured claims must cite a source; `Claim` raises otherwise. +- **Publish the noise floor**, and never on a saturated cell. +- **Never conflate `$/hr` and `$/token`.** They are different economics. +- **Never put someone else's published number and ours in one cell as a ratio.** + Different model, different hardware, different methodology — it is a comparison of + approaches, not a head-to-head. +- **No endpoints in source.** Receipts record a hostname; private hosts are redacted. +- **No benchmark data is vendored.** Adapters read files the user fetched, under the + upstream licence. + ## Provider Support | Provider | Transcription | Vision/Chat | Embeddings | Setup | @@ -125,6 +170,7 @@ See `av --help` for details. | OpenAI (API key) | whisper-1 | gpt-4-1 | text-embedding-3-small | Paste `sk-...` | | Anthropic | -- | claude-sonnet-4-5 | -- | Paste API key | | Gemini | -- | gemini-2.5-flash | text-embedding-004 | Paste API key | +| DeepSeek-V4.1-Flash | -- | deepseek-v4.1-flash (self-hosted SGLang) | -- | `AV_API_BASE_URL` + `DEEPSEEK_API_KEY` | When a capability is unavailable (e.g. Anthropic has no Whisper), the pipeline skips that stage and warns. @@ -140,6 +186,8 @@ When a capability is unavailable (e.g. Anthropic has no Whisper), the pipeline s | `AV_EMBED_MODEL` | `text-embedding-3-small` | Embedding model | | `AV_CHAT_MODEL` | `gpt-4-1` | Chat/RAG model | | `AV_DB_PATH` | `~/.config/av/av.db` | Database location | +| `DEEPSEEK_API_KEY` | (none) | Key for a self-hosted DeepSeek-V4.1-Flash server | +| `SGLANG_API_KEY` | (none) | Alias for the same, matching SGLang's own naming | ## Database @@ -182,5 +230,7 @@ Near-term priorities for contributors: - [ ] Cross-video search improvements (search across all indexed videos at once) - [ ] Streaming ingest progress (SSE-style output for long videos) - [ ] Profile presets for dense captioning (security, retail, meeting, etc.) +- [ ] `av bench` precision measurement (currently only recall against reference windows) +- [ ] Video fetch helper for benchmark task files (yt-dlp, with link-rot reporting) - [x] CI/CD with GitHub Actions (lint + test on PR) — `.github/workflows/ci.yml` - [x] PyPI publish workflow — `.github/workflows/publish.yml` + `PUBLISHING.md` diff --git a/README.md b/README.md index d6ea3f5..eb31bf9 100644 --- a/README.md +++ b/README.md @@ -65,6 +65,12 @@ av sentinel video.mp4 # Detect events (all 4 alert types) av sentinel video.mp4 --alerts FALL # Fall detection only av sentinel video.mp4 -p ollama # Self-hosted (free) av sentinel videos/ -c cam_lobby # Batch with camera tracking + +# Benchmarking +av bench probe # What can this deployment actually do? +av bench gate # Can it order frames at all? Run this first. +av bench run task.jsonl # Dense vs agentic, with tokens and dollars +av bench sweep captions.jsonl vids/ # Where does recall collapse as frames thin out? ``` ## Sentinel — Surveillance Event Detection @@ -102,6 +108,105 @@ Video → 30s chunks (5s overlap) Built on 107 experiments across 21 vision models. Key insight: structural extraction + temporal rules beats generic "detect anomalies" prompts. +## Bench — Cost/Accuracy Frontier + +Selling video understanding on hardware you own means one number decides everything: +**video-hours analysed per dollar**. `av bench` measures it, and measures what it +costs you in accuracy to get there. + +Two headline axes, chosen so results read against published agentic-video +comparisons: **tokens per query** and **accuracy**. Alongside them sits the axis an +API vendor cannot report — **dollars per query on your own box** — because per-token +billing and per-hour hardware are different economics and the tool never conflates +them. + +### Run the gate first + +```bash +av bench gate --sizes 2,4,8 +``` + +Deterministic ffmpeg fixtures carrying a known order, one question, exact-match +scoring. A model that cannot report the order of eight flat colours cannot be +meaningfully scored on long-video reasoning, and any throughput number measured +against it describes a machine doing the wrong thing quickly. The gate costs cents +and it can save the whole exercise. + +### Establish what is tunable before sweeping it + +```bash +av bench probe +av bench plan --widths 512,768,1024,1536 --budgets 200,400,800 +``` + +`probe` tests two candidate knobs against your live endpoint — the OpenAI `detail` +hint and the resolution actually uploaded — because a server may honour one and +silently ignore the other. If neither moves the per-frame token count, the +tokens-per-frame axis is reported as fixed rather than faked. `plan` predicts the +same thing offline from a published preprocessor algorithm, and shows the two walls +worth knowing: an upscale floor below which shrinking frames buys nothing, and a +token ceiling above which extra resolution is discarded. + +### Dense versus agentic + +```bash +av bench prepare minerva minerva.json --out task.jsonl --max-questions 40 --max-videos 6 +av bench run task.jsonl --arms dense,agentic --cost hourly:25.0:20000 +``` + +The **dense** arm samples the whole window at a fixed rate and asks once. The +**agentic** arm takes a cheap coarse look, decides which moments it needs, then +fetches only those — and is charged for both requests. Nothing else differs between +them. + +`av bench prepare` adapts a public benchmark's annotations into the task format. +**No benchmark data ships with av and no videos are downloaded.** Fetch annotations +yourself and mind their licences: MINERVA's are CC BY 4.0, LVBench's are +CC BY-NC-SA with an explicit commercial-use prohibition, and neither grants any +rights to the videos themselves. + +### Where does it collapse? + +```bash +av bench sweep captions.jsonl videos/ --intervals 1,2,5,10,30 --cost token:0.30:2.50 +``` + +Event detection against sampling interval on real footage. The interval at which +detection collapses is the cheapest safe sampling rate — and it is a per-task +answer, not a global one. Smoke tolerates sparse frames; a door opening does not. + +### Noise floor + +```bash +av bench noise --repeats 5 +``` + +Runs one unchanged cell repeatedly and publishes the spread. This is the number that +makes every other number readable: a delta smaller than the spread is noise. Point it +at a cell the model does not already solve perfectly — a saturated cell has no +headroom to vary, and the tool says so rather than reporting a meaningless zero. + +### Receipts + +Every subcommand writes a JSON receipt to `./bench-receipts/` carrying the provider, +the determinism controls, the exact ffmpeg invocations, fixture hashes, the cost +model, and every cell. Claims are labelled `measured`, `derived`, `documented`, +`community-reported`, or `untested`, and a non-measured claim must cite a source. +Endpoints are reduced to a hostname, and private or tunnelled hosts never appear at +all — receipts are meant to be published. + +### Cost model + +```bash +av bench cost --tokens-per-frame 1024 --context-tokens 1048576 \ + --prefill-tok-s 20000 --hourly-usd 25 --kv-bytes-per-token 890 \ + --source "your measurements" +``` + +Pure arithmetic, no API calls, every input recorded. Supply `--cost hourly:RATE` for +hardware you own or `--cost token:IN:OUT` for a vendor API — they are different +shapes and reporting one in the other's units produces a number that means nothing. + ## Configuration ### Interactive Setup (Recommended) @@ -110,14 +215,21 @@ Built on 107 experiments across 21 vision models. Key insight: structural extrac av config setup ``` -Choose from four providers: +Choose from six providers: | # | Provider | Auth | Transcription | Embeddings | |---|----------|------|---------------|------------| | 1 | **OpenAI (Codex OAuth)** | Auto-detected | Whisper | text-embedding-3-small | | 2 | **OpenAI (API key)** | `sk-...` key | Whisper | text-embedding-3-small | -| 3 | **Anthropic (Claude)** | API key | Not supported | Not supported | -| 4 | **Google (Gemini)** | API key | Not supported | text-embedding-004 | +| 3 | **PixelML (OpenRouter)** | API key | Not supported | Not supported | +| 4 | **Anthropic (Claude)** | API key | Not supported | Not supported | +| 5 | **Google (Gemini)** | API key | Not supported | text-embedding-004 | +| 6 | **DeepSeek-V4.1-Flash** | Your own endpoint | Not supported | Not supported | + +**DeepSeek-V4.1-Flash** talks to an OpenAI-compatible SGLang server that you run. +No endpoint ships with `av` — the preset defaults to SGLang's own local bind +address, and you point `AV_API_BASE_URL` at your deployment. Set `DEEPSEEK_API_KEY` +if your server requires one; leave it unset if it does not. Config is saved to `~/.config/av/config.json` and persists across sessions. @@ -134,6 +246,11 @@ export AV_TRANSCRIBE_MODEL="whisper" export AV_VISION_MODEL="gpt-4-1" export AV_EMBED_MODEL="text-embedding-3-small" export AV_CHAT_MODEL="gpt-4-1" + +# Self-hosted DeepSeek-V4.1-Flash via SGLang +export AV_PROVIDER="deepseek" +export AV_API_BASE_URL="http://your-sglang-host:30000/v1" +export DEEPSEEK_API_KEY="..." # only if your server requires one ``` ## Requirements @@ -156,6 +273,14 @@ export AV_CHAT_MODEL="gpt-4-1" | `av transcript ` | Output transcript (VTT/SRT/text) | | `av export` | Export as JSONL/VTT/SRT | | `av open --at ` | Open video at timestamp | +| `av bench gate` | Temporal-ordering capability gate | +| `av bench probe` | Measure a deployment's image-token and multi-image behaviour | +| `av bench plan` | Predict per-frame token cost against resolution (offline) | +| `av bench prepare` | Adapt a public benchmark's annotations into a task file | +| `av bench run` | Dense vs agentic arms, with tokens and dollars | +| `av bench sweep` | Event recall against sampling interval | +| `av bench noise` | Spread across identical runs | +| `av bench cost` | Cost arithmetic with labelled inputs (offline) | | `av version` | Print version JSON | ## License diff --git a/samples/bench-receipts/README.md b/samples/bench-receipts/README.md new file mode 100644 index 0000000..40cc53a --- /dev/null +++ b/samples/bench-receipts/README.md @@ -0,0 +1,48 @@ +# Sample bench receipts + +Real output from `av bench`, committed so the claims in the README have something +behind them. Every file here was produced by the commands named below on +2026-09-10, against **Gemini 2.5 Flash via its OpenAI-compatible endpoint** — chosen +because it was the provider already configured on the machine, not because it is the +subject of the benchmark. + +Read these as a demonstration that the harness measures what it says it measures. +They are **not** a model comparison, and no result here is a claim about any other +model or deployment. + +| File | Command | What it shows | +|------|---------|---------------| +| `gate-color-2-4-6-8.json` | `av bench gate --kind color --sizes 2,4,6,8` | Exact-order accuracy at 2, 4, 6 and 8 frames | +| `probe-tokens-per-frame.json` | `av bench probe` | Whether per-frame token cost is tunable on this endpoint | +| `noise-floor-saturated.json` | `av bench noise --n 6 --repeats 5` | A saturated cell, and the harness refusing to call its zero spread a noise floor | +| `sweep-two-probes-five-intervals.json` | `av bench sweep .../captions.jsonl .../videos --probes door_activity,person_enters --intervals 1,2,5,10,30 --max-per-probe 8 --cost token:0.30:2.50` | The full frontier: two probes collapsing at different rates | +| `sweep-door-activity.json` | `av bench sweep .../captions.jsonl .../videos --probes door_activity --intervals 1,5,30 --max-per-probe 4 --cost token:0.30:2.50` | Where detection collapses as frames thin out | +| `plan-image-tokens.json` | `av bench plan --budgets 200,400,800` | Predicted per-frame token cost against resolution (offline) | +| `cost-model-arithmetic.json` | `av bench cost --tokens-per-frame 1024 --context-tokens 1048576 --prefill-tok-s 20000 --hourly-usd 25 --kv-bytes-per-token 890` | The frontier arithmetic, with inputs recorded (offline) | + +## Reading a receipt + +- `claims` carries the labelled statements. `measured` means this run produced it; + `derived` means arithmetic over inputs; `documented` and `community-reported` must + cite a source; `untested` means we did not check. +- `determinism` carries temperature, seed, the exact ffmpeg invocations, and fixture + hashes, so a cell can be regenerated byte-for-byte. +- `provider.endpoint_host` is a hostname only. Private and tunnelled hosts are + redacted to `` — receipts are meant to be published. +- `notes` carries caveats that apply to the whole run, including the reference caveat + on event sweeps. + +## Two caveats that apply to the sweep receipt + +1. **The reference is model-generated.** Event windows come from the dense captioning + run shipped in `samples/epstein-cctv`, not from human annotation. The metric is + named `reference_recall` for that reason and must not be quoted as recall. +2. **The sample is small.** Four to eight windows per probe. It demonstrates the + shape of the frontier; it does not establish a rate to two significant figures. + +## What is deliberately absent + +No receipt here was produced against a self-hosted DeepSeek-V4.1-Flash deployment. +The provider and its image-token model are implemented and unit-tested, but the +harness has not been pointed at a running server, so no measured claim about that +model appears anywhere in this repository. diff --git a/samples/bench-receipts/cost-model-arithmetic.json b/samples/bench-receipts/cost-model-arithmetic.json new file mode 100644 index 0000000..23b54ff --- /dev/null +++ b/samples/bench-receipts/cost-model-arithmetic.json @@ -0,0 +1,109 @@ +{ + "receipt_version": 2, + "run_id": "eeffcfad-e5a1-477e-90bf-dd75557dfefb", + "created_utc": "2026-09-10T14:02:02+00:00", + "kind": "cost", + "av_version": "0.1.0", + "environment": { + "python": "3.11.11", + "platform": "macOS-26.3.1-arm64-arm-64bit", + "ffmpeg": "ffmpeg version 7.1 Copyright (c) 2000-2024 the FFmpeg developers" + }, + "provider": { + "note": "no provider contacted \u2014 this subcommand is arithmetic only" + }, + "determinism": { + "tokens_per_frame": 1024, + "context_tokens": 1048576, + "prefill_tok_per_s": 20000.0, + "hourly_usd": 25.0, + "kv_bytes_per_token": 890.0, + "intervals_sec": [ + 1.0, + 2.0, + 5.0, + 10.0, + 30.0, + 60.0 + ], + "input_source": "brief-supplied figures" + }, + "cost_model": { + "mode": "per_hour", + "hourly_usd": 25.0, + "prefill_tok_per_s": 20000.0, + "decode_tok_per_s": null + }, + "summary": { + "tokens_per_frame": 1024, + "frames_per_request_max": 1024, + "single_request_minutes_at_1fps": 17.07, + "target_video_hours_per_dollar": 100.0, + "required_interval_sec": 128.0, + "required_interval_description": "one frame every 128 seconds at 1024 tokens/frame" + }, + "cells": [ + { + "interval_sec": 1.0, + "frames_per_video_hour": 3600.0, + "tokens_per_video_hour": 3686400.0, + "gpu_seconds_per_video_hour": 184.32, + "cost_usd_per_video_hour": 1.28, + "video_hours_per_dollar": 0.781, + "kv_gib_per_video_hour": 3.0556 + }, + { + "interval_sec": 2.0, + "frames_per_video_hour": 1800.0, + "tokens_per_video_hour": 1843200.0, + "gpu_seconds_per_video_hour": 92.16, + "cost_usd_per_video_hour": 0.64, + "video_hours_per_dollar": 1.563, + "kv_gib_per_video_hour": 1.5278 + }, + { + "interval_sec": 5.0, + "frames_per_video_hour": 720.0, + "tokens_per_video_hour": 737280.0, + "gpu_seconds_per_video_hour": 36.864, + "cost_usd_per_video_hour": 0.256, + "video_hours_per_dollar": 3.906, + "kv_gib_per_video_hour": 0.6111 + }, + { + "interval_sec": 10.0, + "frames_per_video_hour": 360.0, + "tokens_per_video_hour": 368640.0, + "gpu_seconds_per_video_hour": 18.432, + "cost_usd_per_video_hour": 0.128, + "video_hours_per_dollar": 7.812, + "kv_gib_per_video_hour": 0.3056 + }, + { + "interval_sec": 30.0, + "frames_per_video_hour": 120.0, + "tokens_per_video_hour": 122880.0, + "gpu_seconds_per_video_hour": 6.144, + "cost_usd_per_video_hour": 0.042667, + "video_hours_per_dollar": 23.438, + "kv_gib_per_video_hour": 0.1019 + }, + { + "interval_sec": 60.0, + "frames_per_video_hour": 60.0, + "tokens_per_video_hour": 61440.0, + "gpu_seconds_per_video_hour": 3.072, + "cost_usd_per_video_hour": 0.021333, + "video_hours_per_dollar": 46.875, + "kv_gib_per_video_hour": 0.0509 + } + ], + "claims": [ + { + "label": "derived", + "statement": "Every figure in this receipt is arithmetic over the inputs supplied on the command line. None of it was measured against a running model.", + "source": null + } + ], + "notes": [] +} diff --git a/samples/bench-receipts/gate-color-2-4-6-8.json b/samples/bench-receipts/gate-color-2-4-6-8.json new file mode 100644 index 0000000..a98be0b --- /dev/null +++ b/samples/bench-receipts/gate-color-2-4-6-8.json @@ -0,0 +1,219 @@ +{ + "receipt_version": 2, + "run_id": "27fc6f81-943f-4a4b-88f7-21b39c2676ee", + "created_utc": "2026-09-10T14:02:59+00:00", + "kind": "gate", + "av_version": "0.1.0", + "environment": { + "python": "3.11.11", + "platform": "macOS-26.3.1-arm64-arm-64bit", + "ffmpeg": "ffmpeg version 7.1 Copyright (c) 2000-2024 the FFmpeg developers" + }, + "provider": { + "provider": "gemini", + "model": "gemini-2.5-flash", + "endpoint_host": "generativelanguage.googleapis.com", + "temperature": 0.0, + "seed": 0 + }, + "determinism": { + "temperature": 0.0, + "seed": 0, + "fixture_version": 1, + "ffmpeg_commands": [ + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFF0000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_02_00.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x00A000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_02_01.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFF0000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_04_00.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x00A000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_04_01.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x0000FF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_04_02.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFFFF00:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_04_03.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFF0000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_06_00.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x00A000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_06_01.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x0000FF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_06_02.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFFFF00:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_06_03.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFF00FF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_06_04.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x00FFFF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_06_05.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFF0000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_08_00.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x00A000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_08_01.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x0000FF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_08_02.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFFFF00:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_08_03.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFF00FF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_08_04.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x00FFFF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_08_05.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFFFFFF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_08_06.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFF8000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_gate_ag5c6e27/color_08_07.jpg" + ] + }, + "cost_model": null, + "summary": { + "fixture_kind": "color", + "sizes": [ + 2, + 4, + 6, + 8 + ], + "passed_sizes": [ + 2, + 4, + 6, + 8 + ], + "failed_sizes": [], + "largest_passing_n": 8, + "verdict": "pass" + }, + "cells": [ + { + "n": 2, + "fixture_kind": "color", + "expected": [ + "red", + "green" + ], + "reported": [ + "red", + "green" + ], + "exact_order": true, + "correct_prefix": 2, + "n_reported": 2, + "tokens_in": 608, + "tokens_out": 3, + "ttft_sec": 2.935, + "wall_sec": 2.9354, + "ok": true, + "error": null, + "multi_image_unsupported": false, + "raw_text": "red, green", + "frame_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7" + ] + }, + { + "n": 4, + "fixture_kind": "color", + "expected": [ + "red", + "green", + "blue", + "yellow" + ], + "reported": [ + "red", + "green", + "blue", + "yellow" + ], + "exact_order": true, + "correct_prefix": 4, + "n_reported": 4, + "tokens_in": 1124, + "tokens_out": 7, + "ttft_sec": 3.0127, + "wall_sec": 3.0169, + "ok": true, + "error": null, + "multi_image_unsupported": false, + "raw_text": "red,green,blue,yellow", + "frame_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7", + "7f440a482f75672a0173093d8433c28c1bc5e72e1df4199f05bf17c2d4f7db55", + "5323ca4428d59c27803079bc9f3f0f2aa34c8e3347c46a2b159c008d32b8bd4f" + ] + }, + { + "n": 6, + "fixture_kind": "color", + "expected": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "reported": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "exact_order": true, + "correct_prefix": 6, + "n_reported": 6, + "tokens_in": 1640, + "tokens_out": 11, + "ttft_sec": 2.766, + "wall_sec": 2.7666, + "ok": true, + "error": null, + "multi_image_unsupported": false, + "raw_text": "red,green,blue,yellow,magenta,cyan", + "frame_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7", + "7f440a482f75672a0173093d8433c28c1bc5e72e1df4199f05bf17c2d4f7db55", + "5323ca4428d59c27803079bc9f3f0f2aa34c8e3347c46a2b159c008d32b8bd4f", + "0033bce2024225a82c84aad67ab0f7c75b7af20d2d1eead163e4098ab11bc33d", + "a6cfb65c424e9027c7c54b563825cb01032df85bcc7f11550b69094df0d85b6b" + ] + }, + { + "n": 8, + "fixture_kind": "color", + "expected": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan", + "white", + "orange" + ], + "reported": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan", + "white", + "orange" + ], + "exact_order": true, + "correct_prefix": 8, + "n_reported": 8, + "tokens_in": 2156, + "tokens_out": 15, + "ttft_sec": 2.9276, + "wall_sec": 2.9299, + "ok": true, + "error": null, + "multi_image_unsupported": false, + "raw_text": "red, green, blue, yellow, magenta, cyan, white, orange", + "frame_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7", + "7f440a482f75672a0173093d8433c28c1bc5e72e1df4199f05bf17c2d4f7db55", + "5323ca4428d59c27803079bc9f3f0f2aa34c8e3347c46a2b159c008d32b8bd4f", + "0033bce2024225a82c84aad67ab0f7c75b7af20d2d1eead163e4098ab11bc33d", + "a6cfb65c424e9027c7c54b563825cb01032df85bcc7f11550b69094df0d85b6b", + "fd4c8bdd5d6c1f480b5ee1a433a8fe39378264b1e87d076164bc8ff8aed9f4a4", + "5cbcf6917389edc7d1b6d3db56c678bc35c60a5caafa527d51da9b8252408ef7" + ] + } + ], + "claims": [ + { + "label": "measured", + "statement": "Exact-order accuracy was 100% at every tested size [2, 4, 6, 8].", + "source": null + } + ], + "notes": [] +} diff --git a/samples/bench-receipts/noise-floor-saturated.json b/samples/bench-receipts/noise-floor-saturated.json new file mode 100644 index 0000000..498eacf --- /dev/null +++ b/samples/bench-receipts/noise-floor-saturated.json @@ -0,0 +1,269 @@ +{ + "receipt_version": 2, + "run_id": "4928d71e-dc7f-430b-b386-9d06f1780ade", + "created_utc": "2026-09-10T14:05:10+00:00", + "kind": "noise", + "av_version": "0.1.0", + "environment": { + "python": "3.11.11", + "platform": "macOS-26.3.1-arm64-arm-64bit", + "ffmpeg": "ffmpeg version 7.1 Copyright (c) 2000-2024 the FFmpeg developers" + }, + "provider": { + "provider": "gemini", + "model": "gemini-2.5-flash", + "endpoint_host": "generativelanguage.googleapis.com", + "temperature": 0.0, + "seed": 0 + }, + "determinism": { + "temperature": 0.0, + "seed": 0, + "repeats": 5, + "fixture_kind": "color", + "fixture_n": 6, + "ffmpeg_commands": [ + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFF0000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_noise_q7y6hpcg/color_06_00.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x00A000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_noise_q7y6hpcg/color_06_01.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x0000FF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_noise_q7y6hpcg/color_06_02.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFFFF00:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_noise_q7y6hpcg/color_06_03.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFF00FF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_noise_q7y6hpcg/color_06_04.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x00FFFF:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_noise_q7y6hpcg/color_06_05.jpg" + ], + "fixture_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7", + "7f440a482f75672a0173093d8433c28c1bc5e72e1df4199f05bf17c2d4f7db55", + "5323ca4428d59c27803079bc9f3f0f2aa34c8e3347c46a2b159c008d32b8bd4f", + "0033bce2024225a82c84aad67ab0f7c75b7af20d2d1eead163e4098ab11bc33d", + "a6cfb65c424e9027c7c54b563825cb01032df85bcc7f11550b69094df0d85b6b" + ] + }, + "cost_model": null, + "summary": { + "score_spread": { + "n": 5, + "unit": " score", + "values": [ + 1.0, + 1.0, + 1.0, + 1.0, + 1.0 + ], + "min": 1.0, + "max": 1.0, + "mean": 1.0, + "stdev": 0.0, + "range": 0.0 + }, + "prompt_token_spread": 0, + "interpretation": "every run hit the ceiling, so this spread measures nothing. Re-measure the noise floor on a cell the model does not solve perfectly.", + "saturated": true + }, + "cells": [ + { + "n": 6, + "fixture_kind": "color", + "expected": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "reported": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "exact_order": true, + "correct_prefix": 6, + "n_reported": 6, + "tokens_in": 1640, + "tokens_out": 11, + "ttft_sec": 2.9744, + "wall_sec": 2.9747, + "ok": true, + "error": null, + "multi_image_unsupported": false, + "raw_text": "red, green, blue, yellow, magenta, cyan", + "frame_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7", + "7f440a482f75672a0173093d8433c28c1bc5e72e1df4199f05bf17c2d4f7db55", + "5323ca4428d59c27803079bc9f3f0f2aa34c8e3347c46a2b159c008d32b8bd4f", + "0033bce2024225a82c84aad67ab0f7c75b7af20d2d1eead163e4098ab11bc33d", + "a6cfb65c424e9027c7c54b563825cb01032df85bcc7f11550b69094df0d85b6b" + ] + }, + { + "n": 6, + "fixture_kind": "color", + "expected": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "reported": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "exact_order": true, + "correct_prefix": 6, + "n_reported": 6, + "tokens_in": 1640, + "tokens_out": 11, + "ttft_sec": 2.8376, + "wall_sec": 2.8422, + "ok": true, + "error": null, + "multi_image_unsupported": false, + "raw_text": "red, green, blue, yellow, magenta, cyan", + "frame_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7", + "7f440a482f75672a0173093d8433c28c1bc5e72e1df4199f05bf17c2d4f7db55", + "5323ca4428d59c27803079bc9f3f0f2aa34c8e3347c46a2b159c008d32b8bd4f", + "0033bce2024225a82c84aad67ab0f7c75b7af20d2d1eead163e4098ab11bc33d", + "a6cfb65c424e9027c7c54b563825cb01032df85bcc7f11550b69094df0d85b6b" + ] + }, + { + "n": 6, + "fixture_kind": "color", + "expected": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "reported": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "exact_order": true, + "correct_prefix": 6, + "n_reported": 6, + "tokens_in": 1640, + "tokens_out": 11, + "ttft_sec": 3.1038, + "wall_sec": 3.1043, + "ok": true, + "error": null, + "multi_image_unsupported": false, + "raw_text": "red, green, blue, yellow, magenta, cyan", + "frame_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7", + "7f440a482f75672a0173093d8433c28c1bc5e72e1df4199f05bf17c2d4f7db55", + "5323ca4428d59c27803079bc9f3f0f2aa34c8e3347c46a2b159c008d32b8bd4f", + "0033bce2024225a82c84aad67ab0f7c75b7af20d2d1eead163e4098ab11bc33d", + "a6cfb65c424e9027c7c54b563825cb01032df85bcc7f11550b69094df0d85b6b" + ] + }, + { + "n": 6, + "fixture_kind": "color", + "expected": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "reported": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "exact_order": true, + "correct_prefix": 6, + "n_reported": 6, + "tokens_in": 1640, + "tokens_out": 11, + "ttft_sec": 2.5817, + "wall_sec": 2.5832, + "ok": true, + "error": null, + "multi_image_unsupported": false, + "raw_text": "red, green, blue, yellow, magenta, cyan", + "frame_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7", + "7f440a482f75672a0173093d8433c28c1bc5e72e1df4199f05bf17c2d4f7db55", + "5323ca4428d59c27803079bc9f3f0f2aa34c8e3347c46a2b159c008d32b8bd4f", + "0033bce2024225a82c84aad67ab0f7c75b7af20d2d1eead163e4098ab11bc33d", + "a6cfb65c424e9027c7c54b563825cb01032df85bcc7f11550b69094df0d85b6b" + ] + }, + { + "n": 6, + "fixture_kind": "color", + "expected": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "reported": [ + "red", + "green", + "blue", + "yellow", + "magenta", + "cyan" + ], + "exact_order": true, + "correct_prefix": 6, + "n_reported": 6, + "tokens_in": 1640, + "tokens_out": 11, + "ttft_sec": 2.6432, + "wall_sec": 2.6442, + "ok": true, + "error": null, + "multi_image_unsupported": false, + "raw_text": "red,green,blue,yellow,magenta,cyan", + "frame_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7", + "7f440a482f75672a0173093d8433c28c1bc5e72e1df4199f05bf17c2d4f7db55", + "5323ca4428d59c27803079bc9f3f0f2aa34c8e3347c46a2b159c008d32b8bd4f", + "0033bce2024225a82c84aad67ab0f7c75b7af20d2d1eead163e4098ab11bc33d", + "a6cfb65c424e9027c7c54b563825cb01032df85bcc7f11550b69094df0d85b6b" + ] + } + ], + "claims": [ + { + "label": "untested", + "statement": "Every run scored full marks, so this cell has no headroom to vary and its zero spread is not a usable noise floor. Re-run --n at a size the model does not solve perfectly.", + "source": null + } + ], + "notes": [] +} diff --git a/samples/bench-receipts/plan-image-tokens.json b/samples/bench-receipts/plan-image-tokens.json new file mode 100644 index 0000000..5d145bb --- /dev/null +++ b/samples/bench-receipts/plan-image-tokens.json @@ -0,0 +1,161 @@ +{ + "receipt_version": 2, + "run_id": "4adafa6a-06d9-4eb4-ac10-87c18cc559d6", + "created_utc": "2026-09-10T14:01:46+00:00", + "kind": "plan", + "av_version": "0.1.0", + "environment": { + "python": "3.11.11", + "platform": "macOS-26.3.1-arm64-arm-64bit", + "ffmpeg": "ffmpeg version 7.1 Copyright (c) 2000-2024 the FFmpeg developers" + }, + "provider": { + "note": "no provider contacted \u2014 this subcommand is arithmetic only" + }, + "determinism": { + "aspect_ratio": 1.7777777777777777, + "widths": "256,512,768,1024,1536,1920", + "budgets": "200,400,800" + }, + "cost_model": null, + "summary": { + "min_pixels_floor": 295936, + "min_tokens_per_frame": 184, + "max_tokens_per_frame": 1024, + "widths_for_budgets": { + "200": 742, + "400": 1050, + "800": 1554 + } + }, + "cells": [ + { + "width": 256, + "height": 144, + "source": [ + 256, + 144 + ], + "processed": [ + 728, + 420 + ], + "grid": [ + 10, + 18 + ], + "tokens": 192, + "upscaled": true, + "downscaled": false, + "approximate": false + }, + { + "width": 512, + "height": 288, + "source": [ + 512, + 288 + ], + "processed": [ + 728, + 420 + ], + "grid": [ + 10, + 18 + ], + "tokens": 192, + "upscaled": true, + "downscaled": false, + "approximate": false + }, + { + "width": 768, + "height": 432, + "source": [ + 768, + 432 + ], + "processed": [ + 770, + 434 + ], + "grid": [ + 11, + 19 + ], + "tokens": 222, + "upscaled": false, + "downscaled": false, + "approximate": false + }, + { + "width": 1024, + "height": 576, + "source": [ + 1024, + 576 + ], + "processed": [ + 1036, + 588 + ], + "grid": [ + 14, + 25 + ], + "tokens": 366, + "upscaled": false, + "downscaled": false, + "approximate": false + }, + { + "width": 1536, + "height": 864, + "source": [ + 1536, + 864 + ], + "processed": [ + 1540, + 868 + ], + "grid": [ + 21, + 37 + ], + "tokens": 800, + "upscaled": false, + "downscaled": false, + "approximate": false + }, + { + "width": 1920, + "height": 1080, + "source": [ + 1920, + 1080 + ], + "processed": [ + 1722, + 966 + ], + "grid": [ + 23, + 41 + ], + "tokens": 968, + "upscaled": false, + "downscaled": true, + "approximate": true + } + ], + "claims": [ + { + "label": "derived", + "statement": "Token counts are computed from the model's published vision preprocessor algorithm and reproduce its published worked examples exactly below the token ceiling. Rows marked approximate ran a shrink search and may differ by one grid step; measure those with `av bench probe`.", + "source": "model vision_config and reference image processor" + } + ], + "notes": [] +} diff --git a/samples/bench-receipts/probe-tokens-per-frame.json b/samples/bench-receipts/probe-tokens-per-frame.json new file mode 100644 index 0000000..5262bba --- /dev/null +++ b/samples/bench-receipts/probe-tokens-per-frame.json @@ -0,0 +1,121 @@ +{ + "receipt_version": 2, + "run_id": "986625a7-4bf6-4582-a1f9-086949910a70", + "created_utc": "2026-09-10T14:03:43+00:00", + "kind": "probe", + "av_version": "0.1.0", + "environment": { + "python": "3.11.11", + "platform": "macOS-26.3.1-arm64-arm-64bit", + "ffmpeg": "ffmpeg version 7.1 Copyright (c) 2000-2024 the FFmpeg developers" + }, + "provider": { + "provider": "gemini", + "model": "gemini-2.5-flash", + "endpoint_host": "generativelanguage.googleapis.com", + "temperature": 0.0, + "seed": 0 + }, + "determinism": { + "temperature": 0.0, + "seed": 0, + "fixture_ffmpeg_commands": [ + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0xFF0000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_probe_8wxduo1r/color_02_00.jpg", + "ffmpeg -hide_banner -loglevel error -nostdin -y -f lavfi -i color=c=0x00A000:s=512x512:d=1 -frames:v 1 -q:v 2 /var/folders/wf/wt27p9s965s_w_3tq16vpyvc0000gq/T/av_bench_probe_8wxduo1r/color_02_01.jpg" + ], + "fixture_sha256": [ + "c86fbf9f1ba254965df1d5846ae5b7481bdf65a1824dfac19536ab2b1838cd03", + "91a25b1c70ca2b4344162e17c3cc10d7b4f6e1e72297f13dc1e0caa4fb1624f7" + ] + }, + "cost_model": null, + "summary": { + "tokens_per_frame_verdict": "fixed", + "effective_knobs": [], + "distinct_image_token_counts": [ + 258 + ], + "multi_image_supported": true, + "multi_image_error": null + }, + "cells": [ + { + "baseline_text_only_prompt_tokens": 8, + "baseline_ok": true, + "baseline_error": null, + "detail_observations": [ + { + "detail": "(unset)", + "ok": true, + "error": null, + "prompt_tokens": 266, + "tokens_attributable_to_image": 258 + }, + { + "detail": "low", + "ok": true, + "error": null, + "prompt_tokens": 266, + "tokens_attributable_to_image": 258 + }, + { + "detail": "high", + "ok": true, + "error": null, + "prompt_tokens": 266, + "tokens_attributable_to_image": 258 + } + ], + "resolution_observations": [ + { + "width": 256, + "ok": true, + "error": null, + "prompt_tokens": 266, + "tokens_attributable_to_image": 258 + }, + { + "width": 512, + "ok": true, + "error": null, + "prompt_tokens": 266, + "tokens_attributable_to_image": 258 + }, + { + "width": 768, + "ok": true, + "error": null, + "prompt_tokens": 266, + "tokens_attributable_to_image": 258 + }, + { + "width": 1024, + "ok": true, + "error": null, + "prompt_tokens": 266, + "tokens_attributable_to_image": 258 + }, + { + "width": 1536, + "ok": true, + "error": null, + "prompt_tokens": 266, + "tokens_attributable_to_image": 258 + } + ], + "distinct_image_token_counts": [ + 258 + ], + "effective_knobs": [], + "verdict": "fixed" + } + ], + "claims": [ + { + "label": "measured", + "statement": "Per-frame token count did not change across any tested detail setting or input resolution on this deployment. The tokens-per-frame axis is dropped rather than simulated.", + "source": null + } + ], + "notes": [] +} diff --git a/samples/bench-receipts/sweep-door-activity.json b/samples/bench-receipts/sweep-door-activity.json new file mode 100644 index 0000000..3bce78b --- /dev/null +++ b/samples/bench-receipts/sweep-door-activity.json @@ -0,0 +1,130 @@ +{ + "receipt_version": 2, + "run_id": "e2a0e1dd-4c8c-4a2f-88c2-b03873a1f383", + "created_utc": "2026-09-10T14:12:19+00:00", + "kind": "sweep", + "av_version": "0.1.0", + "environment": { + "python": "3.11.11", + "platform": "macOS-26.3.1-arm64-arm-64bit", + "ffmpeg": "ffmpeg version 7.1 Copyright (c) 2000-2024 the FFmpeg developers" + }, + "provider": { + "provider": "gemini", + "model": "gemini-2.5-flash", + "endpoint_host": "generativelanguage.googleapis.com", + "temperature": 0.0, + "seed": 0 + }, + "determinism": { + "temperature": 0.0, + "seed": 0, + "intervals_sec": [ + 1.0, + 5.0, + 30.0 + ], + "scale_width": 768, + "artifacts_file": "/Users/sean/WIP/Antigravity-Workspace/av/samples/epstein-cctv/hf-dataset/captions.jsonl", + "artifacts_sha256": "4bd43793ec44affcdc11ddf926efdc4332b141f832fd233e25b1f7ad7f12586a", + "probe_questions": { + "door_activity": "Does anyone open, close, or interact with a door or gate in these frames?" + } + }, + "cost_model": { + "mode": "per_token", + "input_usd_per_mtok": 0.3, + "output_usd_per_mtok": 2.5, + "cached_input_usd_per_mtok": null + }, + "summary": { + "probes": [ + "door_activity" + ], + "intervals_sec": [ + 1.0, + 5.0, + 30.0 + ], + "reference_windows": 4, + "frontier": { + "door_activity": { + "baseline_interval_sec": 1.0, + "baseline_score": 1.0, + "safe_interval_sec": 1.0, + "safe_score": 1.0, + "collapse_interval_sec": 5.0, + "collapse_score": 0.25, + "tolerance": 0.1 + } + } + }, + "cells": [ + { + "interval_sec": 1.0, + "probe": "door_activity", + "events_total": 4, + "events_detected": 4, + "reference_recall": 1.0, + "unparseable": 0, + "frames_total": 120, + "tokens_in": 31100, + "tokens_out": 20, + "wall_sec": 29.884, + "errors": [], + "score": 1.0, + "cost_usd": 0.00938, + "video_hours_per_dollar": 3.5536602700781805, + "cost_basis": "measured" + }, + { + "interval_sec": 5.0, + "probe": "door_activity", + "events_total": 4, + "events_detected": 1, + "reference_recall": 0.25, + "unparseable": 0, + "frames_total": 24, + "tokens_in": 6332, + "tokens_out": 30, + "wall_sec": 18.205, + "errors": [], + "score": 0.25, + "cost_usd": 0.001975, + "video_hours_per_dollar": 16.881056078868294, + "cost_basis": "measured" + }, + { + "interval_sec": 30.0, + "probe": "door_activity", + "events_total": 4, + "events_detected": 1, + "reference_recall": 0.25, + "unparseable": 0, + "frames_total": 4, + "tokens_in": 1172, + "tokens_out": 20, + "wall_sec": 15.647, + "errors": [], + "score": 0.25, + "cost_usd": 0.000402, + "video_hours_per_dollar": 83.00132802124833, + "cost_basis": "measured" + } + ], + "claims": [ + { + "label": "measured", + "statement": "Detection rates and token counts are from this run.", + "source": null + }, + { + "label": "untested", + "statement": "Precision was not measured: only windows the reference marks as containing the event were shown, so false positives on empty windows are unknown.", + "source": null + } + ], + "notes": [ + "Reference windows come from a dense vision-model run shipped with this repository, not from human annotation. This metric measures agreement with that reference run and must not be quoted as recall against ground truth." + ] +} diff --git a/samples/bench-receipts/sweep-two-probes-five-intervals.json b/samples/bench-receipts/sweep-two-probes-five-intervals.json new file mode 100644 index 0000000..6ad5dff --- /dev/null +++ b/samples/bench-receipts/sweep-two-probes-five-intervals.json @@ -0,0 +1,266 @@ +{ + "receipt_version": 2, + "run_id": "025703b6-f3ce-4113-a093-427a960be54a", + "created_utc": "2026-09-10T14:19:59+00:00", + "kind": "sweep", + "av_version": "0.1.0", + "environment": { + "python": "3.11.11", + "platform": "macOS-26.3.1-arm64-arm-64bit", + "ffmpeg": "ffmpeg version 7.1 Copyright (c) 2000-2024 the FFmpeg developers" + }, + "provider": { + "provider": "gemini", + "model": "gemini-2.5-flash", + "endpoint_host": "generativelanguage.googleapis.com", + "temperature": 0.0, + "seed": 0 + }, + "determinism": { + "temperature": 0.0, + "seed": 0, + "intervals_sec": [ + 1.0, + 2.0, + 5.0, + 10.0, + 30.0 + ], + "scale_width": 768, + "artifacts_file": "/Users/sean/WIP/Antigravity-Workspace/av/samples/epstein-cctv/hf-dataset/captions.jsonl", + "artifacts_sha256": "4bd43793ec44affcdc11ddf926efdc4332b141f832fd233e25b1f7ad7f12586a", + "probe_questions": { + "door_activity": "Does anyone open, close, or interact with a door or gate in these frames?", + "person_enters": "Does any person enter the scene in these frames?" + } + }, + "cost_model": { + "mode": "per_token", + "input_usd_per_mtok": 0.3, + "output_usd_per_mtok": 2.5, + "cached_input_usd_per_mtok": null + }, + "summary": { + "probes": [ + "door_activity", + "person_enters" + ], + "intervals_sec": [ + 1.0, + 2.0, + 5.0, + 10.0, + 30.0 + ], + "reference_windows": 16, + "frontier": { + "door_activity": { + "baseline_interval_sec": 1.0, + "baseline_score": 0.625, + "safe_interval_sec": 1.0, + "safe_score": 0.625, + "collapse_interval_sec": 2.0, + "collapse_score": 0.5, + "tolerance": 0.1 + }, + "person_enters": { + "baseline_interval_sec": 1.0, + "baseline_score": 0.875, + "safe_interval_sec": 1.0, + "safe_score": 0.875, + "collapse_interval_sec": 2.0, + "collapse_score": 0.625, + "tolerance": 0.1 + } + } + }, + "cells": [ + { + "interval_sec": 1.0, + "probe": "door_activity", + "events_total": 8, + "events_detected": 5, + "reference_recall": 0.625, + "unparseable": 0, + "frames_total": 222, + "tokens_in": 57556, + "tokens_out": 50, + "wall_sec": 52.181, + "errors": [], + "score": 0.625, + "cost_usd": 0.017392, + "video_hours_per_dollar": 3.5465837827788578, + "cost_basis": "measured" + }, + { + "interval_sec": 2.0, + "probe": "door_activity", + "events_total": 8, + "events_detected": 4, + "reference_recall": 0.5, + "unparseable": 0, + "frames_total": 111, + "tokens_in": 28918, + "tokens_out": 60, + "wall_sec": 54.793, + "errors": [], + "score": 0.5, + "cost_usd": 0.008825, + "video_hours_per_dollar": 6.989085574969217, + "cost_basis": "measured" + }, + { + "interval_sec": 5.0, + "probe": "door_activity", + "events_total": 8, + "events_detected": 1, + "reference_recall": 0.125, + "unparseable": 0, + "frames_total": 44, + "tokens_in": 11632, + "tokens_out": 50, + "wall_sec": 28.027, + "errors": [], + "score": 0.125, + "cost_usd": 0.003615, + "video_hours_per_dollar": 17.06453710876261, + "cost_basis": "measured" + }, + { + "interval_sec": 10.0, + "probe": "door_activity", + "events_total": 8, + "events_detected": 1, + "reference_recall": 0.125, + "unparseable": 0, + "frames_total": 22, + "tokens_in": 5956, + "tokens_out": 40, + "wall_sec": 29.511, + "errors": [], + "score": 0.125, + "cost_usd": 0.001887, + "video_hours_per_dollar": 32.691051427460955, + "cost_basis": "measured" + }, + { + "interval_sec": 30.0, + "probe": "door_activity", + "events_total": 8, + "events_detected": 1, + "reference_recall": 0.125, + "unparseable": 0, + "frames_total": 7, + "tokens_in": 2051, + "tokens_out": 35, + "wall_sec": 29.358, + "errors": [ + "no frames at 0s in EFTA00028842.mp4" + ], + "score": 0.125, + "cost_usd": 0.000703, + "video_hours_per_dollar": 87.76533271675204, + "cost_basis": "measured" + }, + { + "interval_sec": 1.0, + "probe": "person_enters", + "events_total": 8, + "events_detected": 7, + "reference_recall": 0.875, + "unparseable": 0, + "frames_total": 240, + "tokens_in": 62144, + "tokens_out": 68, + "wall_sec": 62.344, + "errors": [], + "score": 0.875, + "cost_usd": 0.018813, + "video_hours_per_dollar": 3.543611223325467, + "cost_basis": "measured" + }, + { + "interval_sec": 2.0, + "probe": "person_enters", + "events_total": 8, + "events_detected": 5, + "reference_recall": 0.625, + "unparseable": 0, + "frames_total": 120, + "tokens_in": 31184, + "tokens_out": 64, + "wall_sec": 40.365, + "errors": [], + "score": 0.625, + "cost_usd": 0.009515, + "video_hours_per_dollar": 7.006333725688022, + "cost_basis": "measured" + }, + { + "interval_sec": 5.0, + "probe": "person_enters", + "events_total": 8, + "events_detected": 4, + "reference_recall": 0.5, + "unparseable": 0, + "frames_total": 48, + "tokens_in": 12608, + "tokens_out": 60, + "wall_sec": 33.782, + "errors": [], + "score": 0.5, + "cost_usd": 0.003932, + "video_hours_per_dollar": 16.95317532973926, + "cost_basis": "measured" + }, + { + "interval_sec": 10.0, + "probe": "person_enters", + "events_total": 8, + "events_detected": 4, + "reference_recall": 0.5, + "unparseable": 0, + "frames_total": 24, + "tokens_in": 6416, + "tokens_out": 55, + "wall_sec": 27.762, + "errors": [], + "score": 0.5, + "cost_usd": 0.002062, + "video_hours_per_dollar": 32.32636700124456, + "cost_basis": "measured" + }, + { + "interval_sec": 30.0, + "probe": "person_enters", + "events_total": 8, + "events_detected": 2, + "reference_recall": 0.25, + "unparseable": 0, + "frames_total": 8, + "tokens_in": 2288, + "tokens_out": 65, + "wall_sec": 34.325, + "errors": [], + "score": 0.25, + "cost_usd": 0.000849, + "video_hours_per_dollar": 78.53300349471866, + "cost_basis": "measured" + } + ], + "claims": [ + { + "label": "measured", + "statement": "Detection rates and token counts are from this run.", + "source": null + }, + { + "label": "untested", + "statement": "Precision was not measured: only windows the reference marks as containing the event were shown, so false positives on empty windows are unknown.", + "source": null + } + ], + "notes": [ + "Reference windows come from a dense vision-model run shipped with this repository, not from human annotation. This metric measures agreement with that reference run and must not be quoted as recall against ground truth." + ] +} diff --git a/src/av/bench/__init__.py b/src/av/bench/__init__.py new file mode 100644 index 0000000..8f68b36 --- /dev/null +++ b/src/av/bench/__init__.py @@ -0,0 +1,23 @@ +"""av bench — cost/accuracy frontier measurement for video understanding. + +Two headline axes, chosen to be readable against published agentic-video results: +tokens per query and task accuracy. A third axis that API vendors cannot report is +added alongside: dollars per query and video-hours per dollar on hardware you own. + +Everything here writes a receipt. No number in a report should exist without one. +""" + +from __future__ import annotations + +from av.bench.cost import CostModel, HourlyCost, PerTokenCost, parse_cost_model +from av.bench.receipts import Claim, Receipt, write_receipt + +__all__ = [ + "Claim", + "CostModel", + "HourlyCost", + "PerTokenCost", + "Receipt", + "parse_cost_model", + "write_receipt", +] diff --git a/src/av/bench/cost.py b/src/av/bench/cost.py new file mode 100644 index 0000000..4670abd --- /dev/null +++ b/src/av/bench/cost.py @@ -0,0 +1,176 @@ +"""Cost models for benchmark cells. + +Self-hosted and API economics are different shapes and must never be conflated: + +- ``HourlyCost`` — you rent or own the box. Cost accrues with wall-clock time, + regardless of how many tokens you push through it. Cheaper per token the busier + the box is. +- ``PerTokenCost`` — a vendor bills each token. Cost is independent of wall clock, + and an idle box costs nothing. + +Reporting one in the other's units produces a number that means nothing, so the +receipt always records which model produced a figure. +""" + +from __future__ import annotations + +import json +from dataclasses import asdict, dataclass +from pathlib import Path + + +@dataclass +class CellUsage: + """Measured usage for one benchmark cell.""" + + tokens_in: int = 0 + tokens_out: int = 0 + wall_total_sec: float = 0.0 + # Time to first token. On a streaming API this is the closest measurable + # proxy for prefill; it is not prefill itself and is labelled as a proxy. + ttft_sec: float | None = None + requests: int = 0 + + @property + def tokens_total(self) -> int: + return self.tokens_in + self.tokens_out + + +class CostModel: + """Base class. Subclasses turn measured usage into dollars.""" + + mode: str = "unknown" + + def cost_usd(self, usage: CellUsage) -> float: # pragma: no cover - abstract + raise NotImplementedError + + def describe(self) -> dict: # pragma: no cover - trivial + raise NotImplementedError + + def basis(self) -> str: + """`measured` if dollars follow from measured quantities alone.""" + return "measured" + + +@dataclass +class PerTokenCost(CostModel): + """Vendor API pricing, quoted per million tokens.""" + + input_usd_per_mtok: float + output_usd_per_mtok: float + cached_input_usd_per_mtok: float | None = None + mode: str = "per_token" + + def cost_usd(self, usage: CellUsage) -> float: + return ( + usage.tokens_in / 1_000_000 * self.input_usd_per_mtok + + usage.tokens_out / 1_000_000 * self.output_usd_per_mtok + ) + + def describe(self) -> dict: + return {"mode": "per_token", **asdict(self)} + + +@dataclass +class HourlyCost(CostModel): + """Self-hosted or rented hardware, billed by the hour. + + With no throughput assumption, cost is measured wall clock x hourly rate — an + honest single-stream number that under-uses the box. Supply + ``prefill_tok_per_s`` / ``decode_tok_per_s`` to model a saturated box instead; + that figure is *derived*, not measured, and the receipt says so. + """ + + hourly_usd: float + prefill_tok_per_s: float | None = None + decode_tok_per_s: float | None = None + mode: str = "per_hour" + + def cost_usd(self, usage: CellUsage) -> float: + if self.prefill_tok_per_s: + seconds = usage.tokens_in / self.prefill_tok_per_s + if self.decode_tok_per_s and usage.tokens_out: + seconds += usage.tokens_out / self.decode_tok_per_s + else: + seconds = usage.wall_total_sec + return seconds / 3600.0 * self.hourly_usd + + def basis(self) -> str: + return "derived" if self.prefill_tok_per_s else "measured" + + def describe(self) -> dict: + return {"mode": "per_hour", **asdict(self)} + + +def video_hours_per_dollar(video_seconds: float, cost_usd: float) -> float | None: + """The business metric: video-hours analysed per dollar spent.""" + if cost_usd <= 0: + return None + return (video_seconds / 3600.0) / cost_usd + + +def parse_cost_model(spec: str | None) -> CostModel | None: + """Parse a ``--cost`` spec into a cost model. + + Accepted forms:: + + hourly:25.0 measured wall clock x $25/hr + hourly:25.0:20000 modelled at 20,000 prefill tok/s (derived) + hourly:25.0:20000:1200 ... and 1,200 decode tok/s + token:0.30:2.50 $0.30/Mtok in, $2.50/Mtok out + @/path/to/cost.json a JSON file with the same fields + + Returns ``None`` when *spec* is empty, so callers can report token counts + without inventing a price. + """ + if not spec: + return None + + spec = spec.strip() + if spec.startswith("@"): + data = json.loads(Path(spec[1:]).expanduser().read_text()) + mode = data.get("mode") + if mode == "per_token": + return PerTokenCost( + input_usd_per_mtok=float(data["input_usd_per_mtok"]), + output_usd_per_mtok=float(data["output_usd_per_mtok"]), + cached_input_usd_per_mtok=( + float(data["cached_input_usd_per_mtok"]) + if data.get("cached_input_usd_per_mtok") is not None + else None + ), + ) + if mode == "per_hour": + return HourlyCost( + hourly_usd=float(data["hourly_usd"]), + prefill_tok_per_s=( + float(data["prefill_tok_per_s"]) if data.get("prefill_tok_per_s") else None + ), + decode_tok_per_s=( + float(data["decode_tok_per_s"]) if data.get("decode_tok_per_s") else None + ), + ) + raise ValueError(f"cost file has unknown mode: {mode!r}") + + parts = spec.split(":") + kind = parts[0].lower() + + if kind in ("hourly", "hour", "hr"): + if len(parts) < 2: + raise ValueError("hourly cost needs a rate, e.g. hourly:25.0") + return HourlyCost( + hourly_usd=float(parts[1]), + prefill_tok_per_s=float(parts[2]) if len(parts) > 2 and parts[2] else None, + decode_tok_per_s=float(parts[3]) if len(parts) > 3 and parts[3] else None, + ) + + if kind in ("token", "tokens", "per-token"): + if len(parts) < 3: + raise ValueError("token cost needs input and output rates, e.g. token:0.30:2.50") + return PerTokenCost( + input_usd_per_mtok=float(parts[1]), + output_usd_per_mtok=float(parts[2]), + cached_input_usd_per_mtok=float(parts[3]) if len(parts) > 3 and parts[3] else None, + ) + + raise ValueError(f"unknown cost spec: {spec!r} (expected hourly:..., token:..., or @file.json)") diff --git a/src/av/bench/datasets.py b/src/av/bench/datasets.py new file mode 100644 index 0000000..8303b97 --- /dev/null +++ b/src/av/bench/datasets.py @@ -0,0 +1,190 @@ +"""Adapters for public video-QA benchmarks. + +No benchmark data is vendored into this repository. These adapters read an +annotation file you fetched yourself and emit the harness's own JSONL task format, +so licences stay with their owners and this Apache-2.0 tree stays clean. LVBench in +particular is CC BY-NC-SA with an explicit commercial-use prohibition — parsing it +is fine, redistributing it here would not be. + +Videos are never fetched by these adapters. Both benchmarks reference YouTube ids, +which means yt-dlp, bandwidth, and link rot are the caller's problem and the caller's +decision. Each emitted row records the source id so a fetch step can be run separately. +""" + +from __future__ import annotations + +import json +import re +from dataclasses import dataclass +from pathlib import Path + +# Where the annotation files come from. Recorded here so a receipt can name a source +# rather than a vague dataset name. +SOURCES: dict[str, dict[str, str]] = { + "minerva": { + "name": "MINERVA", + "annotations_url": "https://storage.googleapis.com/neptunedata/minerva.json", + "repo": "https://github.com/google-deepmind/neptune", + "paper": "arXiv:2505.00681", + "annotations_licence": "CC BY 4.0", + "video_licence": "not granted — YouTube ids only", + "format": "5-way multiple choice", + }, + "lvbench": { + "name": "LVBench", + "annotations_url": ( + "https://huggingface.co/datasets/zai-org/LVBench/resolve/main/video_info.meta.jsonl" + ), + "repo": "https://github.com/zai-org/LVBench", + "paper": "arXiv:2406.08035", + "annotations_licence": "CC BY-NC-SA 4.0 (non-commercial; see upstream)", + "video_licence": "not granted — YouTube ids only", + "format": "4-way multiple choice", + }, +} + +_LETTERS = "ABCDEFGH" +_OPTION_LINE = re.compile(r"^\(([A-H])\)\s*(.*)$") +_TIME_RANGE = re.compile(r"^(\d{1,2}):(\d{2})(?::(\d{2}))?-(\d{1,2}):(\d{2})(?::(\d{2}))?$") + + +@dataclass +class AdaptedRow: + id: str + video_id: str + question: str + options: list[str] + answer: str + start_sec: float | None = None + end_sec: float | None = None + meta: dict | None = None + + def to_task_row(self, video_template: str) -> dict: + row: dict = { + "id": self.id, + "video": video_template.format(video_id=self.video_id, id=self.id), + "question": self.question, + "options": self.options, + "answer": self.answer, + "meta": {"video_id": self.video_id, **(self.meta or {})}, + } + if self.start_sec is not None: + row["start_sec"] = self.start_sec + if self.end_sec is not None: + row["end_sec"] = self.end_sec + return row + + +def _parse_timespan(value: str) -> tuple[float, float] | None: + """LVBench's ``time_reference`` — ``MM:SS-MM:SS`` or ``HH:MM:SS-HH:MM:SS``.""" + m = _TIME_RANGE.match((value or "").strip()) + if not m: + return None + a1, a2, a3, b1, b2, b3 = m.groups() + start = (int(a1) * 3600 + int(a2) * 60 + int(a3)) if a3 else (int(a1) * 60 + int(a2)) + end = (int(b1) * 3600 + int(b2) * 60 + int(b3)) if b3 else (int(b1) * 60 + int(b2)) + return (float(start), float(end)) if end > start else None + + +def adapt_minerva(annotations: list[dict]) -> list[AdaptedRow]: + """MINERVA: flat list, choices in ``answer_choice_N``, gold index in ``answer_id``.""" + rows: list[AdaptedRow] = [] + for item in annotations: + choices = [] + i = 0 + while f"answer_choice_{i}" in item: + choices.append(str(item[f"answer_choice_{i}"])) + i += 1 + if not choices: + continue + gold = item.get("answer_id") + if not isinstance(gold, int) or not 0 <= gold < len(choices): + continue + rows.append( + AdaptedRow( + id=str(item["key"]), + video_id=str(item["video_id"]), + question=str(item["question"]), + options=[f"{_LETTERS[i]}. {c}" for i, c in enumerate(choices)], + answer=_LETTERS[gold], + meta={ + "question_type": item.get("question_type"), + "split": item.get("split"), + "category": item.get("category"), + }, + ) + ) + return rows + + +def adapt_lvbench(annotations: list[dict]) -> list[AdaptedRow]: + """LVBench: one object per video, options inlined in the question text as ``(A) ...``.""" + rows: list[AdaptedRow] = [] + for video in annotations: + video_id = str(video.get("key", "")) + for qa in video.get("qa") or []: + lines = str(qa.get("question", "")).splitlines() + stem_lines: list[str] = [] + options: list[str] = [] + for line in lines: + m = _OPTION_LINE.match(line.strip()) + if m: + options.append(f"{m.group(1)}. {m.group(2)}") + elif not options: + stem_lines.append(line) + if not options: + continue + span = _parse_timespan(str(qa.get("time_reference", ""))) + rows.append( + AdaptedRow( + id=f"{video_id}:{qa.get('uid')}", + video_id=video_id, + question="\n".join(stem_lines).strip(), + options=options, + answer=str(qa.get("answer", "")).strip().upper(), + start_sec=span[0] if span else None, + end_sec=span[1] if span else None, + meta={ + "question_type": qa.get("question_type"), + "video_type": video.get("type"), + "time_reference": qa.get("time_reference"), + }, + ) + ) + return rows + + +ADAPTERS = {"minerva": adapt_minerva, "lvbench": adapt_lvbench} + + +def load_annotations(path: Path) -> list[dict]: + """Read either a JSON array (MINERVA) or JSONL (LVBench).""" + text = Path(path).expanduser().read_text() + stripped = text.lstrip() + if stripped.startswith("["): + return json.loads(text) + return [json.loads(line) for line in text.splitlines() if line.strip()] + + +def subset_by_video( + rows: list[AdaptedRow], *, max_questions: int, max_videos: int | None = None +) -> list[AdaptedRow]: + """Take a subset grouped by video, so a small run does not pull many hours of video. + + Rows are ordered by their stable id first, so the same arguments always select the + same questions. Sampling by question instead of by video is how a 30-question run + turns into a 40-hour download. + """ + by_video: dict[str, list[AdaptedRow]] = {} + for row in sorted(rows, key=lambda r: r.id): + by_video.setdefault(row.video_id, []).append(row) + + chosen: list[AdaptedRow] = [] + for n, (_, group) in enumerate(sorted(by_video.items())): + if max_videos is not None and n >= max_videos: + break + for row in group: + if len(chosen) >= max_questions: + return chosen + chosen.append(row) + return chosen diff --git a/src/av/bench/fixtures.py b/src/av/bench/fixtures.py new file mode 100644 index 0000000..32a26b7 --- /dev/null +++ b/src/av/bench/fixtures.py @@ -0,0 +1,168 @@ +"""Deterministic synthetic fixtures for the temporal-ordering capability gate. + +Every fixture is generated by a pinned ffmpeg invocation from a solid-colour +``lavfi`` source and font-free ``drawbox`` overlays, so it reproduces byte-for-byte +on any machine with the same ffmpeg build. No fonts, no random seeds, no assets. + +Three ordered properties, all trivially checkable: + +``color`` frame *k* is a named solid colour from a fixed palette +``count`` frame *k* carries exactly *k* white squares on black +``motion`` a single white square advances left to right, one step per frame +""" + +from __future__ import annotations + +import subprocess +from dataclasses import dataclass +from pathlib import Path + +from av.core.exceptions import FFmpegError + +FIXTURE_SIZE = 512 +FIXTURE_VERSION = 1 + +# Ordered, unambiguously nameable colours. Order is part of the contract: +# changing it changes every colour fixture, so append only. +COLOR_PALETTE: list[tuple[str, str]] = [ + ("red", "0xFF0000"), + ("green", "0x00A000"), + ("blue", "0x0000FF"), + ("yellow", "0xFFFF00"), + ("magenta", "0xFF00FF"), + ("cyan", "0x00FFFF"), + ("white", "0xFFFFFF"), + ("orange", "0xFF8000"), + ("purple", "0x8000FF"), + ("pink", "0xFF80C0"), + ("brown", "0x804000"), + ("gray", "0x808080"), +] + +FIXTURE_KINDS = ("color", "count", "motion") + +# Practical ceilings. `color` is limited by how many colours a person (or a model) +# can name without ambiguity; the others by how many marks fit legibly in a frame. +MAX_N = {"color": len(COLOR_PALETTE), "count": 32, "motion": 32} + + +@dataclass +class Fixture: + kind: str + n: int + frame_paths: list[Path] + labels: list[str] + ffmpeg_commands: list[str] + + @property + def expected_answer(self) -> list[str]: + return list(self.labels) + + +def _run_ffmpeg(cmd: list[str]) -> None: + try: + subprocess.run(cmd, capture_output=True, text=True, check=True, timeout=60) + except FileNotFoundError: + raise FFmpegError("ffmpeg not found. Install ffmpeg: brew install ffmpeg", cmd=" ".join(cmd)) + except subprocess.CalledProcessError as e: + raise FFmpegError( + f"Fixture generation failed: {e.stderr}", cmd=" ".join(cmd), returncode=e.returncode + ) + except subprocess.TimeoutExpired: + raise FFmpegError("Fixture generation timed out", cmd=" ".join(cmd)) + + +def _base_cmd(source: str, out_path: Path, vf: str | None = None) -> list[str]: + cmd = [ + "ffmpeg", "-hide_banner", "-loglevel", "error", "-nostdin", "-y", + "-f", "lavfi", "-i", source, + ] + if vf: + cmd += ["-vf", vf] + cmd += ["-frames:v", "1", "-q:v", "2", str(out_path)] + return cmd + + +def _count_boxes(k: int) -> str: + """`k` white squares on a fixed 6-column grid, filled in reading order.""" + box, gap, margin = 48, 16, 40 + parts = [] + for i in range(k): + col, row = i % 6, i // 6 + x = margin + col * (box + gap) + y = margin + row * (box + gap) + parts.append(f"drawbox=x={x}:y={y}:w={box}:h={box}:color=white:t=fill") + return ",".join(parts) + + +def _motion_box(k: int, n: int) -> str: + """One white square whose x position is a fixed function of frame index.""" + box = 64 + span = FIXTURE_SIZE - box - 2 * 32 + x = 32 + (span * k // max(n - 1, 1)) + y = (FIXTURE_SIZE - box) // 2 + return f"drawbox=x={x}:y={y}:w={box}:h={box}:color=white:t=fill" + + +def generate_fixture(kind: str, n: int, out_dir: Path) -> Fixture: + """Generate an ``n``-frame ordered fixture of the given kind.""" + if kind not in FIXTURE_KINDS: + raise ValueError(f"unknown fixture kind {kind!r}; expected one of {FIXTURE_KINDS}") + if n < 2: + raise ValueError("a fixture needs at least 2 frames to carry an order") + if n > MAX_N[kind]: + raise ValueError(f"{kind} fixtures support at most {MAX_N[kind]} frames (asked for {n})") + + out_dir = Path(out_dir).expanduser() + out_dir.mkdir(parents=True, exist_ok=True) + + size = f"{FIXTURE_SIZE}x{FIXTURE_SIZE}" + frame_paths: list[Path] = [] + labels: list[str] = [] + commands: list[str] = [] + + for k in range(n): + out_path = out_dir / f"{kind}_{n:02d}_{k:02d}.jpg" + if kind == "color": + name, hex_code = COLOR_PALETTE[k] + cmd = _base_cmd(f"color=c={hex_code}:s={size}:d=1", out_path) + label = name + elif kind == "count": + cmd = _base_cmd(f"color=c=black:s={size}:d=1", out_path, _count_boxes(k + 1)) + label = str(k + 1) + else: # motion + cmd = _base_cmd(f"color=c=black:s={size}:d=1", out_path, _motion_box(k, n)) + label = str(k + 1) + + _run_ffmpeg(cmd) + frame_paths.append(out_path) + labels.append(label) + commands.append(" ".join(cmd)) + + return Fixture(kind=kind, n=n, frame_paths=frame_paths, labels=labels, ffmpeg_commands=commands) + + +def fixture_prompt(kind: str, n: int) -> str: + """The question put to the model. Fixed text — changing it changes the benchmark.""" + common = ( + f"You are shown {n} images, in the order they were given to you. " + "They are consecutive frames of a video. " + "Answer with the ordered list only, comma-separated, no other words.\n" + ) + if kind == "color": + names = ", ".join(name for name, _ in COLOR_PALETTE) + return common + ( + "Each image is a single solid colour. " + f"List the colour of every image in order, first to last. " + f"Use only these colour names: {names}." + ) + if kind == "count": + return common + ( + "Each image shows some white squares on a black background. " + "List how many squares each image contains, in order, first to last." + ) + return common + ( + "Each image shows one white square that moves horizontally between frames. " + "Number the images 1 to N in the order given, then list those numbers sorted " + "by the square's horizontal position, leftmost first." + ) diff --git a/src/av/bench/frames.py b/src/av/bench/frames.py new file mode 100644 index 0000000..d458d8d --- /dev/null +++ b/src/av/bench/frames.py @@ -0,0 +1,133 @@ +"""Pinned frame extraction for benchmarking. + +``pipeline/ffmpeg.extract_frames`` is tuned for ingestion: it caps frame counts and +uses ingest defaults. Benchmarking needs the invocation itself to be part of the +record, so these helpers keep their command line fixed, return it verbatim for the +receipt, and never silently change sampling behind the caller's back. +""" + +from __future__ import annotations + +import subprocess +import tempfile +from dataclasses import dataclass, field +from pathlib import Path + +from av.core.exceptions import FFmpegError + +# Fixed encoder settings. Changing any of these changes every measurement, so they +# live here as constants rather than as call-site defaults. +JPEG_QUALITY = "2" +SCALE_WIDTH_DEFAULT = 768 + +# Real surveillance footage is frequently limited-range YUV, which the mjpeg encoder +# refuses outright. Normalising the pixel format keeps the pinned invocation working +# on ordinary CCTV files instead of failing on the footage that matters most. +PIXEL_FORMAT = "yuvj420p" + + +@dataclass +class FrameSet: + paths: list[Path] + timestamps: list[float] + commands: list[str] = field(default_factory=list) + interval_sec: float | None = None + + def __len__(self) -> int: + return len(self.paths) + + +def _run(cmd: list[str], timeout: int = 600) -> None: + try: + subprocess.run(cmd, capture_output=True, text=True, check=True, timeout=timeout) + except FileNotFoundError: + raise FFmpegError("ffmpeg not found. Install ffmpeg: brew install ffmpeg", cmd=" ".join(cmd)) + except subprocess.CalledProcessError as e: + raise FFmpegError( + f"Frame extraction failed: {e.stderr}", cmd=" ".join(cmd), returncode=e.returncode + ) + except subprocess.TimeoutExpired: + raise FFmpegError("Frame extraction timed out", cmd=" ".join(cmd)) + + +def _vf(interval_sec: float, scale_width: int | None) -> str: + parts = [f"fps=1/{interval_sec}"] + if scale_width: + parts.append(f"scale={scale_width}:-2") + parts.append(f"format={PIXEL_FORMAT}") + return ",".join(parts) + + +def sample_interval( + video_path: Path, + interval_sec: float, + *, + start_sec: float = 0.0, + duration_sec: float | None = None, + max_frames: int = 1024, + scale_width: int | None = SCALE_WIDTH_DEFAULT, + out_dir: Path | None = None, +) -> FrameSet: + """Sample one frame every ``interval_sec`` seconds. + + ``max_frames`` is a hard ceiling, not a target: a request that would exceed it is + truncated and the caller is expected to record the truncation. + """ + if interval_sec <= 0: + raise ValueError("interval_sec must be positive") + out_dir = out_dir or Path(tempfile.mkdtemp(prefix="av_bench_frames_")) + out_dir.mkdir(parents=True, exist_ok=True) + + cmd = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-nostdin", "-y"] + if start_sec > 0: + cmd += ["-ss", f"{start_sec:.3f}"] + cmd += ["-i", str(video_path)] + if duration_sec is not None: + cmd += ["-t", f"{duration_sec:.3f}"] + cmd += [ + "-vf", _vf(interval_sec, scale_width), + "-frames:v", str(max_frames), + "-q:v", JPEG_QUALITY, + str(out_dir / "f_%06d.jpg"), + ] + _run(cmd) + + paths = sorted(out_dir.glob("f_*.jpg")) + timestamps = [start_sec + i * interval_sec for i in range(len(paths))] + return FrameSet(paths=paths, timestamps=timestamps, commands=[" ".join(cmd)], interval_sec=interval_sec) + + +def sample_at( + video_path: Path, + timestamps: list[float], + *, + scale_width: int | None = SCALE_WIDTH_DEFAULT, + out_dir: Path | None = None, +) -> FrameSet: + """Grab one frame at each requested timestamp — the agentic arm's fetch step.""" + out_dir = out_dir or Path(tempfile.mkdtemp(prefix="av_bench_targeted_")) + out_dir.mkdir(parents=True, exist_ok=True) + + paths: list[Path] = [] + kept: list[float] = [] + commands: list[str] = [] + for i, ts in enumerate(sorted(set(round(t, 3) for t in timestamps))): + out_path = out_dir / f"t_{i:04d}_{int(ts * 1000):09d}.jpg" + cmd = [ + "ffmpeg", "-hide_banner", "-loglevel", "error", "-nostdin", "-y", + "-ss", f"{max(ts, 0.0):.3f}", "-i", str(video_path), + "-frames:v", "1", "-q:v", JPEG_QUALITY, + ] + vf = [f"scale={scale_width}:-2"] if scale_width else [] + vf.append(f"format={PIXEL_FORMAT}") + cmd += ["-vf", ",".join(vf), str(out_path)] + try: + _run(cmd, timeout=120) + except FFmpegError: + continue # a seek past the end is not a harness failure + if out_path.exists() and out_path.stat().st_size > 0: + paths.append(out_path) + kept.append(ts) + commands.append(" ".join(cmd)) + + return FrameSet(paths=paths, timestamps=kept, commands=commands, interval_sec=None) diff --git a/src/av/bench/receipts.py b/src/av/bench/receipts.py new file mode 100644 index 0000000..f9bdb23 --- /dev/null +++ b/src/av/bench/receipts.py @@ -0,0 +1,146 @@ +"""Receipts — every benchmark number traces back to one of these files. + +A receipt records what was run, against which provider, under which cost model, +and with which determinism controls. Claims carried in a receipt are explicitly +labelled so a reader never has to guess whether a figure was measured here, +computed from other figures, taken from someone else's write-up, or not tested. +""" + +from __future__ import annotations + +import hashlib +import json +import platform +import subprocess +import sys +import uuid +from dataclasses import asdict, dataclass, field +from datetime import datetime, timezone +from pathlib import Path +from urllib.parse import urlparse + +# Evidence labels. Anything reported must carry one of these. +MEASURED = "measured" # produced by this run, on this machine +DERIVED = "derived" # arithmetic over measured or documented figures +DOCUMENTED = "documented" # stated by the vendor's own docs +COMMUNITY = "community-reported" # someone else's number, reproduced verbatim +UNTESTED = "untested" # asserted nowhere; we did not check + +EVIDENCE_LABELS = (MEASURED, DERIVED, DOCUMENTED, COMMUNITY, UNTESTED) + +RECEIPT_VERSION = 2 + + +@dataclass +class Claim: + """One labelled statement. ``source`` is required for non-measured labels.""" + + label: str + statement: str + source: str | None = None + + def __post_init__(self) -> None: + if self.label not in EVIDENCE_LABELS: + raise ValueError(f"unknown evidence label {self.label!r}; expected one of {EVIDENCE_LABELS}") + if self.label in (COMMUNITY, DOCUMENTED) and not self.source: + raise ValueError(f"{self.label} claims must cite a source: {self.statement!r}") + + +def redact_endpoint(base_url: str | None) -> str | None: + """Reduce a base URL to its host. + + Receipts are meant to be published. A private or tunnelled inference endpoint + is not something a public artifact should carry, so only the host survives, + and hosts that are plainly private are replaced with a placeholder. + """ + if not base_url: + return None + try: + host = urlparse(base_url).hostname or "" + except ValueError: + return "" + if not host: + return "" + private_markers = (".ts.net", ".internal", ".local", ".lan") + if ( + host in ("localhost", "127.0.0.1", "::1") + or host.startswith(("10.", "192.168.", "172.16.", "100.")) + or host.endswith(private_markers) + ): + return "" + return host + + +def sha256_file(path: Path) -> str: + h = hashlib.sha256() + with open(path, "rb") as f: + for block in iter(lambda: f.read(1 << 20), b""): + h.update(block) + return h.hexdigest() + + +def sha256_text(text: str) -> str: + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _ffmpeg_version() -> str | None: + try: + out = subprocess.run( + ["ffmpeg", "-version"], capture_output=True, text=True, timeout=15 + ).stdout + except Exception: + return None + return out.splitlines()[0].strip() if out else None + + +@dataclass +class Receipt: + kind: str + provider: dict = field(default_factory=dict) + determinism: dict = field(default_factory=dict) + cost_model: dict | None = None + cells: list[dict] = field(default_factory=list) + claims: list[Claim] = field(default_factory=list) + summary: dict = field(default_factory=dict) + notes: list[str] = field(default_factory=list) + + run_id: str = field(default_factory=lambda: str(uuid.uuid4())) + created_utc: str = field( + default_factory=lambda: datetime.now(timezone.utc).isoformat(timespec="seconds") + ) + + def add_claim(self, label: str, statement: str, source: str | None = None) -> None: + self.claims.append(Claim(label=label, statement=statement, source=source)) + + def to_dict(self) -> dict: + from av import __version__ + + return { + "receipt_version": RECEIPT_VERSION, + "run_id": self.run_id, + "created_utc": self.created_utc, + "kind": self.kind, + "av_version": __version__, + "environment": { + "python": sys.version.split()[0], + "platform": platform.platform(), + "ffmpeg": _ffmpeg_version(), + }, + "provider": self.provider, + "determinism": self.determinism, + "cost_model": self.cost_model, + "summary": self.summary, + "cells": self.cells, + "claims": [asdict(c) for c in self.claims], + "notes": self.notes, + } + + +def write_receipt(receipt: Receipt, out_dir: Path) -> Path: + """Write a receipt to ``out_dir`` and return its path.""" + out_dir = Path(out_dir).expanduser() + out_dir.mkdir(parents=True, exist_ok=True) + stamp = receipt.created_utc.replace(":", "").replace("-", "") + path = out_dir / f"{receipt.kind}-{stamp}-{receipt.run_id[:8]}.json" + path.write_text(json.dumps(receipt.to_dict(), indent=2, default=str) + "\n") + return path diff --git a/src/av/bench/runner.py b/src/av/bench/runner.py new file mode 100644 index 0000000..37a5023 --- /dev/null +++ b/src/av/bench/runner.py @@ -0,0 +1,151 @@ +"""Sweep axes, noise floor, and the frontier summary. + +The noise floor exists because a benchmark that publishes a single run invites +readers to over-read small deltas. Repeating one unchanged cell and publishing the +spread is what makes a difference legible: any gap smaller than the spread is noise, +and the harness says so rather than leaving the reader to work it out. +""" + +from __future__ import annotations + +import statistics +from dataclasses import dataclass +from typing import Callable + +# The cost/accuracy frontier axis, in seconds between sampled frames. +# 1 fps is the dense baseline that published agentic-video comparisons use. +DEFAULT_FRAME_INTERVALS: tuple[float, ...] = (1.0, 2.0, 5.0, 10.0, 30.0, 60.0) + +DEFAULT_NOISE_REPEATS = 5 + + +@dataclass +class Spread: + values: list[float] + unit: str = "" + + @property + def n(self) -> int: + return len(self.values) + + @property + def minimum(self) -> float | None: + return min(self.values) if self.values else None + + @property + def maximum(self) -> float | None: + return max(self.values) if self.values else None + + @property + def mean(self) -> float | None: + return statistics.fmean(self.values) if self.values else None + + @property + def stdev(self) -> float | None: + return statistics.stdev(self.values) if len(self.values) > 1 else 0.0 + + @property + def range(self) -> float | None: + if not self.values: + return None + return max(self.values) - min(self.values) + + def to_dict(self) -> dict: + return { + "n": self.n, + "unit": self.unit, + "values": [round(v, 6) for v in self.values], + "min": self.minimum, + "max": self.maximum, + "mean": round(self.mean, 6) if self.mean is not None else None, + "stdev": round(self.stdev, 6) if self.stdev is not None else None, + "range": round(self.range, 6) if self.range is not None else None, + } + + +def noise_floor( + run_once: Callable[[int], float | None], + *, + repeats: int = DEFAULT_NOISE_REPEATS, + unit: str = "", + on_run: Callable[[int, float | None], None] | None = None, +) -> Spread: + """Run one unchanged cell ``repeats`` times and collect the spread. + + ``run_once`` receives the zero-based iteration index and returns the metric being + watched, or ``None`` if that run failed. Failed runs are dropped from the spread + and the reduced ``n`` makes the omission visible. + """ + values: list[float] = [] + for i in range(repeats): + value = run_once(i) + if on_run: + on_run(i, value) + if value is not None: + values.append(float(value)) + return Spread(values=values, unit=unit) + + +def is_saturated(spread: Spread, ceiling: float = 1.0) -> bool: + """True when every run hit the ceiling, so the spread says nothing about variance. + + A cell the model solves perfectly every time has no headroom to vary, and quoting + its zero spread as *the* noise floor understates variance everywhere else. The + floor has to be measured on a cell that is actually contested. + """ + return bool(spread.values) and spread.range == 0.0 and spread.values[0] >= ceiling + + +def interpret_delta(delta: float, spread: Spread) -> str: + """Classify an observed difference against the measured noise floor.""" + if spread.range is None or spread.n < 2: + return "no noise floor measured — treat any delta as unverified" + if is_saturated(spread): + return ( + "every run hit the ceiling, so this spread measures nothing. Re-measure the " + "noise floor on a cell the model does not solve perfectly." + ) + if abs(delta) <= spread.range: + return ( + f"within the noise floor (spread {spread.range:.4g}{spread.unit} over " + f"{spread.n} identical runs) — not a real difference" + ) + return ( + f"exceeds the noise floor (spread {spread.range:.4g}{spread.unit} over " + f"{spread.n} identical runs)" + ) + + +def collapse_point( + rows: list[dict], + *, + score_key: str = "score", + tolerance: float = 0.1, +) -> dict: + """Find the widest interval whose score is still within ``tolerance`` of the densest. + + Returns the safe interval and the one after it, so a reader sees both the + recommendation and the evidence for where it stops being safe. + """ + scored = [r for r in sorted(rows, key=lambda c: c.get("interval_sec") or 0.0) + if r.get(score_key) is not None] + if not scored: + return {"safe_interval_sec": None, "reason": "no scored cells"} + + baseline = scored[0][score_key] + safe = scored[0] + first_collapse = None + for row in scored[1:]: + if baseline - row[score_key] <= tolerance: + safe = row + elif first_collapse is None: + first_collapse = row + return { + "baseline_interval_sec": scored[0].get("interval_sec"), + "baseline_score": baseline, + "safe_interval_sec": safe.get("interval_sec"), + "safe_score": safe.get(score_key), + "collapse_interval_sec": first_collapse.get("interval_sec") if first_collapse else None, + "collapse_score": first_collapse.get(score_key) if first_collapse else None, + "tolerance": tolerance, + } diff --git a/src/av/bench/tasks/__init__.py b/src/av/bench/tasks/__init__.py new file mode 100644 index 0000000..1d33b28 --- /dev/null +++ b/src/av/bench/tasks/__init__.py @@ -0,0 +1,3 @@ +"""Benchmark task families.""" + +from __future__ import annotations diff --git a/src/av/bench/tasks/events.py b/src/av/bench/tasks/events.py new file mode 100644 index 0000000..e1dfc0f --- /dev/null +++ b/src/av/bench/tasks/events.py @@ -0,0 +1,261 @@ +"""Event detection on real footage, as a function of frame interval. + +This is where the frontier stops being an abstraction: the interval at which recall +collapses is the cheapest sampling rate that is still safe for the task, and it is +different for every task. Smoke tolerates sparse frames; a door opening does not. + +**Reference caveat, stated up front.** The ``samples/epstein-cctv`` artifacts shipped +with this repository were themselves produced by a vision model. Scoring against them +measures *agreement with a dense reference run*, not agreement with human ground truth. +Every receipt this module writes carries that caveat, and the metric is named +``reference_recall`` rather than ``recall`` so nobody can quote it as the latter. +""" + +from __future__ import annotations + +import json +import re +import shutil +import tempfile +from dataclasses import dataclass, field +from pathlib import Path + +from av.bench.frames import sample_interval +from av.bench.vlm import BenchVLM, strip_code_fence +from av.core.exceptions import FFmpegError + +# Event probes. Each is a label, the phrases that mark it in a reference caption, and +# the question put to the model. Keep the wording of both sides fixed: changing either +# changes the benchmark. +EVENT_PROBES: dict[str, dict] = { + "door_activity": { + "reference_terms": ["door", "gate", "doorway"], + "question": "Does anyone open, close, or interact with a door or gate in these frames?", + }, + "person_enters": { + "reference_terms": ["enters", "entering", "walks in", "arrives"], + "question": "Does any person enter the scene in these frames?", + }, + "person_exits": { + "reference_terms": ["exits", "exiting", "leaves", "walks out", "departs"], + "question": "Does any person leave or exit the scene in these frames?", + }, + "group_present": { + "reference_terms": ["group", "several people", "multiple people", "three ", "four "], + "question": "Are three or more people visible together in these frames?", + }, + "escort": { + "reference_terms": ["escort", "escorted", "restrain", "handcuff"], + "question": "Is anyone being escorted, guided, or restrained by another person in these frames?", + }, +} + +ANSWER_INSTRUCTION = 'Reply with JSON only: {"present": true} or {"present": false}' + + +@dataclass +class ReferenceEvent: + video: Path + start_sec: float + end_sec: float + probe: str + source_text: str + + +def load_reference_events( + artifacts_path: Path, + video_dir: Path, + *, + probes: list[str] | None = None, + max_per_probe: int = 10, +) -> list[ReferenceEvent]: + """Derive event windows from a shipped artifacts JSONL. + + A window counts as containing an event when its caption text mentions one of the + probe's reference terms. Windows whose video file is missing are skipped, since a + benchmark that scores questions it cannot show frames for is measuring nothing. + """ + wanted = probes or list(EVENT_PROBES) + video_dir = Path(video_dir).expanduser() + counts = {p: 0 for p in wanted} + events: list[ReferenceEvent] = [] + + for line in Path(artifacts_path).expanduser().read_text().splitlines(): + line = line.strip() + if not line: + continue + try: + row = json.loads(line) + except json.JSONDecodeError: + continue + text = str(row.get("text") or "") + if not text or row.get("end_sec") in (None, ""): + continue + video = video_dir / str(row.get("filename") or "") + if not video.exists(): + continue + low = text.lower() + for probe in wanted: + if counts[probe] >= max_per_probe: + continue + terms = EVENT_PROBES[probe]["reference_terms"] + if any(t in low for t in terms): + events.append( + ReferenceEvent( + video=video, + start_sec=float(row["start_sec"]), + end_sec=float(row["end_sec"]), + probe=probe, + source_text=text, + ) + ) + counts[probe] += 1 + return events + + +_TRUE = re.compile(r"\btrue\b|\byes\b", re.I) +_FALSE = re.compile(r"\bfalse\b|\bno\b", re.I) +# Matches the field even when a token limit cut the closing brace off. +_PRESENT_FIELD = re.compile(r'"present"\s*:\s*(true|false)', re.I) + + +def parse_presence(text: str) -> bool | None: + """Read a presence reply. ``None`` means unparseable, which is not a detection. + + Kept deliberately forgiving about formatting and strict about content: a reply + the harness cannot read is recorded separately from a reply that says "no", so a + parsing problem never masquerades as a detection failure. + """ + if not text: + return None + cleaned = strip_code_fence(text) + m = re.search(r"\{.*\}", cleaned, re.S) + if m: + try: + value = json.loads(m.group(0)).get("present") + if isinstance(value, bool): + return value + except json.JSONDecodeError: + pass + field = _PRESENT_FIELD.search(cleaned) + if field: + return field.group(1).lower() == "true" + if _TRUE.search(cleaned) and not _FALSE.search(cleaned): + return True + if _FALSE.search(cleaned): + return False + return None + + +@dataclass +class EventCell: + interval_sec: float + probe: str + events_total: int + events_detected: int + unparseable: int + frames_total: int + tokens_in: int + tokens_out: int + wall_sec: float + errors: list[str] = field(default_factory=list) + + @property + def reference_recall(self) -> float | None: + if not self.events_total: + return None + return self.events_detected / self.events_total + + def to_dict(self) -> dict: + return { + "interval_sec": self.interval_sec, + "probe": self.probe, + "events_total": self.events_total, + "events_detected": self.events_detected, + "reference_recall": ( + round(self.reference_recall, 4) if self.reference_recall is not None else None + ), + "unparseable": self.unparseable, + "frames_total": self.frames_total, + "tokens_in": self.tokens_in, + "tokens_out": self.tokens_out, + "wall_sec": round(self.wall_sec, 3), + "errors": self.errors[:5], + } + + +def run_event_cell( + vlm: BenchVLM, + events: list[ReferenceEvent], + interval_sec: float, + *, + probe: str, + max_frames: int = 64, + scale_width: int | None = 512, + on_event=None, +) -> EventCell: + """Ask the presence question over every reference window at one sampling interval.""" + subset = [e for e in events if e.probe == probe] + question = EVENT_PROBES[probe]["question"] + "\n\n" + ANSWER_INSTRUCTION + + detected = unparseable = frames_total = 0 + tokens_in = tokens_out = 0 + wall = 0.0 + errors: list[str] = [] + + for event in subset: + work = Path(tempfile.mkdtemp(prefix="av_bench_event_")) + try: + span = max(event.end_sec - event.start_sec, interval_sec) + try: + frames = sample_interval( + event.video, interval_sec, + start_sec=event.start_sec, duration_sec=span, + max_frames=max_frames, scale_width=scale_width, out_dir=work, + ) + except FFmpegError as e: + # A window we cannot decode is a gap in coverage, not a reason to + # abandon the sweep. It is recorded and the cell continues. + errors.append(f"{event.video.name}@{event.start_sec:.0f}s: {e}") + continue + if not frames.paths: + errors.append(f"no frames at {event.start_sec:.0f}s in {event.video.name}") + continue + frames_total += len(frames.paths) + res = vlm.ask(frames.paths, question) + wall += res.wall_sec + tokens_in += res.tokens_in or 0 + tokens_out += res.tokens_out or 0 + if not res.ok: + errors.append(res.error or "unknown error") + unparseable += 1 + continue + verdict = parse_presence(res.text) + if verdict is True: + detected += 1 + elif verdict is None: + unparseable += 1 + if on_event: + on_event(event, verdict, len(frames.paths)) + finally: + shutil.rmtree(work, ignore_errors=True) + + return EventCell( + interval_sec=interval_sec, + probe=probe, + events_total=len(subset), + events_detected=detected, + unparseable=unparseable, + frames_total=frames_total, + tokens_in=tokens_in, + tokens_out=tokens_out, + wall_sec=wall, + errors=errors, + ) + + +REFERENCE_CAVEAT = ( + "Reference windows come from a dense vision-model run shipped with this repository, " + "not from human annotation. This metric measures agreement with that reference run " + "and must not be quoted as recall against ground truth." +) diff --git a/src/av/bench/tasks/ordering.py b/src/av/bench/tasks/ordering.py new file mode 100644 index 0000000..201061e --- /dev/null +++ b/src/av/bench/tasks/ordering.py @@ -0,0 +1,168 @@ +"""Temporal-ordering capability gate. + +This is not a headline benchmark. It is the check that runs first, because a model +that cannot report the order of a handful of images cannot be meaningfully scored +on long-video reasoning — any accuracy it posts there would be measuring something +other than temporal understanding. + +The gate is deliberately cheap: a few frames of flat colour, one question, exact +string comparison against ground truth generated by ffmpeg. It costs cents and can +save the rest of the sweep. +""" + +from __future__ import annotations + +import re +from dataclasses import dataclass, field +from pathlib import Path + +from av.bench.fixtures import Fixture, fixture_prompt, generate_fixture +from av.bench.receipts import sha256_file +from av.bench.vlm import BenchVLM + +_PUNCT = re.compile(r"[^a-z0-9]+") + + +def normalise_answer(text: str, kind: str) -> list[str]: + """Turn free-form model output into a comparable ordered list.""" + if not text: + return [] + # Strip common wrappers before splitting so "Answer: red, blue" parses. + text = re.sub(r"(?is)^\s*(the\s+)?(answer|order|sequence)\s*(is)?\s*[:\-]\s*", "", text.strip()) + text = text.strip().strip("`").strip() + + if kind in ("count", "motion"): + return re.findall(r"\d+", text) + + parts = re.split(r"[,\n;]+|\s+->\s+|\s+then\s+", text) + out: list[str] = [] + for part in parts: + token = _PUNCT.sub("", part.strip().lower()) + token = re.sub(r"^\d+", "", token) # drop list numbering like "1red" + if token: + out.append(token) + return out + + +@dataclass +class OrderingCell: + n: int + kind: str + expected: list[str] + reported: list[str] + exact_order: bool + correct_prefix: int + n_reported: int + tokens_in: int | None + tokens_out: int | None + ttft_sec: float | None + wall_sec: float + ok: bool + error: str | None = None + multi_image_unsupported: bool = False + raw_text: str = "" + frame_sha256: list[str] = field(default_factory=list) + + def to_dict(self) -> dict: + return { + "n": self.n, + "fixture_kind": self.kind, + "expected": self.expected, + "reported": self.reported, + "exact_order": self.exact_order, + "correct_prefix": self.correct_prefix, + "n_reported": self.n_reported, + "tokens_in": self.tokens_in, + "tokens_out": self.tokens_out, + "ttft_sec": round(self.ttft_sec, 4) if self.ttft_sec is not None else None, + "wall_sec": round(self.wall_sec, 4), + "ok": self.ok, + "error": self.error, + "multi_image_unsupported": self.multi_image_unsupported, + "raw_text": self.raw_text[:2000], + "frame_sha256": self.frame_sha256, + } + + +def score_ordering(expected: list[str], reported: list[str]) -> tuple[bool, int]: + """Exact-order match, plus how many leading items were right. + + ``correct_prefix`` is what distinguishes "wrong order" from "saw one frame and + stopped" — the failure mode reported upstream for a six-image colour fixture. + """ + prefix = 0 + for e, r in zip(expected, reported): + if e != r: + break + prefix += 1 + return (expected == reported), prefix + + +def run_ordering_cell(vlm: BenchVLM, fixture: Fixture) -> OrderingCell: + prompt = fixture_prompt(fixture.kind, fixture.n) + result = vlm.ask(fixture.frame_paths, prompt) + reported = normalise_answer(result.text, fixture.kind) + expected = fixture.expected_answer + exact, prefix = score_ordering(expected, reported) + return OrderingCell( + n=fixture.n, + kind=fixture.kind, + expected=expected, + reported=reported, + exact_order=exact and result.ok, + correct_prefix=prefix, + n_reported=len(reported), + tokens_in=result.tokens_in, + tokens_out=result.tokens_out, + ttft_sec=result.ttft_sec, + wall_sec=result.wall_sec, + ok=result.ok, + error=result.error, + multi_image_unsupported=result.multi_image_unsupported, + raw_text=result.text, + frame_sha256=[sha256_file(p) for p in fixture.frame_paths], + ) + + +def run_gate( + vlm: BenchVLM, + *, + kind: str = "color", + sizes: tuple[int, ...] = (2, 4, 8), + work_dir: Path, + on_cell=None, +) -> tuple[list[OrderingCell], dict]: + """Run the gate across fixture sizes. Returns cells and a verdict summary.""" + cells: list[OrderingCell] = [] + fixtures: list[Fixture] = [] + for n in sizes: + fixture = generate_fixture(kind, n, work_dir) + fixtures.append(fixture) + cell = run_ordering_cell(vlm, fixture) + cells.append(cell) + if on_cell: + on_cell(cell) + + passed = [c.n for c in cells if c.exact_order] + failed = [c.n for c in cells if not c.exact_order] + largest_pass = max(passed) if passed else None + + if any(c.multi_image_unsupported for c in cells): + verdict = "no_multi_image" + elif not passed: + verdict = "fail" + elif not failed: + verdict = "pass" + else: + verdict = "partial" + + summary = { + "fixture_kind": kind, + "sizes": list(sizes), + "passed_sizes": passed, + "failed_sizes": failed, + "largest_passing_n": largest_pass, + "verdict": verdict, + "ffmpeg_commands": [c for f in fixtures for c in f.ffmpeg_commands], + } + return cells, summary diff --git a/src/av/bench/tasks/videoqa.py b/src/av/bench/tasks/videoqa.py new file mode 100644 index 0000000..0a4f80a --- /dev/null +++ b/src/av/bench/tasks/videoqa.py @@ -0,0 +1,393 @@ +"""Video question answering under two arms: dense and agentic. + +The comparison mirrors the shape of published agentic-video results so the numbers +are readable side by side: + +``dense`` every frame at a fixed rate is stuffed into one request. Simple, + expensive, and the baseline everyone reports. +``agentic`` a cheap coarse pass over widely-spaced frames decides *where to look*, + then a second request fetches only those moments at full rate. The + scouting pass is charged to the arm — its tokens count. + +Both arms answer the same questions with the same prompt and the same scorer, so +the only difference between them is which frames the model got to see. +""" + +from __future__ import annotations + +import json +import re +import shutil +import tempfile +from dataclasses import dataclass, field +from pathlib import Path + +from av.bench.cost import CellUsage +from av.bench.frames import sample_at, sample_interval +from av.bench.vlm import BenchVLM, VLMResult, strip_code_fence + +ARMS = ("dense", "agentic") + +# Frame budget for the agentic arm's targeted fetch, per question. +DEFAULT_AGENTIC_BUDGET = 16 +# Coarse pass spacing: one frame every N seconds. Deliberately sparse — the point +# is that a cheap look is enough to decide where the answer lives. +DEFAULT_COARSE_INTERVAL_SEC = 60.0 +DEFAULT_COARSE_MAX_FRAMES = 32 + + +@dataclass +class Question: + id: str + video: Path + question: str + options: list[str] = field(default_factory=list) + answer: str = "" + start_sec: float = 0.0 + end_sec: float | None = None + meta: dict = field(default_factory=dict) + + @property + def is_multiple_choice(self) -> bool: + return bool(self.options) + + +def load_questions(path: Path, video_root: Path | None = None) -> list[Question]: + """Load a JSONL task file. + + One object per line:: + + {"id": "q1", "video": "clips/a.mp4", "question": "...", + "options": ["A. ...", "B. ..."], "answer": "B", + "start_sec": 0, "end_sec": 600} + + ``options`` may be omitted for open-ended questions. Relative ``video`` paths + resolve against ``video_root`` (default: the task file's directory), which keeps + a task file portable across machines. + """ + path = Path(path).expanduser() + root = Path(video_root).expanduser() if video_root else path.parent + questions: list[Question] = [] + for lineno, line in enumerate(path.read_text().splitlines(), start=1): + line = line.strip() + if not line or line.startswith("#"): + continue + try: + row = json.loads(line) + except json.JSONDecodeError as e: + raise ValueError(f"{path}:{lineno}: invalid JSON — {e}") from e + + video = Path(row["video"]).expanduser() + if not video.is_absolute(): + video = (root / video).resolve() + + questions.append( + Question( + id=str(row.get("id") or f"q{lineno}"), + video=video, + question=row["question"], + options=list(row.get("options") or []), + answer=str(row.get("answer", "")), + start_sec=float(row.get("start_sec") or 0.0), + end_sec=float(row["end_sec"]) if row.get("end_sec") is not None else None, + meta=row.get("meta") or {}, + ) + ) + return questions + + +# --- prompting --------------------------------------------------------------- + +_ANSWER_INSTRUCTION_MC = ( + "Answer with the letter of the correct option only — a single character, " + "no explanation, no punctuation." +) +_ANSWER_INSTRUCTION_OPEN = "Answer in as few words as possible. No explanation." + + +def build_question_prompt(q: Question, frame_timestamps: list[float]) -> str: + stamps = ", ".join(f"{t:.1f}s" for t in frame_timestamps) + header = ( + f"You are shown {len(frame_timestamps)} frames sampled from a video, in " + f"chronological order, at these timestamps: {stamps}.\n\n" + ) + body = f"Question: {q.question}\n" + if q.is_multiple_choice: + body += "Options:\n" + "\n".join(q.options) + "\n\n" + _ANSWER_INSTRUCTION_MC + else: + body += "\n" + _ANSWER_INSTRUCTION_OPEN + return header + body + + +_SELECTION_INSTRUCTION = ( + "These frames are a coarse, widely-spaced preview of a longer video.\n\n" + "Question you will later have to answer: {question}\n\n" + "Do not answer it yet. Decide which moments of the video you need to see at a " + "finer sampling rate to answer it confidently.\n" + "Reply with JSON only, in exactly this form, and nothing else:\n" + '{{"timestamps": [12.0, 13.5, 40.0]}}\n' + "Give at most {budget} timestamps, in seconds, within 0 and {duration:.1f}." +) + + +def build_selection_prompt(q: Question, frame_timestamps: list[float], budget: int, duration: float) -> str: + stamps = ", ".join(f"{t:.1f}s" for t in frame_timestamps) + return ( + f"You are shown {len(frame_timestamps)} frames at these timestamps: {stamps}.\n\n" + + _SELECTION_INSTRUCTION.format(question=q.question, budget=budget, duration=duration) + ) + + +# --- scoring ----------------------------------------------------------------- + +_LETTER = re.compile(r"\b([A-H])\b") + + +def _norm(text: str) -> str: + return re.sub(r"[^a-z0-9 ]+", "", text.lower()).strip() + + +def extract_choice(text: str, options: list[str]) -> str | None: + """Pull a choice letter out of a model reply, tolerating light chattiness.""" + if not text: + return None + stripped = strip_code_fence(text).strip().strip(".").strip() + if len(stripped) == 1 and stripped.upper().isalpha(): + return stripped.upper() + + # Match the option text before the bare letter. A reply like "a red car" contains + # a standalone "a", and reading that as choice A would score a correct answer wrong. + norm_reply = _norm(stripped) + for i, opt in enumerate(options): + body = _norm(re.sub(r"^\s*[A-Ha-h][.)]\s*", "", opt)) + if len(body) > 1 and body in norm_reply: + return chr(ord("A") + i) + + m = _LETTER.search(stripped.upper()) + return m.group(1) if m else None + + +def score_answer(q: Question, text: str) -> tuple[bool, str | None]: + """Return (correct, parsed answer). Unparseable replies score as incorrect.""" + if q.is_multiple_choice: + choice = extract_choice(text, q.options) + return (choice is not None and choice == q.answer.strip().upper()), choice + parsed = text.strip() + gold = _norm(q.answer) + return (bool(gold) and gold in _norm(parsed)), parsed or None + + +def parse_timestamps(text: str, duration: float, budget: int) -> list[float]: + """Read the selection reply. Falls back to bare numbers if the JSON is malformed.""" + if not text: + return [] + text = strip_code_fence(text) + candidates: list[float] = [] + m = re.search(r"\{.*\}", text, re.S) + if m: + try: + data = json.loads(m.group(0)) + candidates = [float(t) for t in (data.get("timestamps") or [])] + except (json.JSONDecodeError, TypeError, ValueError): + candidates = [] + if not candidates: + candidates = [float(x) for x in re.findall(r"\d+(?:\.\d+)?", text)] + kept = [t for t in candidates if 0.0 <= t <= max(duration, 0.0)] + return sorted(set(kept))[:budget] + + +# --- arms -------------------------------------------------------------------- + +@dataclass +class QuestionResult: + question_id: str + arm: str + correct: bool + parsed_answer: str | None + gold_answer: str + frames_sent: int + frames_extracted: int + requests: int + tokens_in: int | None + tokens_out: int | None + ttft_sec: float | None + wall_sec: float + ok: bool + error: str | None = None + selected_timestamps: list[float] = field(default_factory=list) + raw_text: str = "" + + def to_dict(self) -> dict: + return { + "question_id": self.question_id, + "arm": self.arm, + "correct": self.correct, + "parsed_answer": self.parsed_answer, + "gold_answer": self.gold_answer, + "frames_sent": self.frames_sent, + "frames_extracted": self.frames_extracted, + "requests": self.requests, + "tokens_in": self.tokens_in, + "tokens_out": self.tokens_out, + "ttft_sec": round(self.ttft_sec, 4) if self.ttft_sec is not None else None, + "wall_sec": round(self.wall_sec, 4), + "ok": self.ok, + "error": self.error, + "selected_timestamps": self.selected_timestamps, + "raw_text": self.raw_text[:1000], + } + + +def window_for(q: Question, duration: float) -> tuple[float, float]: + start = max(q.start_sec, 0.0) + end = q.end_sec if q.end_sec is not None else duration + end = min(end, duration) if duration else end + return start, max(end - start, 0.0) + + +def _accumulate(results: list[VLMResult]) -> tuple[int | None, int | None, float | None, float]: + tin = sum(r.tokens_in for r in results if r.tokens_in is not None) or None + tout = sum(r.tokens_out for r in results if r.tokens_out is not None) or None + ttfts = [r.ttft_sec for r in results if r.ttft_sec is not None] + wall = sum(r.wall_sec for r in results) + return tin, tout, (ttfts[0] if ttfts else None), wall + + +def run_dense( + vlm: BenchVLM, + q: Question, + duration: float, + *, + interval_sec: float = 1.0, + max_frames: int = 1024, + scale_width: int | None = 768, +) -> QuestionResult: + """Sample the whole window at a fixed rate and ask once.""" + work = Path(tempfile.mkdtemp(prefix="av_bench_dense_")) + try: + start, span = window_for(q, duration) + frames = sample_interval( + q.video, interval_sec, + start_sec=start, duration_sec=span or None, + max_frames=max_frames, scale_width=scale_width, out_dir=work, + ) + if not frames.paths: + return QuestionResult( + question_id=q.id, arm="dense", correct=False, parsed_answer=None, + gold_answer=q.answer, frames_sent=0, frames_extracted=0, requests=0, + tokens_in=None, tokens_out=None, ttft_sec=None, wall_sec=0.0, + ok=False, error="no frames extracted", + ) + prompt = build_question_prompt(q, frames.timestamps) + res = vlm.ask(frames.paths, prompt) + correct, parsed = score_answer(q, res.text) if res.ok else (False, None) + tin, tout, ttft, wall = _accumulate([res]) + return QuestionResult( + question_id=q.id, arm="dense", correct=correct, parsed_answer=parsed, + gold_answer=q.answer, frames_sent=len(frames.paths), + frames_extracted=len(frames.paths), requests=1, + tokens_in=tin, tokens_out=tout, ttft_sec=ttft, wall_sec=wall, + ok=res.ok, error=res.error, raw_text=res.text, + ) + finally: + shutil.rmtree(work, ignore_errors=True) + + +def run_agentic( + vlm: BenchVLM, + q: Question, + duration: float, + *, + coarse_interval_sec: float = DEFAULT_COARSE_INTERVAL_SEC, + coarse_max_frames: int = DEFAULT_COARSE_MAX_FRAMES, + budget_frames: int = DEFAULT_AGENTIC_BUDGET, + scale_width: int | None = 768, + seed_timestamps: list[float] | None = None, +) -> QuestionResult: + """Coarse look, then targeted fetch. Both requests are charged to this arm. + + ``seed_timestamps`` lets an external retriever (for example ``av``'s FTS5 search + over already-ingested captions) propose moments before the model looks at all. + They are merged with the model's own selection rather than replacing it. + """ + work = Path(tempfile.mkdtemp(prefix="av_bench_agentic_")) + try: + start, span = window_for(q, duration) + coarse = sample_interval( + q.video, coarse_interval_sec, + start_sec=start, duration_sec=span or None, + max_frames=coarse_max_frames, scale_width=scale_width, + out_dir=work / "coarse", + ) + if not coarse.paths: + return QuestionResult( + question_id=q.id, arm="agentic", correct=False, parsed_answer=None, + gold_answer=q.answer, frames_sent=0, frames_extracted=0, requests=0, + tokens_in=None, tokens_out=None, ttft_sec=None, wall_sec=0.0, + ok=False, error="no coarse frames extracted", + ) + + sel_prompt = build_selection_prompt(q, coarse.timestamps, budget_frames, start + (span or duration)) + sel = vlm.ask(coarse.paths, sel_prompt) + chosen = parse_timestamps(sel.text, start + (span or duration), budget_frames) if sel.ok else [] + if seed_timestamps: + chosen = sorted(set(chosen) | set(seed_timestamps))[:budget_frames] + if not chosen: + # The model declined to choose. Falling back to the coarse frames keeps + # the arm answerable, and the empty selection is recorded either way. + chosen = list(coarse.timestamps[:budget_frames]) + + targeted = sample_at(q.video, chosen, scale_width=scale_width, out_dir=work / "targeted") + if not targeted.paths: + targeted = coarse + + ans_prompt = build_question_prompt(q, targeted.timestamps) + ans = vlm.ask(targeted.paths, ans_prompt) + correct, parsed = score_answer(q, ans.text) if ans.ok else (False, None) + tin, tout, ttft, wall = _accumulate([sel, ans]) + return QuestionResult( + question_id=q.id, arm="agentic", correct=correct, parsed_answer=parsed, + gold_answer=q.answer, + frames_sent=len(coarse.paths) + len(targeted.paths), + frames_extracted=len(coarse.paths) + len(targeted.paths), + requests=2, + tokens_in=tin, tokens_out=tout, ttft_sec=ttft, wall_sec=wall, + ok=sel.ok and ans.ok, error=ans.error or sel.error, + selected_timestamps=chosen, raw_text=ans.text, + ) + finally: + shutil.rmtree(work, ignore_errors=True) + + +def aggregate(results: list[QuestionResult]) -> dict: + """Arm-level summary: the two headline axes plus what they were computed over.""" + answered = [r for r in results if r.ok] + n = len(results) + correct = sum(1 for r in results if r.correct) + with_tokens = [r for r in results if r.tokens_in is not None] + total_in = sum(r.tokens_in or 0 for r in with_tokens) + total_out = sum(r.tokens_out or 0 for r in with_tokens) + return { + "questions": n, + "answered_ok": len(answered), + "correct": correct, + "accuracy": round(correct / n, 4) if n else None, + "tokens_per_query_in": round(total_in / len(with_tokens), 1) if with_tokens else None, + "tokens_per_query_out": round(total_out / len(with_tokens), 1) if with_tokens else None, + "tokens_per_query_total": ( + round((total_in + total_out) / len(with_tokens), 1) if with_tokens else None + ), + "tokens_reported_for": len(with_tokens), + "frames_per_query": round(sum(r.frames_sent for r in results) / n, 1) if n else None, + "wall_sec_total": round(sum(r.wall_sec for r in results), 3), + } + + +def usage_for(results: list[QuestionResult]) -> CellUsage: + return CellUsage( + tokens_in=sum(r.tokens_in or 0 for r in results), + tokens_out=sum(r.tokens_out or 0 for r in results), + wall_total_sec=sum(r.wall_sec for r in results), + ttft_sec=next((r.ttft_sec for r in results if r.ttft_sec is not None), None), + requests=sum(r.requests for r in results), + ) diff --git a/src/av/bench/vlm.py b/src/av/bench/vlm.py new file mode 100644 index 0000000..beb7ec2 --- /dev/null +++ b/src/av/bench/vlm.py @@ -0,0 +1,367 @@ +"""Provider-agnostic multi-image calls with usage accounting. + +The benchmark's headline axis is tokens per query, so every call here reports the +token usage the provider itself returned. Nothing is estimated: when a provider +omits ``usage``, the receipt records ``null`` rather than a guess. + +Time to first token is captured by streaming. It is the closest measurable proxy +for prefill on a chat-completions API and is labelled as a proxy everywhere it +appears — it is not a prefill measurement. +""" + +from __future__ import annotations + +import base64 +import time +from dataclasses import dataclass, field +from pathlib import Path + +from av.core.config import AVConfig +from av.providers.openai import _client + +# Providers differ on whether an unsupported request field is ignored or rejected. +# These fragments mark a rejection we should retry without the field. +_UNSUPPORTED_MARKERS = ( + "unknown field", + "unknown name", + "unrecognized", + "unexpected keyword", + "not supported", + "unsupported", + "invalid_request_error", + "invalid_argument", + "extra fields not permitted", + "does not support", +) + +# Fragments that mark the provider refusing multi-image input outright. That is a +# capability result, not a bug in the harness. +_MULTI_IMAGE_MARKERS = ( + "only one image", + "single image", + "at most 1 image", + "too many images", + "image count", + "multiple images", +) + + +def strip_code_fence(text: str) -> str: + """Drop a markdown code fence around a reply. + + Models routinely wrap JSON in ```json fences, and a fence that gets truncated by a + token limit leaves the payload unterminated. Both are formatting, not content, so + parsers strip them before trying to read an answer. + """ + if not text: + return "" + cleaned = text.strip() + if cleaned.startswith("```"): + cleaned = cleaned.split("\n", 1)[1] if "\n" in cleaned else cleaned[3:] + if cleaned.endswith("```"): + cleaned = cleaned[:-3] + return cleaned.strip() + + +def encode_image(path: Path) -> str: + """Base64 data URL for an image file, in the OpenAI ``image_url`` format.""" + ext = path.suffix.lstrip(".").lower() + if ext == "jpg": + ext = "jpeg" + data = base64.b64encode(path.read_bytes()).decode() + return f"data:image/{ext};base64,{data}" + + +@dataclass +class VLMResult: + text: str = "" + tokens_in: int | None = None + tokens_out: int | None = None + ttft_sec: float | None = None + wall_sec: float = 0.0 + ok: bool = True + error: str | None = None + # True when the provider refused the request because of the image count. + multi_image_unsupported: bool = False + detail_accepted: bool | None = None + # False when the provider rejected `seed` and the call was retried without it. + seed_accepted: bool = True + raw_usage: dict = field(default_factory=dict) + + @property + def tokens_total(self) -> int | None: + if self.tokens_in is None and self.tokens_out is None: + return None + return (self.tokens_in or 0) + (self.tokens_out or 0) + + +class BenchVLM: + """A thin, deterministic wrapper over any OpenAI-compatible chat endpoint.""" + + def __init__( + self, + config: AVConfig, + *, + model: str | None = None, + temperature: float = 0.0, + seed: int | None = 0, + detail: str | None = None, + max_tokens: int = 1024, + stream: bool = True, + timeout: float = 600.0, + ) -> None: + self.config = config + self.model = model or config.vision_model + self.temperature = temperature + self.seed = seed + self.detail = detail + self.max_tokens = max_tokens + self.stream = stream + self.timeout = timeout + self.client = _client(config) + + # -- request construction ------------------------------------------------- + + def _content(self, images: list[Path], prompt: str) -> list[dict]: + content: list[dict] = [{"type": "text", "text": prompt}] + for img in images: + image_url: dict = {"url": encode_image(img)} + if self.detail: + image_url["detail"] = self.detail + content.append({"type": "image_url", "image_url": image_url}) + return content + + def _kwargs(self, images: list[Path], prompt: str, *, with_seed: bool) -> dict: + kwargs: dict = { + "model": self.model, + "messages": [{"role": "user", "content": self._content(images, prompt)}], + "max_tokens": self.max_tokens, + "temperature": self.temperature, + } + if with_seed and self.seed is not None: + kwargs["seed"] = self.seed + return kwargs + + # -- calling -------------------------------------------------------------- + + def ask(self, images: list[Path], prompt: str) -> VLMResult: + """One question over ``images``. Never raises for provider-side refusals. + + Determinism is requested, not assumed. Providers that reject ``seed`` outright + get one retry without it, and the result records that the run was unseeded so + a reader knows the noise floor is doing more work on this provider. + """ + result = self._attempt(images, prompt, with_seed=True) + if result.ok or not result.error or self.seed is None: + return result + low = result.error.lower() + if "seed" in low or any(m in low for m in _UNSUPPORTED_MARKERS): + retry = self._attempt(images, prompt, with_seed=False) + retry.seed_accepted = False + return retry + return result + + def _attempt(self, images: list[Path], prompt: str, *, with_seed: bool) -> VLMResult: + kwargs = self._kwargs(images, prompt, with_seed=with_seed) + started = time.perf_counter() + try: + if self.stream: + return self._stream(kwargs, started) + return self._blocking(kwargs, started) + except Exception as e: # provider-side failure is data, not a crash + msg = str(e) + low = msg.lower() + return VLMResult( + ok=False, + error=msg, + wall_sec=time.perf_counter() - started, + multi_image_unsupported=len(images) > 1 and any(m in low for m in _MULTI_IMAGE_MARKERS), + detail_accepted=None if self.detail is None else False, + ) + + def _blocking(self, kwargs: dict, started: float) -> VLMResult: + response = self.client.chat.completions.create(timeout=self.timeout, **kwargs) + wall = time.perf_counter() - started + usage = getattr(response, "usage", None) + return VLMResult( + text=(response.choices[0].message.content or "").strip(), + tokens_in=getattr(usage, "prompt_tokens", None) if usage else None, + tokens_out=getattr(usage, "completion_tokens", None) if usage else None, + ttft_sec=None, + wall_sec=wall, + raw_usage=_usage_dict(usage), + detail_accepted=None if self.detail is None else True, + ) + + def _stream(self, kwargs: dict, started: float) -> VLMResult: + stream = self.client.chat.completions.create( + stream=True, + stream_options={"include_usage": True}, + timeout=self.timeout, + **kwargs, + ) + chunks: list[str] = [] + ttft: float | None = None + usage = None + for event in stream: + if getattr(event, "usage", None): + usage = event.usage + for choice in getattr(event, "choices", None) or []: + piece = getattr(choice.delta, "content", None) + if piece: + if ttft is None: + ttft = time.perf_counter() - started + chunks.append(piece) + wall = time.perf_counter() - started + return VLMResult( + text="".join(chunks).strip(), + tokens_in=getattr(usage, "prompt_tokens", None) if usage else None, + tokens_out=getattr(usage, "completion_tokens", None) if usage else None, + ttft_sec=ttft, + wall_sec=wall, + raw_usage=_usage_dict(usage), + detail_accepted=None if self.detail is None else True, + ) + + +def _usage_dict(usage) -> dict: + if usage is None: + return {} + if hasattr(usage, "model_dump"): + try: + return usage.model_dump() + except Exception: + pass + return { + k: getattr(usage, k) + for k in ("prompt_tokens", "completion_tokens", "total_tokens") + if getattr(usage, k, None) is not None + } + + +DEFAULT_PROBE_WIDTHS: tuple[int, ...] = (256, 512, 768, 1024, 1536) + + +def _resized_copy(image: Path, width: int, out_dir: Path) -> Path | None: + """Aspect-preserving resize via ffmpeg, so the probe controls the one knob that + actually moves per-frame token cost on most vision stacks: input resolution.""" + import subprocess + + out_path = out_dir / f"probe_w{width}{image.suffix or '.jpg'}" + cmd = [ + "ffmpeg", "-hide_banner", "-loglevel", "error", "-nostdin", "-y", + "-i", str(image), "-vf", f"scale={width}:-2", "-q:v", "2", str(out_path), + ] + try: + subprocess.run(cmd, capture_output=True, check=True, timeout=60) + except Exception: + return None + return out_path if out_path.exists() and out_path.stat().st_size else None + + +def probe_tokens_per_frame( + config: AVConfig, + image: Path, + *, + model: str | None = None, + details: tuple[str | None, ...] = (None, "low", "high"), + widths: tuple[int, ...] = DEFAULT_PROBE_WIDTHS, + baseline_prompt: str = "Reply with the single word: ok", +) -> dict: + """Measure whether tokens-per-frame is tunable on this deployment, and by which knob. + + Two candidate levers are tested separately, because they are not equivalent and + a provider may honour one and ignore the other: + + ``detail`` the OpenAI ``image_url`` hint. Some hosted APIs implement it; + some self-hosted servers parse it and never read it back, in + which case it is inert and must not be reported as a knob. + ``resolution`` the pixels actually uploaded. Where a vision tower maps a frame + onto a grid, this is the real lever and the client owns it. + + Per-image cost is the prompt-token count minus a text-only baseline. When every + setting on both axes yields the same count, the axis is genuinely fixed for this + deployment and the sweep should drop it rather than fake it. + """ + import shutil + import tempfile + + text_only = BenchVLM(config, model=model, stream=False, max_tokens=16) + base = text_only.ask([], baseline_prompt) + + def attributable(res: VLMResult) -> int | None: + if res.tokens_in is None or base.tokens_in is None: + return None + return res.tokens_in - base.tokens_in + + detail_obs: list[dict] = [] + for detail in details: + vlm = BenchVLM(config, model=model, detail=detail, stream=False, max_tokens=16) + res = vlm.ask([image], baseline_prompt) + detail_obs.append( + { + "detail": detail or "(unset)", + "ok": res.ok, + "error": res.error, + "prompt_tokens": res.tokens_in, + "tokens_attributable_to_image": attributable(res), + } + ) + + width_obs: list[dict] = [] + work = Path(tempfile.mkdtemp(prefix="av_bench_probe_widths_")) + try: + for width in widths: + resized = _resized_copy(image, width, work) + if resized is None: + width_obs.append({"width": width, "ok": False, "error": "resize failed"}) + continue + vlm = BenchVLM(config, model=model, stream=False, max_tokens=16) + res = vlm.ask([resized], baseline_prompt) + width_obs.append( + { + "width": width, + "ok": res.ok, + "error": res.error, + "prompt_tokens": res.tokens_in, + "tokens_attributable_to_image": attributable(res), + } + ) + finally: + shutil.rmtree(work, ignore_errors=True) + + def distinct(observations: list[dict]) -> list[int]: + return sorted( + { + o["tokens_attributable_to_image"] + for o in observations + if o.get("ok") and o.get("tokens_attributable_to_image") is not None + } + ) + + detail_counts = distinct(detail_obs) + width_counts = distinct(width_obs) + all_counts = sorted(set(detail_counts) | set(width_counts)) + + detail_is_knob = len(detail_counts) > 1 + width_is_knob = len(width_counts) > 1 + + if width_is_knob or detail_is_knob: + verdict = "tunable" + elif all_counts: + verdict = "fixed" + else: + verdict = "unknown" + + knobs = [k for k, on in (("detail", detail_is_knob), ("resolution", width_is_knob)) if on] + + return { + "baseline_text_only_prompt_tokens": base.tokens_in, + "baseline_ok": base.ok, + "baseline_error": base.error, + "detail_observations": detail_obs, + "resolution_observations": width_obs, + "distinct_image_token_counts": all_counts, + "effective_knobs": knobs, + "verdict": verdict, + } diff --git a/src/av/cli/app.py b/src/av/cli/app.py index 5f49124..449c1ae 100644 --- a/src/av/cli/app.py +++ b/src/av/cli/app.py @@ -36,6 +36,7 @@ def version_cmd() -> None: from av.cli.config_cmd import config_app # noqa: E402 from av.cli.sentinel import register as register_sentinel # noqa: E402 from av.cli.sentinel_doctor import register_doctor # noqa: E402 +from av.cli.bench import register as register_bench # noqa: E402 register_ingest(app) register_search(app) @@ -47,6 +48,7 @@ def version_cmd() -> None: register_open(app) register_sentinel(app) register_doctor(app) +register_bench(app) app.add_typer(config_app, name="config", help="Show/set configuration") diff --git a/src/av/cli/bench.py b/src/av/cli/bench.py new file mode 100644 index 0000000..c3f6f48 --- /dev/null +++ b/src/av/cli/bench.py @@ -0,0 +1,876 @@ +"""av bench — measure the cost/accuracy frontier for video understanding. + +Headline axes, chosen so results read against published agentic-video comparisons: +**tokens per query** and **accuracy**. Alongside them sits the axis an API vendor +cannot report — **dollars per query** and **video-hours per dollar** on hardware you +own — because that is the number that decides whether in-shore deployment pays. + +Subcommands, in the order you should run them: + +``av bench probe`` what can this deployment actually do? Is tokens-per-frame tunable? +``av bench gate`` can the model order a handful of images at all? Run this first. +``av bench prepare`` turn a public benchmark's annotations into a task file. +``av bench run`` dense vs agentic arms over a task file. +``av bench sweep`` event recall against sampling interval, on real footage. +``av bench noise`` the spread across identical runs — the floor below which deltas are noise. +``av bench cost`` the arithmetic, with every input labelled. No API calls. + +Every subcommand writes a receipt. JSON to stdout, progress to stderr. +""" + +from __future__ import annotations + +import json +import tempfile +from pathlib import Path + +import typer + +from av.bench.cost import CellUsage, HourlyCost, parse_cost_model, video_hours_per_dollar +from av.bench.datasets import ADAPTERS, SOURCES, load_annotations, subset_by_video +from av.bench.fixtures import FIXTURE_KINDS, MAX_N, generate_fixture +from av.bench.receipts import ( + COMMUNITY, + DERIVED, + MEASURED, + UNTESTED, + Receipt, + redact_endpoint, + sha256_file, + write_receipt, +) +from av.bench.runner import ( + DEFAULT_FRAME_INTERVALS, + DEFAULT_NOISE_REPEATS, + collapse_point, + interpret_delta, + is_saturated, + noise_floor, +) +from av.bench.tasks import events as events_task +from av.bench.tasks import videoqa +from av.bench.tasks.ordering import run_gate +from av.bench.vlm import BenchVLM, probe_tokens_per_frame +from av.cli.output import error, output_json, progress +from av.core.config import get_config +from av.pipeline.ffmpeg import get_video_info + +bench_app = typer.Typer(help="Measure the cost/accuracy frontier for video understanding.") + +DEFAULT_RECEIPTS_DIR = Path("./bench-receipts") + + +# --- shared helpers ---------------------------------------------------------- + +def _provider_record(config, model: str | None, *, temperature: float, seed: int | None) -> dict: + """What the receipt records about the provider. Endpoints are reduced to a host.""" + return { + "provider": config.provider or "(unset)", + "model": model or config.vision_model, + "endpoint_host": redact_endpoint(config.api_base_url), + "temperature": temperature, + "seed": seed, + } + + +def _resolve_config(provider: str, model: str, base_url: str): + """Build a config from flags, falling back to the user's saved configuration. + + A provider override never reaches into source for an endpoint: it takes the + preset's default, which for self-hosted providers is a local placeholder, and + expects the endpoint from ``--base-url``, ``AV_API_BASE_URL``, or config.json. + """ + config = get_config() + if provider: + from av.core.constants import PROVIDER_PRESETS + + preset = PROVIDER_PRESETS.get(provider) + if preset is None: + raise typer.BadParameter( + f"unknown provider {provider!r}; known: {', '.join(sorted(PROVIDER_PRESETS))}" + ) + config = config.model_copy(update={"provider": provider, **preset}) + if base_url: + config = config.model_copy(update={"api_base_url": base_url}) + if model: + config = config.model_copy(update={"vision_model": model, "chat_model": model}) + return config + + +def _parse_intervals(spec: str) -> tuple[float, ...]: + if not spec: + return DEFAULT_FRAME_INTERVALS + return tuple(float(part) for part in spec.split(",") if part.strip()) + + +def _parse_sizes(spec: str) -> tuple[int, ...]: + return tuple(int(part) for part in spec.split(",") if part.strip()) + + +def _emit(receipt: Receipt, payload: dict, receipts_dir: Path) -> None: + path = write_receipt(receipt, receipts_dir) + payload["receipt"] = str(path) + payload["run_id"] = receipt.run_id + output_json(payload) + + +# --- probe ------------------------------------------------------------------- + +@bench_app.command("probe") +def probe_cmd( + provider: str = typer.Option("", "--provider", help="Provider preset to use"), + model: str = typer.Option("", "--model", "-m", help="Model id override"), + base_url: str = typer.Option("", "--base-url", help="Endpoint base URL override"), + receipts: Path = typer.Option(DEFAULT_RECEIPTS_DIR, "--receipts", help="Receipt output directory"), +) -> None: + """Ask a live deployment what it can do, instead of assuming. + + Answers two questions that decide the shape of the whole sweep: does the endpoint + accept multiple images in one request, and is the per-frame token cost tunable? + Two candidate knobs are tested separately — the OpenAI ``detail`` hint and the + resolution actually uploaded — because a server may honour one and silently + ignore the other. If neither moves the count, the tokens-per-frame axis is + dropped rather than faked. + """ + config = _resolve_config(provider, model, base_url) + work = Path(tempfile.mkdtemp(prefix="av_bench_probe_")) + fixture = generate_fixture("color", 2, work) + + progress(f" Probing {config.provider or '(unset)'} / {model or config.vision_model}...") + tokens = probe_tokens_per_frame(config, fixture.frame_paths[0], model=model or None) + + multi = BenchVLM(config, model=model or None, stream=False, max_tokens=16).ask( + fixture.frame_paths, "Reply with the single word: ok" + ) + multi_supported = ( + True if multi.ok else (False if multi.multi_image_unsupported else None) + ) + + receipt = Receipt( + kind="probe", + provider=_provider_record(config, model, temperature=0.0, seed=0), + determinism={ + "temperature": 0.0, + "seed": 0, + "fixture_ffmpeg_commands": fixture.ffmpeg_commands, + "fixture_sha256": [sha256_file(p) for p in fixture.frame_paths], + }, + cells=[tokens], + summary={ + "tokens_per_frame_verdict": tokens["verdict"], + "effective_knobs": tokens["effective_knobs"], + "distinct_image_token_counts": tokens["distinct_image_token_counts"], + "multi_image_supported": multi_supported, + "multi_image_error": multi.error, + }, + ) + + if tokens["verdict"] == "tunable": + receipt.add_claim( + MEASURED, + "Per-frame token count moved on this deployment via " + f"{', '.join(tokens['effective_knobs'])}, so tokens-per-frame is a usable " + "sweep axis here. Drive it with --scale-width.", + ) + elif tokens["verdict"] == "fixed": + receipt.add_claim( + MEASURED, + "Per-frame token count did not change across any tested detail setting or " + "input resolution on this deployment. The tokens-per-frame axis is dropped " + "rather than simulated.", + ) + else: + receipt.add_claim( + UNTESTED, + "Per-frame token count could not be established — the provider returned no " + "usage, or the endpoint was unreachable.", + ) + + receipt.notes.append( + "Measured through an OpenAI-compatible chat endpoint. A provider's native API " + "may bill images differently from its compatibility layer, so this verdict " + "describes the endpoint you are actually calling, not the model in general." + ) + + if multi_supported is False: + receipt.add_claim( + MEASURED, + "This deployment refused a two-image request. That is a capability result, " + "not a harness error: multi-image benchmarks cannot run against it.", + ) + + _emit( + receipt, + { + "verdict": tokens["verdict"], + "effective_knobs": tokens["effective_knobs"], + "probe": tokens, + "multi_image_supported": multi_supported, + }, + receipts, + ) + + +# --- gate -------------------------------------------------------------------- + +@bench_app.command("gate") +def gate_cmd( + kind: str = typer.Option("color", "--kind", help=f"Fixture kind: {', '.join(FIXTURE_KINDS)}"), + sizes: str = typer.Option("2,4,8", "--sizes", help="Comma-separated frame counts"), + provider: str = typer.Option("", "--provider", help="Provider preset to use"), + model: str = typer.Option("", "--model", "-m", help="Model id override"), + base_url: str = typer.Option("", "--base-url", help="Endpoint base URL override"), + keep_fixtures: str = typer.Option("", "--keep-fixtures", help="Directory to keep fixtures in"), + receipts: Path = typer.Option(DEFAULT_RECEIPTS_DIR, "--receipts", help="Receipt output directory"), +) -> None: + """Temporal-ordering capability gate — run this before anything else. + + A model that cannot report the order of a handful of images cannot be + meaningfully scored on long-video reasoning, and any throughput number measured + against it describes a machine doing the wrong thing quickly. The gate is cheap + and it can save the entire sweep. + """ + if kind not in FIXTURE_KINDS: + error(f"unknown fixture kind {kind!r}; expected one of {', '.join(FIXTURE_KINDS)}") + raise typer.Exit(2) + + fixture_sizes = _parse_sizes(sizes) + too_big = [n for n in fixture_sizes if n > MAX_N[kind]] + if too_big: + error(f"{kind} fixtures support at most {MAX_N[kind]} frames; asked for {too_big}") + raise typer.Exit(2) + + config = _resolve_config(provider, model, base_url) + work = Path(keep_fixtures).expanduser() if keep_fixtures else Path( + tempfile.mkdtemp(prefix="av_bench_gate_") + ) + vlm = BenchVLM(config, model=model or None, temperature=0.0, seed=0, max_tokens=512) + + def report(cell) -> None: + mark = "PASS" if cell.exact_order else "FAIL" + progress( + f" n={cell.n:<3} {mark} expected={len(cell.expected)} " + f"reported={cell.n_reported} prefix={cell.correct_prefix}" + ) + + progress(f" Gate: {kind} fixtures, sizes {list(fixture_sizes)}") + cells, summary = run_gate(vlm, kind=kind, sizes=fixture_sizes, work_dir=work, on_cell=report) + + receipt = Receipt( + kind="gate", + provider=_provider_record(config, model, temperature=0.0, seed=0), + determinism={ + "temperature": 0.0, + "seed": 0, + "fixture_version": 1, + "ffmpeg_commands": summary.pop("ffmpeg_commands"), + }, + cells=[c.to_dict() for c in cells], + summary=summary, + ) + + verdict = summary["verdict"] + if verdict == "pass": + receipt.add_claim(MEASURED, f"Exact-order accuracy was 100% at every tested size {list(fixture_sizes)}.") + elif verdict == "no_multi_image": + receipt.add_claim(MEASURED, "The provider refused multi-image input. No ordering result is possible.") + else: + receipt.add_claim( + MEASURED, + f"Ordering failed at sizes {summary['failed_sizes']}. Benchmark scores for this " + "model do not measure temporal understanding until this is fixed.", + ) + + payload = { + "verdict": verdict, + "largest_passing_n": summary["largest_passing_n"], + "cells": [c.to_dict() for c in cells], + } + if keep_fixtures: + payload["fixtures_dir"] = str(work) + _emit(receipt, payload, receipts) + + +# --- prepare ----------------------------------------------------------------- + +@bench_app.command("prepare") +def prepare_cmd( + dataset: str = typer.Argument(help=f"Benchmark name: {', '.join(sorted(ADAPTERS))}"), + annotations: Path = typer.Argument(help="Annotation file you downloaded yourself"), + out: Path = typer.Option(..., "--out", "-o", help="Task JSONL to write"), + video_template: str = typer.Option( + "videos/{video_id}.mp4", "--video-template", + help="Path pattern for the video of each row; {video_id} is substituted", + ), + max_questions: int = typer.Option(50, "--max-questions", help="Cap on questions"), + max_videos: int = typer.Option(0, "--max-videos", help="Cap on distinct videos (0 = no cap)"), +) -> None: + """Convert a public benchmark's annotations into a task file. + + No benchmark data ships with av and no videos are downloaded here. Fetch the + annotation file yourself, mind its licence — LVBench's is non-commercial — and + fetch the videos separately. Subsetting is grouped by video on purpose: sampling + by question is how a 30-question run turns into a 40-hour download. + """ + if dataset not in ADAPTERS: + error(f"unknown dataset {dataset!r}; known: {', '.join(sorted(ADAPTERS))}") + raise typer.Exit(2) + + rows = ADAPTERS[dataset](load_annotations(annotations)) + chosen = subset_by_video( + rows, max_questions=max_questions, max_videos=max_videos or None + ) + + out = Path(out).expanduser() + out.parent.mkdir(parents=True, exist_ok=True) + with open(out, "w") as f: + for row in chosen: + f.write(json.dumps(row.to_task_row(video_template)) + "\n") + + source = SOURCES.get(dataset, {}) + progress(f" Wrote {len(chosen)} questions to {out}") + output_json({ + "dataset": dataset, + "source": source, + "questions_available": len(rows), + "questions_written": len(chosen), + "videos_referenced": len({r.video_id for r in chosen}), + "task_file": str(out), + "video_ids": sorted({r.video_id for r in chosen}), + "note": ( + "Videos are not downloaded. Fetch them separately and check the upstream " + "licence before any downstream use." + ), + }) + + +# --- run --------------------------------------------------------------------- + +@bench_app.command("run") +def run_cmd( + task_file: Path = typer.Argument(help="Task JSONL (see `av bench prepare`)"), + arms: str = typer.Option("dense,agentic", "--arms", help="Comma-separated: dense, agentic"), + interval: float = typer.Option(1.0, "--interval", help="Dense arm sampling interval, seconds"), + coarse_interval: float = typer.Option( + videoqa.DEFAULT_COARSE_INTERVAL_SEC, "--coarse-interval", + help="Agentic arm coarse-pass interval, seconds", + ), + budget: int = typer.Option( + videoqa.DEFAULT_AGENTIC_BUDGET, "--budget", help="Agentic arm targeted frame budget" + ), + max_frames: int = typer.Option(1024, "--max-frames", help="Hard ceiling on frames per request"), + scale_width: int = typer.Option(768, "--scale-width", help="Frame width in pixels (0 = native)"), + limit: int = typer.Option(0, "--limit", help="Only run the first N questions (0 = all)"), + provider: str = typer.Option("", "--provider", help="Provider preset to use"), + model: str = typer.Option("", "--model", "-m", help="Model id override"), + base_url: str = typer.Option("", "--base-url", help="Endpoint base URL override"), + cost: str = typer.Option( + "", "--cost", + help="Cost model: hourly:25.0[:prefill_tok_s[:decode_tok_s]] | token:IN:OUT | @file.json", + ), + receipts: Path = typer.Option(DEFAULT_RECEIPTS_DIR, "--receipts", help="Receipt output directory"), +) -> None: + """Dense versus agentic on the same questions, with token and dollar cost. + + The dense arm samples the whole window at a fixed rate and asks once. The agentic + arm takes a cheap coarse look, decides which moments it needs, then fetches only + those — and is charged for both requests. Nothing else differs between them. + """ + config = _resolve_config(provider, model, base_url) + cost_model = parse_cost_model(cost) + wanted = [a.strip() for a in arms.split(",") if a.strip()] + unknown = [a for a in wanted if a not in videoqa.ARMS] + if unknown: + error(f"unknown arm(s): {unknown}; expected any of {list(videoqa.ARMS)}") + raise typer.Exit(2) + + questions = videoqa.load_questions(task_file) + if limit: + questions = questions[:limit] + missing = [q for q in questions if not q.video.exists()] + if missing: + error( + f"{len(missing)} question(s) reference videos that are not on disk, " + f"e.g. {missing[0].video}. Fetch them first." + ) + raise typer.Exit(1) + if not questions: + error("task file contains no questions") + raise typer.Exit(1) + + vlm = BenchVLM( + config, model=model or None, temperature=0.0, seed=0, + max_tokens=512, stream=True, + ) + width = scale_width or None + + arm_results: dict[str, list[videoqa.QuestionResult]] = {} + video_seconds_total = 0.0 + for arm in wanted: + progress(f" Arm: {arm}") + results: list[videoqa.QuestionResult] = [] + for i, q in enumerate(questions, 1): + duration = get_video_info(q.video).duration_sec + if arm == wanted[0]: + start, span = videoqa.window_for(q, duration) + video_seconds_total += span or duration + if arm == "dense": + res = videoqa.run_dense( + vlm, q, duration, interval_sec=interval, + max_frames=max_frames, scale_width=width, + ) + else: + res = videoqa.run_agentic( + vlm, q, duration, coarse_interval_sec=coarse_interval, + budget_frames=budget, scale_width=width, + ) + results.append(res) + progress( + f" [{i}/{len(questions)}] {q.id} {'OK ' if res.correct else 'X '}" + f"frames={res.frames_sent} tok_in={res.tokens_in} wall={res.wall_sec:.1f}s" + ) + arm_results[arm] = results + + cells: list[dict] = [] + for arm, results in arm_results.items(): + summary = videoqa.aggregate(results) + usage = videoqa.usage_for(results) + cell = { + "arm": arm, + "interval_sec": interval if arm == "dense" else coarse_interval, + **summary, + "results": [r.to_dict() for r in results], + } + if cost_model: + total_cost = cost_model.cost_usd(usage) + cell["cost_usd_total"] = round(total_cost, 6) + cell["cost_usd_per_query"] = round(total_cost / len(results), 6) if results else None + cell["video_hours_per_dollar"] = video_hours_per_dollar(video_seconds_total, total_cost) + cell["cost_basis"] = cost_model.basis() + cells.append(cell) + + receipt = Receipt( + kind="arms", + provider=_provider_record(config, model, temperature=0.0, seed=0), + determinism={ + "temperature": 0.0, + "seed": 0, + "dense_interval_sec": interval, + "agentic_coarse_interval_sec": coarse_interval, + "agentic_budget_frames": budget, + "scale_width": width, + "max_frames": max_frames, + "task_file": str(task_file), + "task_file_sha256": sha256_file(Path(task_file)), + }, + cost_model=cost_model.describe() if cost_model else None, + cells=cells, + summary={ + "questions": len(questions), + "arms": wanted, + "video_seconds_covered": round(video_seconds_total, 1), + "by_arm": { + arm: { + k: v for k, v in videoqa.aggregate(res).items() + if k in ("accuracy", "tokens_per_query_total", "frames_per_query") + } + for arm, res in arm_results.items() + }, + }, + ) + receipt.add_claim(MEASURED, "Accuracy and token counts are from this run; tokens are the provider's own usage figures.") + if cost_model: + receipt.add_claim( + DERIVED if cost_model.basis() == "derived" else MEASURED, + f"Dollar figures follow the supplied {cost_model.mode} cost model.", + ) + receipt.add_claim( + COMMUNITY, + "Published dense-versus-agentic figures for other models were produced on " + "different hardware with different methodology. Read them alongside these " + "numbers as a comparison of approaches, never as a like-for-like ratio.", + source="vendor-published benchmark charts", + ) + receipt.notes.append( + "This run measures one model on one deployment. It is not a head-to-head " + "against any vendor's published result." + ) + + _emit(receipt, {"summary": receipt.summary, "cells": [ + {k: v for k, v in c.items() if k != "results"} for c in cells + ]}, receipts) + + +# --- sweep ------------------------------------------------------------------- + +@bench_app.command("sweep") +def sweep_cmd( + artifacts: Path = typer.Argument(help="Reference artifacts JSONL with start_sec/end_sec/text"), + video_dir: Path = typer.Argument(help="Directory holding the referenced video files"), + probes: str = typer.Option( + "door_activity", "--probes", + help=f"Comma-separated probes: {', '.join(events_task.EVENT_PROBES)}", + ), + intervals: str = typer.Option("", "--intervals", help="Comma-separated seconds between frames"), + max_per_probe: int = typer.Option(10, "--max-per-probe", help="Reference windows per probe"), + scale_width: int = typer.Option(768, "--scale-width", help="Frame width in pixels (0 = native)"), + provider: str = typer.Option("", "--provider", help="Provider preset to use"), + model: str = typer.Option("", "--model", "-m", help="Model id override"), + base_url: str = typer.Option("", "--base-url", help="Endpoint base URL override"), + cost: str = typer.Option("", "--cost", help="Cost model (see `av bench run --help`)"), + tolerance: float = typer.Option(0.1, "--tolerance", help="Score drop still counted as safe"), + receipts: Path = typer.Option(DEFAULT_RECEIPTS_DIR, "--receipts", help="Receipt output directory"), +) -> None: + """Event detection against sampling interval on real footage. + + The interval at which detection collapses is the cheapest safe sampling rate, and + it is a per-task answer, not a global one. Note the reference caveat: the shipped + artifacts were produced by a vision model, so this measures agreement with a dense + reference run, not recall against human ground truth. + """ + config = _resolve_config(provider, model, base_url) + cost_model = parse_cost_model(cost) + axis = _parse_intervals(intervals) + probe_list = [p.strip() for p in probes.split(",") if p.strip()] + unknown = [p for p in probe_list if p not in events_task.EVENT_PROBES] + if unknown: + error(f"unknown probe(s): {unknown}; expected any of {list(events_task.EVENT_PROBES)}") + raise typer.Exit(2) + + reference = events_task.load_reference_events( + artifacts, video_dir, probes=probe_list, max_per_probe=max_per_probe + ) + if not reference: + error( + "no reference windows found — check that the artifacts file has " + "start_sec/end_sec/text and that the videos exist in the given directory" + ) + raise typer.Exit(1) + + # Generous ceiling on purpose: a reasoning model can spend most of a small budget + # before it emits the answer, and a truncated reply would score as a miss. + vlm = BenchVLM(config, model=model or None, temperature=0.0, seed=0, max_tokens=512) + width = scale_width or None + + cells: list[dict] = [] + for probe in probe_list: + windows = [e for e in reference if e.probe == probe] + covered = sum(e.end_sec - e.start_sec for e in windows) + for interval_sec in axis: + progress(f" {probe} @ 1 frame / {interval_sec:g}s over {len(windows)} windows...") + cell = events_task.run_event_cell( + vlm, reference, interval_sec, probe=probe, scale_width=width + ) + row = cell.to_dict() + row["score"] = cell.reference_recall + if cost_model: + usage = CellUsage( + tokens_in=cell.tokens_in, tokens_out=cell.tokens_out, + wall_total_sec=cell.wall_sec, requests=cell.events_total, + ) + cell_cost = cost_model.cost_usd(usage) + row["cost_usd"] = round(cell_cost, 6) + row["video_hours_per_dollar"] = video_hours_per_dollar(covered, cell_cost) + row["cost_basis"] = cost_model.basis() + cells.append(row) + progress( + f" recall={row['reference_recall']} frames={row['frames_total']} " + f"tok_in={row['tokens_in']}" + ) + + frontier = { + probe: collapse_point( + [c for c in cells if c["probe"] == probe], score_key="score", tolerance=tolerance + ) + for probe in probe_list + } + + receipt = Receipt( + kind="sweep", + provider=_provider_record(config, model, temperature=0.0, seed=0), + determinism={ + "temperature": 0.0, + "seed": 0, + "intervals_sec": list(axis), + "scale_width": width, + "artifacts_file": str(artifacts), + "artifacts_sha256": sha256_file(Path(artifacts)), + "probe_questions": {p: events_task.EVENT_PROBES[p]["question"] for p in probe_list}, + }, + cost_model=cost_model.describe() if cost_model else None, + cells=cells, + summary={ + "probes": probe_list, + "intervals_sec": list(axis), + "reference_windows": len(reference), + "frontier": frontier, + }, + notes=[events_task.REFERENCE_CAVEAT], + ) + receipt.add_claim(MEASURED, "Detection rates and token counts are from this run.") + receipt.add_claim( + UNTESTED, + "Precision was not measured: only windows the reference marks as containing " + "the event were shown, so false positives on empty windows are unknown.", + ) + _emit(receipt, {"frontier": frontier, "cells": cells, "caveat": events_task.REFERENCE_CAVEAT}, receipts) + + +# --- noise ------------------------------------------------------------------- + +@bench_app.command("noise") +def noise_cmd( + kind: str = typer.Option("color", "--kind", help=f"Fixture kind: {', '.join(FIXTURE_KINDS)}"), + n: int = typer.Option(6, "--n", help="Frames in the repeated fixture"), + repeats: int = typer.Option(DEFAULT_NOISE_REPEATS, "--repeats", help="Identical runs"), + provider: str = typer.Option("", "--provider", help="Provider preset to use"), + model: str = typer.Option("", "--model", "-m", help="Model id override"), + base_url: str = typer.Option("", "--base-url", help="Endpoint base URL override"), + receipts: Path = typer.Option(DEFAULT_RECEIPTS_DIR, "--receipts", help="Receipt output directory"), +) -> None: + """Run one unchanged cell repeatedly and publish the spread. + + This is the number that makes every other number readable. A benchmark delta + smaller than this spread is noise, and reporting it as a result is how a + measurement turns into a claim it cannot support. + """ + config = _resolve_config(provider, model, base_url) + work = Path(tempfile.mkdtemp(prefix="av_bench_noise_")) + fixture = generate_fixture(kind, n, work) + vlm = BenchVLM(config, model=model or None, temperature=0.0, seed=0, max_tokens=512) + + from av.bench.tasks.ordering import run_ordering_cell + + observations: list[dict] = [] + + def once(i: int) -> float | None: + cell = run_ordering_cell(vlm, fixture) + observations.append(cell.to_dict()) + if not cell.ok: + return None + return cell.correct_prefix / cell.n + + def report(i: int, value: float | None) -> None: + progress(f" run {i + 1}/{repeats}: score={value}") + + progress(f" Noise floor: {repeats} identical runs of a {n}-frame {kind} fixture") + spread = noise_floor(once, repeats=repeats, unit=" score", on_run=report) + + token_values = [o["tokens_in"] for o in observations if o["tokens_in"] is not None] + token_spread = (max(token_values) - min(token_values)) if token_values else None + + receipt = Receipt( + kind="noise", + provider=_provider_record(config, model, temperature=0.0, seed=0), + determinism={ + "temperature": 0.0, + "seed": 0, + "repeats": repeats, + "fixture_kind": kind, + "fixture_n": n, + "ffmpeg_commands": fixture.ffmpeg_commands, + "fixture_sha256": [sha256_file(p) for p in fixture.frame_paths], + }, + cells=observations, + summary={ + "score_spread": spread.to_dict(), + "prompt_token_spread": token_spread, + "interpretation": interpret_delta(0.0, spread), + }, + ) + saturated = is_saturated(spread) + receipt.summary["saturated"] = saturated + if saturated: + receipt.add_claim( + UNTESTED, + "Every run scored full marks, so this cell has no headroom to vary and its " + "zero spread is not a usable noise floor. Re-run --n at a size the model " + "does not solve perfectly.", + ) + else: + receipt.add_claim( + MEASURED, + f"Across {spread.n} identical runs the score spread was " + f"{spread.range if spread.range is not None else 'unmeasured'}. Any " + "single-run difference at or below that spread is noise.", + ) + _emit( + receipt, + { + "score_spread": spread.to_dict(), + "prompt_token_spread": token_spread, + "saturated": saturated, + "interpretation": receipt.summary["interpretation"], + }, + receipts, + ) + + +# --- cost -------------------------------------------------------------------- + +@bench_app.command("cost") +def cost_cmd( + tokens_per_frame: int = typer.Option(..., "--tokens-per-frame", help="Tokens the model retains per frame"), + context_tokens: int = typer.Option(0, "--context-tokens", help="Model context window in tokens"), + prefill_tok_s: float = typer.Option(0.0, "--prefill-tok-s", help="Prefill throughput, tokens/sec"), + hourly_usd: float = typer.Option(0.0, "--hourly-usd", help="Hardware cost, dollars per hour"), + kv_bytes_per_token: float = typer.Option(0.0, "--kv-bytes-per-token", help="KV cache bytes per token"), + intervals: str = typer.Option("", "--intervals", help="Comma-separated seconds between frames"), + target_vhpd: float = typer.Option( + 100.0, "--target-vhpd", help="Target video-hours per dollar to solve for" + ), + source: str = typer.Option( + "", "--source", help="Where the inputs came from, recorded in the receipt" + ), + receipts: Path = typer.Option(DEFAULT_RECEIPTS_DIR, "--receipts", help="Receipt output directory"), +) -> None: + """Work the arithmetic, with every input labelled. Makes no API calls. + + Nothing here is measured — it is all derived from the figures you pass in, which + is exactly why the receipt records them and why ``--source`` exists. Supply your + own numbers and the tool will tell you what sampling rate the target implies, + how many frames fit in one request, and how much KV cache an hour of video costs. + """ + if tokens_per_frame <= 0: + error("--tokens-per-frame must be positive") + raise typer.Exit(2) + + axis = _parse_intervals(intervals) + cost_model = HourlyCost(hourly_usd=hourly_usd, prefill_tok_per_s=prefill_tok_s or None) if hourly_usd else None + + rows: list[dict] = [] + for interval_sec in axis: + frames_per_video_hour = 3600.0 / interval_sec + tokens_per_video_hour = frames_per_video_hour * tokens_per_frame + row: dict = { + "interval_sec": interval_sec, + "frames_per_video_hour": round(frames_per_video_hour, 3), + "tokens_per_video_hour": round(tokens_per_video_hour, 1), + } + if prefill_tok_s: + gpu_seconds = tokens_per_video_hour / prefill_tok_s + row["gpu_seconds_per_video_hour"] = round(gpu_seconds, 3) + if hourly_usd: + dollars = gpu_seconds / 3600.0 * hourly_usd + row["cost_usd_per_video_hour"] = round(dollars, 6) + row["video_hours_per_dollar"] = round(1.0 / dollars, 3) if dollars else None + if kv_bytes_per_token: + row["kv_gib_per_video_hour"] = round( + tokens_per_video_hour * kv_bytes_per_token / (1024 ** 3), 4 + ) + rows.append(row) + + summary: dict = {"tokens_per_frame": tokens_per_frame} + + if context_tokens: + frames_in_context = context_tokens // tokens_per_frame + summary["frames_per_request_max"] = frames_in_context + summary["single_request_minutes_at_1fps"] = round(frames_in_context / 60.0, 2) + + if prefill_tok_s and hourly_usd: + # Solve: 1 / ((3600/I * tpf) / prefill / 3600 * hourly) = target + # => I = target * tpf * hourly / prefill + required_interval = target_vhpd * tokens_per_frame * hourly_usd / prefill_tok_s + summary["target_video_hours_per_dollar"] = target_vhpd + summary["required_interval_sec"] = round(required_interval, 2) + summary["required_interval_description"] = ( + f"one frame every {required_interval:.0f} seconds at {tokens_per_frame} tokens/frame" + ) + + receipt = Receipt( + kind="cost", + provider={"note": "no provider contacted — this subcommand is arithmetic only"}, + determinism={ + "tokens_per_frame": tokens_per_frame, + "context_tokens": context_tokens or None, + "prefill_tok_per_s": prefill_tok_s or None, + "hourly_usd": hourly_usd or None, + "kv_bytes_per_token": kv_bytes_per_token or None, + "intervals_sec": list(axis), + "input_source": source or "(not stated by the caller)", + }, + cost_model=cost_model.describe() if cost_model else None, + cells=rows, + summary=summary, + ) + receipt.add_claim( + DERIVED, + "Every figure in this receipt is arithmetic over the inputs supplied on the " + "command line. None of it was measured against a running model.", + ) + if not source: + receipt.add_claim( + UNTESTED, + "The caller did not state where these inputs came from, so their provenance " + "is unknown. Pass --source to record it.", + ) + _emit(receipt, {"summary": summary, "rows": rows}, receipts) + + +# --- plan -------------------------------------------------------------------- + +@bench_app.command("plan") +def plan_cmd( + widths: str = typer.Option( + "256,512,768,1024,1536,1920", "--widths", help="Comma-separated frame widths in pixels" + ), + aspect: float = typer.Option(16 / 9, "--aspect", help="Frame aspect ratio (width / height)"), + budgets: str = typer.Option( + "", "--budgets", help="Comma-separated token budgets to solve widths for" + ), + receipts: Path = typer.Option(DEFAULT_RECEIPTS_DIR, "--receipts", help="Receipt output directory"), +) -> None: + """Predict per-frame token cost against frame resolution. Makes no API calls. + + Useful before spending anything: it shows where the two walls are — the upscale + floor, below which shrinking frames buys nothing, and the token ceiling, above + which extra resolution is discarded. A tokens-per-frame sweep belongs between + them, and ``--scale-width`` on the other subcommands is how you drive it. + + These are predictions from a published preprocessor algorithm, not measurements + of your server. Confirm them with ``av bench probe``. + """ + from av.providers.deepseek import ( + MAX_IMAGE_TOKENS, + MIN_IMAGE_TOKENS, + MIN_PIXELS, + plan_image_tokens, + widths_for_token_budgets, + ) + + rows = [] + for width in (int(w) for w in widths.split(",") if w.strip()): + height = max(int(round(width / aspect)), 1) + rows.append({"width": width, "height": height, **plan_image_tokens(width, height).to_dict()}) + + solved = ( + widths_for_token_budgets([int(b) for b in budgets.split(",") if b.strip()], aspect) + if budgets + else {} + ) + + receipt = Receipt( + kind="plan", + provider={"note": "no provider contacted — this subcommand is arithmetic only"}, + determinism={"aspect_ratio": aspect, "widths": widths, "budgets": budgets or None}, + cells=rows, + summary={ + "min_pixels_floor": MIN_PIXELS, + "min_tokens_per_frame": MIN_IMAGE_TOKENS, + "max_tokens_per_frame": MAX_IMAGE_TOKENS, + "widths_for_budgets": solved, + }, + ) + receipt.add_claim( + DERIVED, + "Token counts are computed from the model's published vision preprocessor " + "algorithm and reproduce its published worked examples exactly below the " + "token ceiling. Rows marked approximate ran a shrink search and may differ " + "by one grid step; measure those with `av bench probe`.", + source="model vision_config and reference image processor", + ) + _emit(receipt, {"rows": rows, "widths_for_budgets": solved}, receipts) + + +def register(app: typer.Typer) -> None: + app.add_typer(bench_app, name="bench", help="Benchmark the cost/accuracy frontier") diff --git a/src/av/cli/config_cmd.py b/src/av/cli/config_cmd.py index a31f03e..489ff1d 100644 --- a/src/av/cli/config_cmd.py +++ b/src/av/cli/config_cmd.py @@ -21,6 +21,7 @@ ("pixelml", "PixelML (OpenRouter)", "routes to GPT-4.1, Claude, Gemini via PixelML gateway"), ("anthropic", "Anthropic (Claude)", "paste your Anthropic key"), ("gemini", "Google (Gemini)", "paste your Google API key"), + ("deepseek", "DeepSeek-V4.1-Flash (self-hosted)", "SGLang endpoint you run yourself"), ] @@ -135,6 +136,30 @@ def config_setup() -> None: "Use [bold]--no-embed[/bold] or set [bold]AV_OPENAI_API_KEY[/bold] env var for transcription." ) + elif provider_key == "deepseek": + _console.print() + _console.print( + " This provider talks to an OpenAI-compatible SGLang server that " + "[bold]you[/bold] run. No endpoint ships with av." + ) + base_url = Prompt.ask( + " Enter your endpoint base URL", + console=_console, + default=config_data["api_base_url"], + ) + config_data["api_base_url"] = base_url.strip() + api_key = Prompt.ask( + " Enter an API key (press Enter if your server does not require one)", + console=_console, + default="", + ) + config_data["api_key"] = api_key.strip() + _console.print() + _console.print( + " [yellow]Note:[/yellow] Transcription and embeddings are not served by this " + "deployment. Set [bold]AV_OPENAI_API_KEY[/bold] if you want those stages." + ) + elif provider_key == "gemini": api_key = Prompt.ask(" Enter your Google API key", console=_console) config_data["api_key"] = api_key.strip() diff --git a/src/av/core/constants.py b/src/av/core/constants.py index 197fc3e..9142c27 100644 --- a/src/av/core/constants.py +++ b/src/av/core/constants.py @@ -59,6 +59,16 @@ "embed_model": "text-embedding-004", "chat_model": "gemini-2.5-flash", }, + # DeepSeek-V4.1-Flash served by SGLang. The base URL is SGLang's own local + # default — a placeholder for a server you run, not a hosted endpoint. Point + # AV_API_BASE_URL at your deployment. + "deepseek": { + "api_base_url": "http://localhost:30000/v1", + "transcribe_model": "", + "vision_model": "deepseek-v4.1-flash", + "embed_model": "", + "chat_model": "deepseek-v4.1-flash", + }, "pixelml": { "api_base_url": "https://ishi.pixelml.com/v1", "transcribe_model": "", diff --git a/src/av/providers/deepseek.py b/src/av/providers/deepseek.py new file mode 100644 index 0000000..9f4ab97 --- /dev/null +++ b/src/av/providers/deepseek.py @@ -0,0 +1,334 @@ +"""DeepSeek-V4.1-Flash, served behind an OpenAI-compatible API (SGLang). + +This module is deliberately thin. SGLang exposes ``/v1/chat/completions`` with the +standard ``image_url`` content format, so the existing OpenAI-compatible client in +``providers/openai.py`` already speaks to it correctly. Reimplementing that shape +would add a second code path to maintain and would quietly break the repository's +provider-agnostic contract, so what lives here is only what is genuinely different: + +* endpoint resolution that reads config and environment, never source +* the model's declared capability record, so the benchmark can report what it + assumed rather than silently assuming it +* a runtime probe, because the interesting properties of a self-hosted deployment + (does it accept multiple images? is per-frame token count tunable?) are + deployment facts, not model facts, and must be measured against the server you + actually have + +**No endpoint is hard-coded.** The preset default points at SGLang's own local +default port. Any other endpoint comes from ``AV_API_BASE_URL`` or +``~/.config/av/config.json``. +""" + +from __future__ import annotations + +import math +import os +from dataclasses import dataclass, field +from pathlib import Path + +from av.core.config import AVConfig + +PROVIDER_NAME = "deepseek" +DEFAULT_MODEL = "deepseek-v4.1-flash" + +# SGLang's documented default bind address. A placeholder, not a deployment. +DEFAULT_BASE_URL = "http://localhost:30000/v1" + +# Environment variables consulted for credentials, in order. +API_KEY_ENV_VARS = ("AV_API_KEY", "DEEPSEEK_API_KEY", "SGLANG_API_KEY") + +# A self-hosted server usually needs no key; SGLang accepts any bearer token unless +# started with --api-key. This placeholder keeps the OpenAI SDK from refusing to send. +PLACEHOLDER_KEY = "no-key" + + +@dataclass +class CapabilityRecord: + """What the harness believes about a deployment, and on what evidence. + + Fields default to ``None`` — meaning *not established* — rather than to a + plausible value. A benchmark that guesses its own assumptions is not a benchmark. + """ + + model: str = DEFAULT_MODEL + context_tokens: int | None = None + image_tokens_per_frame: int | None = None + image_tokens_tunable: bool | None = None + kv_bytes_per_token: float | None = None + multi_image_supported: bool | None = None + evidence: dict[str, str] = field(default_factory=dict) + + def to_dict(self) -> dict: + return { + "model": self.model, + "context_tokens": self.context_tokens, + "image_tokens_per_frame": self.image_tokens_per_frame, + "image_tokens_tunable": self.image_tokens_tunable, + "kv_bytes_per_token": self.kv_bytes_per_token, + "multi_image_supported": self.multi_image_supported, + "evidence": self.evidence, + } + + def max_frames_per_request(self, prompt_overhead_tokens: int = 512) -> int | None: + """How many frames fit in one request, given the context and per-frame cost. + + This is the constraint that caps a single-request video window, and it is the + reason dense processing of a long video is not merely expensive but impossible + past a certain duration. Returns ``None`` when either input is unestablished, + because the alternative is inventing a limit. + """ + if not self.context_tokens or not self.image_tokens_per_frame: + return None + usable = self.context_tokens - prompt_overhead_tokens + return max(usable // self.image_tokens_per_frame, 0) + + +def resolve_api_key(config: AVConfig | None = None) -> str: + """First non-empty of the configured key, then the environment, then a placeholder.""" + if config is not None: + configured = (config.api_key or "").strip() + if configured and configured.lower() != PLACEHOLDER_KEY: + return configured + for name in API_KEY_ENV_VARS: + value = os.environ.get(name, "").strip() + if value: + return value + return PLACEHOLDER_KEY + + +def resolve_base_url(config: AVConfig | None = None) -> str: + """Endpoint from env, then config, then the local SGLang default.""" + env = os.environ.get("AV_API_BASE_URL", "").strip() + if env: + return env + if config is not None and (config.api_base_url or "").strip(): + return config.api_base_url.strip() + return DEFAULT_BASE_URL + + +def make_config( + base_config: AVConfig | None = None, + *, + model: str | None = None, +) -> AVConfig: + """Build an ``AVConfig`` pointed at a DeepSeek-V4.1-Flash deployment.""" + chosen = model or (base_config.vision_model if base_config else None) or DEFAULT_MODEL + return AVConfig( + provider=PROVIDER_NAME, + api_base_url=resolve_base_url(base_config), + api_key=resolve_api_key(base_config), + transcribe_model="", # not served by this deployment + embed_model="", # not served by this deployment + vision_model=chosen, + chat_model=chosen, + ) + + +def probe( + config: AVConfig, + image: Path, + *, + model: str | None = None, + details: tuple[str | None, ...] = (None, "low", "high"), +) -> CapabilityRecord: + """Measure a live deployment's image-token behaviour and multi-image support. + + Nothing here is assumed from documentation. If the server is unreachable the + record comes back with ``None`` fields and the error recorded as evidence, which + is the honest outcome for an endpoint that is scaled to zero. + """ + from av.bench.vlm import BenchVLM, probe_tokens_per_frame + + record = CapabilityRecord(model=model or config.vision_model or DEFAULT_MODEL) + + tokens = probe_tokens_per_frame(config, image, model=model, details=details) + record.evidence["image_tokens"] = f"probe verdict: {tokens['verdict']}" + counts = tokens.get("distinct_image_token_counts") or [] + if tokens["verdict"] == "tunable": + record.image_tokens_tunable = True + record.image_tokens_per_frame = max(counts) if counts else None + elif tokens["verdict"] == "fixed": + record.image_tokens_tunable = False + record.image_tokens_per_frame = counts[0] if counts else None + else: + record.evidence["image_tokens"] = ( + "probe inconclusive: " + + (tokens.get("baseline_error") or "provider returned no usage") + ) + + two_frames = BenchVLM(config, model=model, stream=False, max_tokens=16) + multi = two_frames.ask([image, image], "Reply with the single word: ok") + if multi.ok: + record.multi_image_supported = True + record.evidence["multi_image"] = "two-image request accepted" + elif multi.multi_image_unsupported: + record.multi_image_supported = False + record.evidence["multi_image"] = f"provider refused multiple images: {multi.error}" + else: + record.evidence["multi_image"] = f"inconclusive: {multi.error}" + + return record + + +# --- image token planning ---------------------------------------------------- +# +# The vision tower is a single aspect-preserving grid, not LLaVA AnyRes tiling and +# not Qwen2-VL min/max pixels. Frames are resized, cut into `patch_size` patches, +# then a 3x3 aligner collapses each 42x42 pixel cell into one LLM token, with one +# newline token per grid row plus a start and end token. +# +# The consequence that matters for video: **per-frame token cost is a function of +# input resolution, and the client controls it.** Resizing before upload is the +# tuning knob. The OpenAI `detail` field is not — SGLang parses it and no +# multimodal processor reads it back, so it is inert against a self-hosted server +# even though the vendor's own hosted API honours it. +# +# These constants mirror the published `vision_config`. They are DOCUMENTED, not +# measured here, and `probe()` exists precisely to check them against a live server. + +PATCH_SIZE = 14 +ALIGNER_DOWNSAMPLE = 3 +PIXELS_PER_LLM_TOKEN = PATCH_SIZE * ALIGNER_DOWNSAMPLE # 42 +MAX_IMAGE_TOKENS = 1024 +MIN_PIXELS = 295_936 # 544 x 544 — smaller frames are upscaled to meet this floor +MIN_IMAGE_TOKENS = 184 # what that floor costs: a 13x13 grid + + +@dataclass +class ImageTokenPlan: + """What one frame will cost, and at what resolution it will be processed.""" + + source_width: int + source_height: int + processed_width: int + processed_height: int + grid_rows: int + grid_cols: int + tokens: int + upscaled: bool + downscaled: bool + + @property + def approximate(self) -> bool: + """True when a shrink search ran, where we may differ by one grid step. + + The published algorithm reproduces exactly for frames at or below the + token ceiling. Above it, the upstream resize solver can land one grid row + away from this one, so a downscaled estimate is a plan, not a promise — + measure the real count with ``av bench probe``. + """ + return self.downscaled + + def to_dict(self) -> dict: + return { + "source": [self.source_width, self.source_height], + "processed": [self.processed_width, self.processed_height], + "grid": [self.grid_rows, self.grid_cols], + "tokens": self.tokens, + "upscaled": self.upscaled, + "downscaled": self.downscaled, + "approximate": self.approximate, + } + + +def _align_up(value: int, multiple: int) -> int: + return int(math.ceil(value / multiple) * multiple) + + +def _grid_for(width: int, height: int) -> tuple[int, int]: + rows = math.ceil(math.ceil(height / PATCH_SIZE) / ALIGNER_DOWNSAMPLE) + cols = math.ceil(math.ceil(width / PATCH_SIZE) / ALIGNER_DOWNSAMPLE) + return rows, cols + + +def tokens_for_grid(rows: int, cols: int) -> int: + """One token per cell, one newline per row, plus image start and end.""" + return rows * (cols + 1) + 2 + + +def plan_image_tokens( + width: int, + height: int, + *, + max_image_tokens: int = MAX_IMAGE_TOKENS, + min_pixels: int = MIN_PIXELS, +) -> ImageTokenPlan: + """Predict the token cost of one frame at a given source resolution. + + Two behaviours are worth knowing before choosing a frame size: + + * Below ``min_pixels`` the frame is **upscaled**, so a 64x64 thumbnail costs + exactly what a 544x544 frame costs. Shrinking past that floor buys nothing. + * Above the token ceiling the frame is shrunk until it fits, so beyond roughly + 1300x1300-equivalent area, extra resolution is discarded rather than charged. + + The useful range is therefore between those two walls, and that is where a + tokens-per-frame sweep should place its samples. + """ + if width <= 0 or height <= 0: + raise ValueError("width and height must be positive") + + ratio = 1.0 + upscaled = False + if width * height < min_pixels: + ratio = math.sqrt(min_pixels / (width * height)) + upscaled = True + + downscaled = False + processed_w = _align_up(int(width * ratio), PATCH_SIZE) + processed_h = _align_up(int(height * ratio), PATCH_SIZE) + rows, cols = _grid_for(processed_w, processed_h) + + # Shrink one patch step at a time on the long edge until the grid fits. + guard = 0 + while tokens_for_grid(rows, cols) > max_image_tokens and guard < 10_000: + downscaled = True + guard += 1 + ratio *= 0.99 + processed_w = _align_up(max(int(width * ratio), PATCH_SIZE), PATCH_SIZE) + processed_h = _align_up(max(int(height * ratio), PATCH_SIZE), PATCH_SIZE) + rows, cols = _grid_for(processed_w, processed_h) + + return ImageTokenPlan( + source_width=width, + source_height=height, + processed_width=processed_w, + processed_height=processed_h, + grid_rows=rows, + grid_cols=cols, + tokens=tokens_for_grid(rows, cols), + upscaled=upscaled, + downscaled=downscaled, + ) + + +def widths_for_token_budgets( + budgets: list[int], aspect_ratio: float = 16 / 9 +) -> dict[int, int]: + """Largest frame width whose predicted cost stays within each token budget. + + This is what turns tokens-per-frame into an actual sweep axis: pass the widths + to ``--scale-width`` and the frames arrive at the intended cost. Budgets below + the upscale floor map to the floor width, since nothing cheaper exists. + """ + out: dict[int, int] = {} + for budget in budgets: + best = 0 + for width in range(PIXELS_PER_LLM_TOKEN, 2400, PATCH_SIZE): + height = max(int(round(width / aspect_ratio)), PATCH_SIZE) + plan = plan_image_tokens(width, height) + # Skip widths past the ceiling: they all collapse to the same processed + # size, so reporting the largest of them would suggest a resolution the + # server will silently discard. + if plan.downscaled: + break + if plan.tokens <= budget: + best = width + out[budget] = best or PIXELS_PER_LLM_TOKEN + return out + + +# Documented and community-reported figures are intentionally NOT baked in as +# defaults. Supply them explicitly via `av bench cost --context-tokens ... etc` so +# that every receipt records where the number came from instead of inheriting it +# from a constant nobody re-checked. diff --git a/src/av/providers/openai.py b/src/av/providers/openai.py index 00e593d..f65028f 100644 --- a/src/av/providers/openai.py +++ b/src/av/providers/openai.py @@ -91,6 +91,11 @@ def _resolve_api_key(config: AVConfig) -> str: if pixelml_key: return pixelml_key + if config.provider == "deepseek": + from av.providers.deepseek import resolve_api_key as deepseek_key + + return deepseek_key(config) + # Prefer OpenClaw auth-profile OAuth (often fresher), then Codex CLI cache. oauth = _openclaw_oauth_token() or _codex_oauth_token() if oauth: diff --git a/tests/test_bench_cost.py b/tests/test_bench_cost.py new file mode 100644 index 0000000..fe84cae --- /dev/null +++ b/tests/test_bench_cost.py @@ -0,0 +1,171 @@ +"""Tests for benchmark cost models, spreads, and the frontier summary.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from av.bench.cost import ( + CellUsage, + HourlyCost, + PerTokenCost, + parse_cost_model, + video_hours_per_dollar, +) +from av.bench.runner import ( + Spread, + collapse_point, + interpret_delta, + is_saturated, + noise_floor, +) + + +# --------------------------------------------------------------------------- +# Cost models +# --------------------------------------------------------------------------- + +def test_per_token_cost() -> None: + model = PerTokenCost(input_usd_per_mtok=0.30, output_usd_per_mtok=2.50) + usage = CellUsage(tokens_in=1_000_000, tokens_out=100_000) + assert model.cost_usd(usage) == pytest.approx(0.30 + 0.25) + assert model.basis() == "measured" + + +def test_per_token_cost_ignores_wall_clock() -> None: + model = PerTokenCost(input_usd_per_mtok=1.0, output_usd_per_mtok=1.0) + fast = CellUsage(tokens_in=1000, wall_total_sec=1.0) + slow = CellUsage(tokens_in=1000, wall_total_sec=1000.0) + assert model.cost_usd(fast) == model.cost_usd(slow) + + +def test_hourly_cost_uses_wall_clock_without_throughput() -> None: + model = HourlyCost(hourly_usd=25.0) + usage = CellUsage(tokens_in=999_999, wall_total_sec=3600.0) + assert model.cost_usd(usage) == pytest.approx(25.0) + assert model.basis() == "measured" + + +def test_hourly_cost_with_throughput_is_derived() -> None: + model = HourlyCost(hourly_usd=25.0, prefill_tok_per_s=20_000) + usage = CellUsage(tokens_in=20_000 * 3600, wall_total_sec=0.0) + assert model.cost_usd(usage) == pytest.approx(25.0) + assert model.basis() == "derived" + + +def test_hourly_and_per_token_are_not_interchangeable() -> None: + """A busy box and an idle one bill identically per token; an API does not.""" + hourly = HourlyCost(hourly_usd=25.0) + per_token = PerTokenCost(input_usd_per_mtok=1.0, output_usd_per_mtok=1.0) + usage = CellUsage(tokens_in=1000, wall_total_sec=3600.0) + assert hourly.cost_usd(usage) != pytest.approx(per_token.cost_usd(usage)) + + +def test_video_hours_per_dollar() -> None: + assert video_hours_per_dollar(3600.0, 1.0) == pytest.approx(1.0) + assert video_hours_per_dollar(3600.0, 0.0) is None + + +def test_target_arithmetic_matches_the_stated_frontier() -> None: + """1,024 tokens/frame at 20k tok/s on a $25/hr box: 1 fps is well under 1 vh/$.""" + model = HourlyCost(hourly_usd=25.0, prefill_tok_per_s=20_000) + one_video_hour_at_1fps = CellUsage(tokens_in=3600 * 1024) + vhpd = video_hours_per_dollar(3600.0, model.cost_usd(one_video_hour_at_1fps)) + assert vhpd == pytest.approx(0.78, abs=0.02) + + # And one frame every 128 s reaches 100 video-hours per dollar. + sparse = CellUsage(tokens_in=int(3600 / 128 * 1024)) + assert video_hours_per_dollar(3600.0, model.cost_usd(sparse)) == pytest.approx(100, rel=0.01) + + +# --------------------------------------------------------------------------- +# Cost spec parsing +# --------------------------------------------------------------------------- + +def test_parse_cost_model_empty_is_none() -> None: + assert parse_cost_model("") is None + assert parse_cost_model(None) is None + + +def test_parse_hourly_spec() -> None: + model = parse_cost_model("hourly:25.0:20000:1200") + assert isinstance(model, HourlyCost) + assert model.hourly_usd == 25.0 + assert model.prefill_tok_per_s == 20_000 + assert model.decode_tok_per_s == 1200 + + +def test_parse_token_spec() -> None: + model = parse_cost_model("token:0.30:2.50") + assert isinstance(model, PerTokenCost) + assert model.output_usd_per_mtok == 2.50 + + +def test_parse_cost_model_from_file(tmp_path: Path) -> None: + path = tmp_path / "cost.json" + path.write_text(json.dumps({"mode": "per_hour", "hourly_usd": 12.5, "prefill_tok_per_s": 5000})) + model = parse_cost_model(f"@{path}") + assert isinstance(model, HourlyCost) + assert model.hourly_usd == 12.5 + + +def test_parse_cost_model_rejects_nonsense() -> None: + with pytest.raises(ValueError): + parse_cost_model("guess:12") + with pytest.raises(ValueError): + parse_cost_model("hourly") + + +# --------------------------------------------------------------------------- +# Noise floor +# --------------------------------------------------------------------------- + +def test_noise_floor_collects_spread() -> None: + values = [0.7, 0.8, 0.75, 0.9, 0.7] + spread = noise_floor(lambda i: values[i], repeats=5, unit=" score") + assert spread.n == 5 + assert spread.range == pytest.approx(0.2) + + +def test_noise_floor_drops_failed_runs() -> None: + spread = noise_floor(lambda i: None if i % 2 else 1.0, repeats=4) + assert spread.n == 2 + + +def test_delta_below_spread_is_called_noise() -> None: + spread = Spread(values=[0.70, 0.74], unit=" score") + assert "not a real difference" in interpret_delta(0.02, spread) + assert "exceeds the noise floor" in interpret_delta(0.5, spread) + + +def test_saturated_spread_is_not_a_noise_floor() -> None: + """A cell solved perfectly every time cannot report variance.""" + spread = Spread(values=[1.0, 1.0, 1.0]) + assert is_saturated(spread) + assert "measures nothing" in interpret_delta(0.01, spread) + + +def test_unmeasured_spread_says_so() -> None: + assert "unverified" in interpret_delta(0.5, Spread(values=[])) + + +# --------------------------------------------------------------------------- +# Frontier +# --------------------------------------------------------------------------- + +def test_collapse_point_finds_widest_safe_interval() -> None: + rows = [ + {"interval_sec": 1.0, "score": 0.90}, + {"interval_sec": 5.0, "score": 0.85}, + {"interval_sec": 10.0, "score": 0.83}, + {"interval_sec": 30.0, "score": 0.40}, + ] + result = collapse_point(rows, tolerance=0.1) + assert result["safe_interval_sec"] == 10.0 + assert result["collapse_interval_sec"] == 30.0 + + +def test_collapse_point_without_scores() -> None: + assert collapse_point([{"interval_sec": 1.0}])["safe_interval_sec"] is None diff --git a/tests/test_bench_fixtures.py b/tests/test_bench_fixtures.py new file mode 100644 index 0000000..f20d132 --- /dev/null +++ b/tests/test_bench_fixtures.py @@ -0,0 +1,171 @@ +"""Tests for synthetic fixtures, the ordering gate, and receipt hygiene.""" + +from __future__ import annotations + +import shutil +from pathlib import Path + +import pytest + +from av.bench.fixtures import ( + COLOR_PALETTE, + FIXTURE_KINDS, + MAX_N, + fixture_prompt, + generate_fixture, +) +from av.bench.receipts import ( + COMMUNITY, + MEASURED, + Claim, + Receipt, + redact_endpoint, + sha256_file, + write_receipt, +) +from av.bench.tasks.ordering import normalise_answer, score_ordering + +needs_ffmpeg = pytest.mark.skipif( + shutil.which("ffmpeg") is None, reason="ffmpeg not installed" +) + + +# --------------------------------------------------------------------------- +# Fixture generation +# --------------------------------------------------------------------------- + +@needs_ffmpeg +def test_color_fixture_frames_and_labels(tmp_path: Path) -> None: + fixture = generate_fixture("color", 4, tmp_path) + assert len(fixture.frame_paths) == 4 + assert fixture.labels == [name for name, _ in COLOR_PALETTE[:4]] + assert all(p.exists() and p.stat().st_size > 0 for p in fixture.frame_paths) + + +@needs_ffmpeg +def test_fixtures_are_byte_identical_across_runs(tmp_path: Path) -> None: + """Determinism is the whole point: the same fixture must hash the same.""" + first = generate_fixture("count", 3, tmp_path / "a") + second = generate_fixture("count", 3, tmp_path / "b") + assert [sha256_file(p) for p in first.frame_paths] == [ + sha256_file(p) for p in second.frame_paths + ] + + +@needs_ffmpeg +def test_frames_within_a_fixture_are_distinct(tmp_path: Path) -> None: + fixture = generate_fixture("motion", 4, tmp_path) + hashes = [sha256_file(p) for p in fixture.frame_paths] + assert len(set(hashes)) == 4 + + +@needs_ffmpeg +@pytest.mark.parametrize("kind", FIXTURE_KINDS) +def test_every_kind_generates(kind: str, tmp_path: Path) -> None: + fixture = generate_fixture(kind, 2, tmp_path / kind) + assert len(fixture.frame_paths) == 2 + assert fixture.ffmpeg_commands and all("ffmpeg" in c for c in fixture.ffmpeg_commands) + + +def test_fixture_rejects_unknown_kind(tmp_path: Path) -> None: + with pytest.raises(ValueError): + generate_fixture("rainbow", 2, tmp_path) + + +def test_fixture_rejects_degenerate_and_oversized(tmp_path: Path) -> None: + with pytest.raises(ValueError): + generate_fixture("color", 1, tmp_path) + with pytest.raises(ValueError): + generate_fixture("color", MAX_N["color"] + 1, tmp_path) + + +def test_prompt_lists_allowed_colour_names() -> None: + prompt = fixture_prompt("color", 3) + assert "magenta" in prompt + assert "3 images" in prompt + + +# --------------------------------------------------------------------------- +# Answer parsing and scoring +# --------------------------------------------------------------------------- + +@pytest.mark.parametrize( + "raw,expected", + [ + ("red, green, blue", ["red", "green", "blue"]), + ("Answer: red,green,blue", ["red", "green", "blue"]), + ("1. red\n2. green\n3. blue", ["red", "green", "blue"]), + ("`red, green`", ["red", "green"]), + ("", []), + ], +) +def test_normalise_colour_answers(raw: str, expected: list[str]) -> None: + assert normalise_answer(raw, "color") == expected + + +def test_normalise_numeric_answers() -> None: + assert normalise_answer("1, 2, 3, 4", "count") == ["1", "2", "3", "4"] + + +def test_score_exact_match() -> None: + exact, prefix = score_ordering(["red", "green"], ["red", "green"]) + assert exact and prefix == 2 + + +def test_score_records_prefix_on_truncated_answer() -> None: + """The upstream failure to detect: a model that reports only the first frame.""" + exact, prefix = score_ordering(["red", "blue", "green"], ["red"]) + assert not exact + assert prefix == 1 + + +def test_score_wrong_order_is_not_exact() -> None: + exact, prefix = score_ordering(["red", "blue"], ["blue", "red"]) + assert not exact + assert prefix == 0 + + +# --------------------------------------------------------------------------- +# Receipts +# --------------------------------------------------------------------------- + +def test_claim_requires_a_source_when_not_measured() -> None: + Claim(label=MEASURED, statement="ran it here") + with pytest.raises(ValueError): + Claim(label=COMMUNITY, statement="someone else's number") + Claim(label=COMMUNITY, statement="someone else's number", source="a published chart") + + +def test_claim_rejects_unknown_label() -> None: + with pytest.raises(ValueError): + Claim(label="probably", statement="...") + + +@pytest.mark.parametrize( + "url,expected", + [ + ("https://api.openai.com/v1", "api.openai.com"), + ("http://localhost:30000/v1", ""), + ("http://10.1.2.3:8000/v1", ""), + ("http://box.tail1234.ts.net:30000/v1", ""), + ("http://100.64.1.2:30000/v1", ""), + (None, None), + ], +) +def test_private_endpoints_never_reach_a_receipt(url: str | None, expected: str | None) -> None: + assert redact_endpoint(url) == expected + + +def test_receipt_round_trips_to_disk(tmp_path: Path) -> None: + receipt = Receipt(kind="gate", summary={"verdict": "pass"}) + receipt.add_claim(MEASURED, "it passed") + path = write_receipt(receipt, tmp_path) + assert path.exists() + + import json + + data = json.loads(path.read_text()) + assert data["kind"] == "gate" + assert data["claims"][0]["label"] == MEASURED + assert data["receipt_version"] >= 1 + assert "av_version" in data diff --git a/tests/test_bench_tasks.py b/tests/test_bench_tasks.py new file mode 100644 index 0000000..fed7402 --- /dev/null +++ b/tests/test_bench_tasks.py @@ -0,0 +1,266 @@ +"""Tests for task loading, scoring, arms, and public-benchmark adapters.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from av.bench.datasets import ( + ADAPTERS, + SOURCES, + adapt_lvbench, + adapt_minerva, + load_annotations, + subset_by_video, +) +from av.bench.tasks.events import EVENT_PROBES, parse_presence +from av.bench.tasks.videoqa import ( + Question, + build_question_prompt, + build_selection_prompt, + extract_choice, + load_questions, + parse_timestamps, + score_answer, +) + + +# --------------------------------------------------------------------------- +# Task file loading +# --------------------------------------------------------------------------- + +def _write_task(tmp_path: Path, rows: list[dict]) -> Path: + path = tmp_path / "task.jsonl" + path.write_text("\n".join(json.dumps(r) for r in rows) + "\n") + return path + + +def test_load_questions_resolves_relative_videos(tmp_path: Path) -> None: + path = _write_task(tmp_path, [ + {"id": "q1", "video": "clips/a.mp4", "question": "what?", "options": ["A. x", "B. y"], "answer": "B"} + ]) + questions = load_questions(path) + assert questions[0].video == (tmp_path / "clips/a.mp4").resolve() + assert questions[0].is_multiple_choice + + +def test_load_questions_honours_video_root(tmp_path: Path) -> None: + path = _write_task(tmp_path, [{"video": "a.mp4", "question": "q"}]) + root = tmp_path / "elsewhere" + assert load_questions(path, root)[0].video == (root / "a.mp4").resolve() + + +def test_load_questions_skips_blank_and_comment_lines(tmp_path: Path) -> None: + path = tmp_path / "task.jsonl" + path.write_text('# a note\n\n{"video": "a.mp4", "question": "q"}\n') + assert len(load_questions(path)) == 1 + + +def test_load_questions_reports_the_bad_line(tmp_path: Path) -> None: + path = tmp_path / "task.jsonl" + path.write_text('{"video": "a.mp4", "question": "q"}\nnot json\n') + with pytest.raises(ValueError, match="task.jsonl:2"): + load_questions(path) + + +# --------------------------------------------------------------------------- +# Scoring +# --------------------------------------------------------------------------- + +@pytest.mark.parametrize( + "reply,expected", + [("B", "B"), ("b", "B"), ("B.", "B"), ("The answer is C", "C"), ("", None), ("hmm", None)], +) +def test_extract_choice(reply: str, expected: str | None) -> None: + assert extract_choice(reply, ["A. x", "B. y", "C. z"]) == expected + + +def test_extract_choice_falls_back_to_option_text() -> None: + assert extract_choice("a red car", ["A. a blue van", "B. a red car"]) == "B" + + +def test_score_multiple_choice() -> None: + q = Question(id="q", video=Path("a.mp4"), question="?", options=["A. x", "B. y"], answer="B") + assert score_answer(q, "B")[0] + assert not score_answer(q, "A")[0] + + +def test_unparseable_reply_scores_incorrect_not_skipped() -> None: + q = Question(id="q", video=Path("a.mp4"), question="?", options=["A. x", "B. y"], answer="B") + correct, parsed = score_answer(q, "I cannot tell from these frames") + assert not correct + assert parsed is None + + +def test_score_open_ended_is_substring_match() -> None: + q = Question(id="q", video=Path("a.mp4"), question="?", answer="a red car") + assert score_answer(q, "It is a red car.")[0] + assert not score_answer(q, "a blue van")[0] + + +# --------------------------------------------------------------------------- +# Prompts and agentic selection +# --------------------------------------------------------------------------- + +def test_question_prompt_carries_timestamps() -> None: + q = Question(id="q", video=Path("a.mp4"), question="what?", options=["A. x"], answer="A") + prompt = build_question_prompt(q, [0.0, 1.0, 2.0]) + assert "0.0s, 1.0s, 2.0s" in prompt + assert "A. x" in prompt + + +def test_selection_prompt_withholds_the_answer_step() -> None: + q = Question(id="q", video=Path("a.mp4"), question="what?", answer="x") + prompt = build_selection_prompt(q, [0.0, 60.0], budget=4, duration=120.0) + assert "Do not answer it yet" in prompt + assert "at most 4" in prompt + + +@pytest.mark.parametrize( + "reply,expected", + [ + ('{"timestamps": [1.0, 2.5]}', [1.0, 2.5]), + ('```json\n{"timestamps": [3]}\n```', [3.0]), + ("I want 5 and 10 seconds", [5.0, 10.0]), + ("", []), + ], +) +def test_parse_timestamps(reply: str, expected: list[float]) -> None: + assert parse_timestamps(reply, duration=100.0, budget=8) == expected + + +def test_parse_timestamps_drops_out_of_range_and_respects_budget() -> None: + assert parse_timestamps('{"timestamps": [1, 5, 500]}', duration=100.0, budget=8) == [1.0, 5.0] + assert len(parse_timestamps('{"timestamps": [1,2,3,4,5]}', duration=100.0, budget=2)) == 2 + + +# --------------------------------------------------------------------------- +# Event presence parsing +# --------------------------------------------------------------------------- + +@pytest.mark.parametrize( + "reply,expected", + [ + ('{"present": true}', True), + ('{"present": false}', False), + ('```json\n{"present": true}\n```', True), + ('```json\n{"present', None), # truncated before the value + ('```json\n{"present": true', True), # truncated after it + ("yes", True), + ("no", False), + ("", None), + ], +) +def test_parse_presence(reply: str, expected: bool | None) -> None: + assert parse_presence(reply) is expected + + +def test_every_probe_has_a_question_and_terms() -> None: + for name, probe in EVENT_PROBES.items(): + assert probe["question"].endswith("?"), name + assert probe["reference_terms"], name + + +# --------------------------------------------------------------------------- +# Public benchmark adapters +# --------------------------------------------------------------------------- + +MINERVA_ROW = { + "key": "vid1:abc", + "video_id": "vid1", + "question": "What happens?", + "answer_choice_0": "nothing", + "answer_choice_1": "something", + "answer_choice_2": "everything", + "answer_id": 1, + "question_type": "Event Occurence", + "split": "Sports", + "category": "Basketball", +} + +LVBENCH_ROW = { + "key": "vidA", + "type": "cartoon", + "qa": [ + { + "uid": "55", + "question": "What year?\n(A) 1636\n(B) 1366\n(C) 1363\n(D) 1633", + "answer": "D", + "question_type": ["key information retrieval"], + "time_reference": "00:15-00:19", + } + ], +} + + +def test_adapt_minerva_maps_index_to_letter() -> None: + rows = adapt_minerva([MINERVA_ROW]) + assert len(rows) == 1 + assert rows[0].answer == "B" + assert rows[0].options[1] == "B. something" + assert rows[0].video_id == "vid1" + + +def test_adapt_minerva_skips_rows_with_a_bad_gold_index() -> None: + assert adapt_minerva([{**MINERVA_ROW, "answer_id": 9}]) == [] + + +def test_adapt_lvbench_splits_inline_options_and_timespan() -> None: + rows = adapt_lvbench([LVBENCH_ROW]) + assert len(rows) == 1 + row = rows[0] + assert row.question == "What year?" + assert row.options == ["A. 1636", "B. 1366", "C. 1363", "D. 1633"] + assert row.answer == "D" + assert (row.start_sec, row.end_sec) == (15.0, 19.0) + + +def test_adapt_lvbench_handles_hour_length_timespans() -> None: + row = {**LVBENCH_ROW} + row["qa"] = [{**LVBENCH_ROW["qa"][0], "time_reference": "01:02:03-01:02:10"}] + assert adapt_lvbench([row])[0].start_sec == 3723.0 + + +def test_task_row_uses_the_video_template() -> None: + row = adapt_minerva([MINERVA_ROW])[0].to_task_row("videos/{video_id}.mp4") + assert row["video"] == "videos/vid1.mp4" + assert row["meta"]["video_id"] == "vid1" + + +def test_subset_groups_by_video_to_limit_downloads() -> None: + rows = adapt_minerva([ + {**MINERVA_ROW, "key": f"v{v}:{q}", "video_id": f"v{v}"} + for v in range(4) for q in range(3) + ]) + chosen = subset_by_video(rows, max_questions=6, max_videos=2) + assert len({r.video_id for r in chosen}) == 2 + assert len(chosen) == 6 + + +def test_subset_is_deterministic() -> None: + rows = adapt_minerva([ + {**MINERVA_ROW, "key": f"v{v}:{q}", "video_id": f"v{v}"} + for v in range(4) for q in range(3) + ]) + first = [r.id for r in subset_by_video(rows, max_questions=5)] + second = [r.id for r in subset_by_video(rows, max_questions=5)] + assert first == second + + +def test_load_annotations_reads_both_json_and_jsonl(tmp_path: Path) -> None: + as_json = tmp_path / "a.json" + as_json.write_text(json.dumps([MINERVA_ROW])) + as_jsonl = tmp_path / "b.jsonl" + as_jsonl.write_text(json.dumps(LVBENCH_ROW) + "\n") + assert len(load_annotations(as_json)) == 1 + assert len(load_annotations(as_jsonl)) == 1 + + +def test_every_adapter_has_a_documented_source() -> None: + """Licences differ sharply between these datasets, so each must name its terms.""" + for name in ADAPTERS: + assert name in SOURCES + assert SOURCES[name]["annotations_licence"] + assert SOURCES[name]["annotations_url"].startswith("https://") diff --git a/tests/test_deepseek.py b/tests/test_deepseek.py new file mode 100644 index 0000000..8370a51 --- /dev/null +++ b/tests/test_deepseek.py @@ -0,0 +1,180 @@ +"""Tests for the DeepSeek-V4.1-Flash provider and its image-token model.""" + +from __future__ import annotations + +import os +from pathlib import Path + +import pytest + +from av.core.config import AVConfig +from av.core.constants import PROVIDER_PRESETS +from av.providers.deepseek import ( + DEFAULT_BASE_URL, + DEFAULT_MODEL, + MAX_IMAGE_TOKENS, + MIN_IMAGE_TOKENS, + PLACEHOLDER_KEY, + CapabilityRecord, + make_config, + plan_image_tokens, + resolve_api_key, + resolve_base_url, + tokens_for_grid, + widths_for_token_budgets, +) +from av.providers.openai import _client, _resolve_api_key + + +# --------------------------------------------------------------------------- +# Preset and configuration +# --------------------------------------------------------------------------- + +def test_preset_is_registered() -> None: + preset = PROVIDER_PRESETS["deepseek"] + assert preset["vision_model"] == DEFAULT_MODEL + assert preset["transcribe_model"] == "" # not served by this deployment + assert preset["embed_model"] == "" + + +def test_preset_ships_no_private_endpoint() -> None: + """A hosted endpoint must never be baked into source; only a local placeholder.""" + url = PROVIDER_PRESETS["deepseek"]["api_base_url"] + assert url == DEFAULT_BASE_URL + assert "localhost" in url + + +def test_base_url_prefers_env_then_config(monkeypatch: pytest.MonkeyPatch) -> None: + config = AVConfig(provider="deepseek", api_base_url="http://from-config:30000/v1") + monkeypatch.delenv("AV_API_BASE_URL", raising=False) + assert resolve_base_url(config) == "http://from-config:30000/v1" + + monkeypatch.setenv("AV_API_BASE_URL", "http://from-env:30000/v1") + assert resolve_base_url(config) == "http://from-env:30000/v1" + + +def test_base_url_falls_back_to_the_local_default(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv("AV_API_BASE_URL", raising=False) + assert resolve_base_url(None) == DEFAULT_BASE_URL + + +def test_api_key_resolution_order(monkeypatch: pytest.MonkeyPatch) -> None: + for name in ("AV_API_KEY", "DEEPSEEK_API_KEY", "SGLANG_API_KEY"): + monkeypatch.delenv(name, raising=False) + + assert resolve_api_key(AVConfig(api_key="explicit")) == "explicit" + + monkeypatch.setenv("DEEPSEEK_API_KEY", "from-env") + assert resolve_api_key(AVConfig(api_key="")) == "from-env" + assert resolve_api_key(AVConfig(api_key=PLACEHOLDER_KEY)) == "from-env" + + +def test_api_key_placeholder_when_server_needs_none(monkeypatch: pytest.MonkeyPatch) -> None: + for name in ("AV_API_KEY", "DEEPSEEK_API_KEY", "SGLANG_API_KEY"): + monkeypatch.delenv(name, raising=False) + assert resolve_api_key(AVConfig(api_key="")) == PLACEHOLDER_KEY + + +def test_openai_client_routes_deepseek_key(monkeypatch: pytest.MonkeyPatch) -> None: + """The shared OpenAI-compatible client must pick up the provider's own env var.""" + monkeypatch.delenv("AV_API_KEY", raising=False) + monkeypatch.setenv("DEEPSEEK_API_KEY", "sk-deepseek") + config = AVConfig(provider="deepseek", api_key="") + assert _resolve_api_key(config) == "sk-deepseek" + + +def test_make_config_disables_unserved_stages(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.delenv("AV_API_BASE_URL", raising=False) + config = make_config(model="deepseek-v4.1-flash") + assert config.provider == "deepseek" + assert config.transcribe_model == "" + assert config.embed_model == "" + assert config.vision_model == config.chat_model == "deepseek-v4.1-flash" + + +def test_client_sends_no_anthropic_header() -> None: + config = AVConfig(provider="deepseek", api_key="k", api_base_url=DEFAULT_BASE_URL) + assert _client(config)._custom_headers.get("anthropic-version") is None + + +# --------------------------------------------------------------------------- +# Image token model +# --------------------------------------------------------------------------- + +def test_token_formula() -> None: + """One token per grid cell, one newline per row, plus start and end.""" + assert tokens_for_grid(13, 13) == 13 * 14 + 2 == 184 + + +@pytest.mark.parametrize( + "width,height,expected_tokens", + [ + (512, 512, 184), # below the pixel floor, upscaled + (640, 480, 206), + (800, 600, 317), + (1024, 1024, 652), + (1280, 720, 578), + (100, 3000, 290), # extreme aspect ratio + ], +) +def test_plan_reproduces_published_examples(width: int, height: int, expected_tokens: int) -> None: + assert plan_image_tokens(width, height).tokens == expected_tokens + + +def test_small_frames_hit_the_upscale_floor() -> None: + """Shrinking past the floor buys nothing: a thumbnail costs what 544x544 costs.""" + tiny = plan_image_tokens(64, 64) + small = plan_image_tokens(512, 512) + assert tiny.upscaled and small.upscaled + assert tiny.tokens == small.tokens == MIN_IMAGE_TOKENS + + +def test_huge_frames_are_capped_and_flagged_approximate() -> None: + big = plan_image_tokens(5000, 5000) + assert big.tokens <= MAX_IMAGE_TOKENS + assert big.downscaled + assert big.approximate # a shrink search ran; measure it rather than trust it + + +def test_extra_resolution_past_the_ceiling_is_discarded() -> None: + assert plan_image_tokens(2000, 2000).tokens == plan_image_tokens(5000, 5000).tokens + + +def test_tokens_rise_with_resolution_between_the_walls() -> None: + counts = [plan_image_tokens(w, int(w * 9 / 16)).tokens for w in (768, 1024, 1280)] + assert counts == sorted(counts) + assert len(set(counts)) == len(counts) + + +def test_plan_rejects_degenerate_sizes() -> None: + with pytest.raises(ValueError): + plan_image_tokens(0, 100) + + +def test_widths_for_budgets_stay_within_budget() -> None: + solved = widths_for_token_budgets([200, 400, 800]) + for budget, width in solved.items(): + height = max(int(round(width / (16 / 9))), 1) + assert plan_image_tokens(width, height).tokens <= budget + assert solved[200] < solved[400] < solved[800] + + +# --------------------------------------------------------------------------- +# Capability record +# --------------------------------------------------------------------------- + +def test_capability_record_defaults_to_unestablished() -> None: + """Nothing is assumed: a field we have not measured stays None.""" + record = CapabilityRecord() + assert record.context_tokens is None + assert record.image_tokens_per_frame is None + assert record.multi_image_supported is None + assert record.max_frames_per_request() is None + + +def test_max_frames_per_request_when_both_inputs_known() -> None: + record = CapabilityRecord(context_tokens=1_048_576, image_tokens_per_frame=1024) + frames = record.max_frames_per_request(prompt_overhead_tokens=0) + assert frames == 1024 + # 1,024 frames at 1 fps is roughly 17 minutes in a single request. + assert 16.5 < frames / 60 < 17.5