Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 15 additions & 0 deletions .add/tasks/eval-run-executor.d/runs/1.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
---
type: Run
runtime: process
task: /tasks/eval-run-executor.md
computation: "uv run pytest tests/evals_runs/test_eval_run_executor.py -q --no-cov -p no:randomly --junitxml tests/evals_runs/_add_junit.xml"
receipt:
kind: command-exit
ids: unknown
exit: 0
freshness: mtime
at: 2026-08-13
stdout: '10 passed in 5.76s'
note: ''
generated: { by: process:run, at: 2026-08-13 }
---
26 changes: 26 additions & 0 deletions .add/tasks/eval-run-executor.d/runs/2.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,26 @@
---
type: Run
runtime: process
task: /tasks/eval-run-executor.md
computation: "uv run pytest tests/evals_runs/test_eval_run_executor.py -q --no-cov -p no:randomly --junitxml=/private/tmp/claude-501/-Users-tindang-workspaces-tind-repo-ai-proxy/88454d31-bfec-4421-87a2-2d625a0ab229/scratchpad/evals_runs_add.xml"
receipt:
kind: test-ids
ids: 10/10 reported
exit: 0
freshness: mtime
at: 2026-08-13
stdout: '10 passed in 5.65s'
note: ''
passed:
- tests.evals_runs.test_eval_run_executor::test_cross_tenant_and_absent_run_uniform_404
- tests.evals_runs.test_eval_run_executor::test_empty_set_run_completes_vacuously
- tests.evals_runs.test_eval_run_executor::test_one_usage_record_per_dialed_case
- tests.evals_runs.test_eval_run_executor::test_over_budget_run_refuses_every_case_no_dial
- tests.evals_runs.test_eval_run_executor::test_per_tenant_breaker_isolation
- tests.evals_runs.test_eval_run_executor::test_results_aligned_to_case_creation_order
- tests.evals_runs.test_eval_run_executor::test_resume_does_not_rebill_terminal_cases
- tests.evals_runs.test_eval_run_executor::test_run_enters_governance_per_case
- tests.evals_runs.test_eval_run_executor::test_timeout_case_errored_run_continues
- tests.evals_runs.test_eval_run_executor::test_zdr_run_refused_atomically_zero_results
generated: { by: process:run, at: 2026-08-13 }
---
29 changes: 29 additions & 0 deletions .add/tasks/eval-run-executor.d/runs/3.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
---
type: Run
runtime: process
task: /tasks/eval-run-executor.md
computation: "uv run pytest tests/evals_runs/test_eval_run_executor.py -q --no-cov -p no:randomly --junitxml=/private/tmp/claude-501/-Users-tindang-workspaces-tind-repo-ai-proxy/88454d31-bfec-4421-87a2-2d625a0ab229/scratchpad/evals_runs_final.xml"
receipt:
kind: test-ids
ids: 13/13 reported
exit: 0
freshness: mtime
at: 2026-08-13
stdout: '13 passed in 7.19s'
note: ''
passed:
- tests.evals_runs.test_eval_run_executor::test_bounded_per_tenant_concurrency
- tests.evals_runs.test_eval_run_executor::test_cross_tenant_and_absent_run_uniform_404
- tests.evals_runs.test_eval_run_executor::test_cross_tenant_launch_and_billing_identity
- tests.evals_runs.test_eval_run_executor::test_empty_set_run_completes_vacuously
- tests.evals_runs.test_eval_run_executor::test_one_usage_record_per_dialed_case
- tests.evals_runs.test_eval_run_executor::test_over_budget_run_refuses_every_case_no_dial
- tests.evals_runs.test_eval_run_executor::test_per_tenant_breaker_isolation
- tests.evals_runs.test_eval_run_executor::test_results_aligned_to_case_creation_order
- tests.evals_runs.test_eval_run_executor::test_resume_does_not_rebill_terminal_cases
- tests.evals_runs.test_eval_run_executor::test_run_enters_governance_per_case
- tests.evals_runs.test_eval_run_executor::test_run_snapshot_fixed_at_launch
- tests.evals_runs.test_eval_run_executor::test_timeout_case_errored_run_continues
- tests.evals_runs.test_eval_run_executor::test_zdr_run_refused_atomically_zero_results
generated: { by: process:run, at: 2026-08-13 }
---
121 changes: 104 additions & 17 deletions .add/tasks/eval-run-executor.md

Large diffs are not rendered by default.

1 change: 1 addition & 0 deletions apps/gateway/migrations/env.py
Original file line number Diff line number Diff line change
Expand Up @@ -111,6 +111,7 @@
import gateway.domain_capture.infrastructure.orm # noqa: F401 — tenant_domain_claims
import gateway.files.infrastructure.orm # noqa: F401 — files
import gateway.evals.infrastructure.orm # noqa: F401 — eval_sets, eval_cases
import gateway.evals.runs.infrastructure.orm # noqa: F401 — eval_runs, eval_case_results
import gateway.finetune.infrastructure.orm # noqa: F401 — finetune_jobs, finetune_job_events
import gateway.guardrail_analytics.infrastructure.orm # noqa: F401 — guardrail_verdict_events
import gateway.payments.infrastructure.orm # noqa: F401 — checkout_sessions
Expand Down
96 changes: 96 additions & 0 deletions apps/gateway/migrations/versions/f5b2d8c41a37_eval_run_executor.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,96 @@
"""eval run + per-case result substrate — eval-run-executor (R7 evals-regression-gate).

Revision ID: f5b2d8c41a37
Revises: e4a1c9d27f60
Create Date: 2026-08-13

eval-run-executor PLAN.md §3 (FROZEN @ sha256:1353206d). Additive, two tenant-scoped tables
that extend the eval-set-store substrate:

- NEW TABLE eval_runs — one row per launched run. Carries the launching key_id (A1 — a run
bills that key, exactly as its live traffic) and the named model. ``status`` is DERIVED
from the run's cases (M7): pending at launch, completed when every snapshot case is
terminal, blocked iff a ZDR flip refused the run mid-flight (M5). FK -> eval_sets CASCADE.
⚠ The raw API key is NEVER persisted (auth-scoped resume, 2026-08-13): only key_id, so a
cross-process resume must re-supply the raw key via a fresh authenticated request.
- NEW TABLE eval_case_results — one row per case DRIVEN. ``response_text`` is the model's
payload-at-rest (the ZDR-gated surface, M5); present only for a `completed` case.
UNIQUE (eval_run_id, eval_case_id) makes a resumed drive idempotent — a terminal case is
never re-dialed or re-billed (M7 / R:DOUBLE_BILL). FK -> eval_runs CASCADE.

Four-manifest rule (see [[gateway-new-table-four-manifests]]): this migration + EXPECTED_TABLES
(tests/migrations) + migrations/env.py's own ORM import + the guardrails NOT-IN allow-list.

Indexes (also declared in the ORM __table_args__ per the v30 two-manifest lesson):
ix_eval_runs_tenant_set_created on eval_runs(tenant_id, eval_set_id, created_at)
ix_eval_case_results_run_created on eval_case_results(tenant_id, eval_run_id, created_at)
"""

from __future__ import annotations

import sqlalchemy as sa
from alembic import op

# revision identifiers, used by Alembic.
revision = "f5b2d8c41a37"
down_revision = "e4a1c9d27f60"
branch_labels = None
depends_on = None


def upgrade() -> None:
op.create_table(
"eval_runs",
sa.Column("id", sa.UUID(), nullable=False, server_default=sa.text("gen_random_uuid()")),
sa.Column("tenant_id", sa.UUID(), nullable=False),
sa.Column("eval_set_id", sa.UUID(), nullable=False),
sa.Column("key_id", sa.UUID(), nullable=False),
sa.Column("model", sa.Text(), nullable=False),
sa.Column("status", sa.Text(), nullable=False),
sa.Column(
"created_at",
sa.DateTime(timezone=True),
nullable=False,
server_default=sa.text("now()"),
),
sa.PrimaryKeyConstraint("id"),
sa.ForeignKeyConstraint(["eval_set_id"], ["eval_sets.id"], ondelete="CASCADE"),
)
op.create_index(
"ix_eval_runs_tenant_set_created",
"eval_runs",
["tenant_id", "eval_set_id", "created_at"],
)

op.create_table(
"eval_case_results",
sa.Column("id", sa.UUID(), nullable=False, server_default=sa.text("gen_random_uuid()")),
sa.Column("tenant_id", sa.UUID(), nullable=False),
sa.Column("eval_run_id", sa.UUID(), nullable=False),
sa.Column("eval_case_id", sa.UUID(), nullable=False),
sa.Column("status", sa.Text(), nullable=False),
sa.Column("response_text", sa.Text(), nullable=True),
sa.Column("reason", sa.Text(), nullable=True),
sa.Column("usage_record_id", sa.UUID(), nullable=True),
sa.Column(
"created_at",
sa.DateTime(timezone=True),
nullable=False,
server_default=sa.text("now()"),
),
sa.PrimaryKeyConstraint("id"),
sa.ForeignKeyConstraint(["eval_run_id"], ["eval_runs.id"], ondelete="CASCADE"),
sa.UniqueConstraint("eval_run_id", "eval_case_id", name="uq_eval_case_results_run_case"),
)
op.create_index(
"ix_eval_case_results_run_created",
"eval_case_results",
["tenant_id", "eval_run_id", "created_at"],
)


def downgrade() -> None:
op.drop_index("ix_eval_case_results_run_created", table_name="eval_case_results")
op.drop_table("eval_case_results")
op.drop_index("ix_eval_runs_tenant_set_created", table_name="eval_runs")
op.drop_table("eval_runs")
9 changes: 9 additions & 0 deletions apps/gateway/src/gateway/core/error_catalog.py
Original file line number Diff line number Diff line change
Expand Up @@ -1689,3 +1689,12 @@ def exc(
"ERR_EVAL_SET_NAME_CONFLICT",
"an eval set with this name already exists for this tenant",
)

#: GET /v1/evals/runs/{id} (or .../cases) whose run id is absent OR owned by another tenant
#: — deliberately indistinguishable (M6, R:RUN_NOT_FOUND; never an enumeration oracle),
#: byte-identical for both causes, exactly like EVAL_SET_NOT_FOUND above.
EVAL_RUN_NOT_FOUND = ErrorSpec(404, "ERR_EVAL_RUN_NOT_FOUND", "Eval run not found")

#: POST .../runs whose `model` is missing or not a non-empty string — 422, nothing launched.
#: Validated on the caller's own input BEFORE the parent set is resolved (no oracle).
EVAL_RUN_INVALID = ErrorSpec(422, "ERR_EVAL_RUN_INVALID", "model must be a non-empty string")
Empty file.
Empty file.
208 changes: 208 additions & 0 deletions apps/gateway/src/gateway/evals/runs/api/run_router.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,208 @@
"""FastAPI router for eval-run execution (/v1/evals/.../runs) — eval-run-executor §3.

Reuses the eval-set-store surface's auth + OpenAI-wire envelope wholesale (``_authenticate``,
``_extract_raw_key``, ``_err``, ``_err_from_problem``, ``_unix`` from
``gateway.evals.api.router``) so the whole /v1/evals surface speaks ONE ``{"error": {...}}``
body, and tenant scope is enforced identically.

Endpoints:
POST /v1/evals/sets/{set_id}/runs {model} -> 201 { id:"er_..", eval_set_id, model, status,
case_count, created_at }
403 ERR_ZDR_PAYLOAD_BLOCKED (ZDR tenant, refused outright at launch, M5)
404 ERR_EVAL_SET_NOT_FOUND (absent OR cross-tenant set, M6)
422 ERR_EVAL_RUN_INVALID (missing/blank model, validated before the set is resolved)
GET /v1/evals/runs/{run_id} -> 200 { id, eval_set_id, model, status, case_count,
counts:{completed,refused,errored,pending}, created_at }
GET /v1/evals/runs/{run_id}/cases -> 200 { object:"list", data:[{eval_case_id, status,
response_text?, reason?, usage_record_id?}] } (A5 order)

Durability (M7): launch enqueues onto ``app.state.eval_run_queue`` when present and returns a
``pending`` run fast; a background worker drives it. When the queue is absent or enqueue fails
(Redis down, or ASGITransport tests), it FAILS OPEN to an inline drive — the vector-store
ingest idiom — so a run is never dropped. The raw key is passed straight through to the inline
drive; it is never persisted (auth-scoped resume).
"""

from __future__ import annotations

import logging
from typing import Annotated, Any

from fastapi import APIRouter, Depends, Request
from fastapi.responses import JSONResponse
from sqlalchemy.ext.asyncio import AsyncSession

from gateway.core.db import get_session
from gateway.core.error_catalog import EVAL_RUN_INVALID, EVAL_RUN_NOT_FOUND, EVAL_SET_NOT_FOUND
from gateway.core.errors import ProblemError

# The eval-set-store router owns the /v1/evals auth + OpenAI-wire envelope; reusing its helpers
# is what keeps the WHOLE surface speaking one body (a fork would drift). Same sanctioned reuse
# pattern as images/audio/embeddings use_cases importing use_cases._fire_record_with_raw.
from gateway.evals.api.router import (
_authenticate, # pyright: ignore[reportPrivateUsage]
_err, # pyright: ignore[reportPrivateUsage]
_err_from_problem, # pyright: ignore[reportPrivateUsage]
_extract_raw_key, # pyright: ignore[reportPrivateUsage]
_unix, # pyright: ignore[reportPrivateUsage]
)
from gateway.evals.infrastructure.repository import SqlAlchemyEvalStore
from gateway.evals.runs.infrastructure.orm import EvalCaseResultRow, EvalRunRow
from gateway.evals.runs.infrastructure.repository import SqlAlchemyEvalRunStore
from gateway.evals.wire_id import (
parse_run_wire_id,
parse_set_wire_id,
to_case_wire_id,
to_run_wire_id,
to_set_wire_id,
)
from gateway.keys.domain.entities import AuthzResult

eval_runs_router = APIRouter(tags=["evals"])

_log = logging.getLogger(__name__)


def _run_store(request: Request) -> SqlAlchemyEvalRunStore:
return SqlAlchemyEvalRunStore(request.app.state.sessionmaker)


def _run_object(row: EvalRunRow, *, case_count: int) -> dict[str, Any]:
return {
"id": to_run_wire_id(row.id),
"object": "eval.run",
"created_at": _unix(row.created_at),
"eval_set_id": to_set_wire_id(row.eval_set_id),
"model": row.model,
"status": row.status,
"case_count": case_count,
}


def _case_result_object(row: EvalCaseResultRow) -> dict[str, Any]:
"""The per-case result wire object. response_text is present only for a `completed` case."""
obj: dict[str, Any] = {
"object": "eval.case_result",
"eval_case_id": to_case_wire_id(row.eval_case_id),
"status": row.status,
}
if row.response_text is not None:
obj["response_text"] = row.response_text
if row.reason is not None:
obj["reason"] = row.reason
if row.usage_record_id is not None:
obj["usage_record_id"] = str(row.usage_record_id)
return obj


@eval_runs_router.post("/v1/evals/sets/{set_id}/runs", status_code=201, response_model=None)
async def launch_eval_run(
set_id: str,
body: dict[str, Any],
request: Request,
authz: Annotated[AuthzResult, Depends(_authenticate)],
session: Annotated[AsyncSession, Depends(get_session)],
) -> dict[str, Any] | JSONResponse:
"""Launch a run of a tenant's set against a model (M1). ZDR tenant refused outright (M5)."""
# Validate the caller's own input FIRST — reveals nothing about whether the set exists.
model = body.get("model")
if not isinstance(model, str) or not model.strip():
return _err(EVAL_RUN_INVALID)

resolved_set = parse_set_wire_id(set_id)
if resolved_set is None:
return _err(EVAL_SET_NOT_FOUND)
# M6: resolve the parent set in tenant scope — absent/cross-tenant is a uniform 404.
parent = await SqlAlchemyEvalStore(session).get_set(
tenant_id=authz.tenant_id, eval_set_id=resolved_set
)
if parent is None:
return _err(EVAL_SET_NOT_FOUND)

executor = request.app.state.eval_run_executor
raw_key = _extract_raw_key(request)
try:
run = await executor.launch(
tenant_id=authz.tenant_id,
key_id=authz.key_id,
raw_key=raw_key,
eval_set_id=resolved_set,
model=model,
)
except ProblemError as exc:
# M5: the ZDR gate refused the run outright at launch. Re-render in this surface's
# envelope (403 ERR_ZDR_PAYLOAD_BLOCKED); nothing was created.
return _err_from_problem(exc)

# Snapshot denominator (A2): cases that existed at launch time.
store = _run_store(request)
snapshot = await store.snapshot_cases(
tenant_id=run.tenant_id, eval_set_id=run.eval_set_id, created_at_max=run.created_at
)
case_count = len(snapshot)

await _enqueue_or_drive(request, executor, run_id=run.id, raw_key=raw_key)

# Re-read so the returned status reflects an inline drive (fail-open / test) if it ran.
refreshed = await store.get_run(tenant_id=run.tenant_id, run_id=run.id) or run
return _run_object(refreshed, case_count=case_count)


async def _enqueue_or_drive(request: Request, executor: Any, *, run_id: Any, raw_key: str) -> None:
"""Enqueue for the durable worker; FAIL OPEN to an inline drive (vector-store idiom, M7)."""
queue = getattr(request.app.state, "eval_run_queue", None)
if queue is not None:
try:
await queue.enqueue(run_id)
return
except Exception:
_log.warning(
"eval_run: enqueue failed for run %s, failing open to inline drive", run_id
)
await executor.drive(run_id, raw_key=raw_key)


@eval_runs_router.get("/v1/evals/runs/{run_id}", status_code=200, response_model=None)
async def get_eval_run(
run_id: str,
request: Request,
authz: Annotated[AuthzResult, Depends(_authenticate)],
) -> dict[str, Any] | JSONResponse:
"""A run's status + per-status rollup (M6-scoped; uniform 404 for absent/cross-tenant)."""
resolved = parse_run_wire_id(run_id)
if resolved is None:
return _err(EVAL_RUN_NOT_FOUND)
store = _run_store(request)
run = await store.get_run(tenant_id=authz.tenant_id, run_id=resolved)
if run is None:
return _err(EVAL_RUN_NOT_FOUND)

snapshot = await store.snapshot_cases(
tenant_id=run.tenant_id, eval_set_id=run.eval_set_id, created_at_max=run.created_at
)
case_count = len(snapshot)
counts = await store.counts_by_status(run.id)
terminal = counts["completed"] + counts["refused"] + counts["errored"]
counts["pending"] = max(0, case_count - terminal)

obj = _run_object(run, case_count=case_count)
obj["counts"] = counts
return obj


@eval_runs_router.get("/v1/evals/runs/{run_id}/cases", status_code=200, response_model=None)
async def list_eval_run_cases(
run_id: str,
request: Request,
authz: Annotated[AuthzResult, Depends(_authenticate)],
) -> dict[str, Any] | JSONResponse:
"""A run's per-case results in the set's creation order (A5). Uniform 404 (M6)."""
resolved = parse_run_wire_id(run_id)
if resolved is None:
return _err(EVAL_RUN_NOT_FOUND)
store = _run_store(request)
run = await store.get_run(tenant_id=authz.tenant_id, run_id=resolved)
if run is None:
return _err(EVAL_RUN_NOT_FOUND)
results = await store.list_case_results(tenant_id=authz.tenant_id, run_id=resolved)
return {"object": "list", "data": [_case_result_object(r) for r in results]}
Empty file.
Loading
Loading