Skip to content

Commit 5599525

Browse files
author
OpenRouter SDK Bot
committed
chore: update OpenAPI spec [sdk-bot]
1 parent 43d88e9 commit 5599525

1 file changed

Lines changed: 160 additions & 9 deletions

File tree

‎.speakeasy/in.openapi.yaml‎

Lines changed: 160 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -24305,23 +24305,130 @@ components:
2430524305
properties:
2430624306
data:
2430724307
items:
24308-
discriminator:
24309-
mapping:
24310-
artificial-analysis: '#/components/schemas/UnifiedBenchmarksAAItem'
24311-
design-arena: '#/components/schemas/UnifiedBenchmarksDAItem'
24312-
openrouter: '#/components/schemas/UnifiedBenchmarksORItem'
24313-
propertyName: 'source'
2431424308
oneOf:
2431524309
- $ref: '#/components/schemas/UnifiedBenchmarksAAItem'
2431624310
- $ref: '#/components/schemas/UnifiedBenchmarksDAItem'
2431724311
- $ref: '#/components/schemas/UnifiedBenchmarksORItem'
24312+
- $ref: '#/components/schemas/UnifiedBenchmarksSearchItem'
2431824313
type: 'array'
2431924314
meta:
2432024315
$ref: '#/components/schemas/UnifiedBenchmarksMeta'
2432124316
required:
2432224317
- 'data'
2432324318
- 'meta'
2432424319
type: 'object'
24320+
UnifiedBenchmarksSearchItem:
24321+
properties:
24322+
avg_cost_per_task:
24323+
description: 'Average cost per task in USD, or null if unavailable.'
24324+
example: 0.031
24325+
format: 'double'
24326+
type:
24327+
- 'number'
24328+
- 'null'
24329+
avg_latency_per_task_ms:
24330+
description: 'Average wall-clock latency per task in milliseconds, or null if unavailable.'
24331+
example: 45210
24332+
format: 'double'
24333+
type:
24334+
- 'number'
24335+
- 'null'
24336+
benchmark_type:
24337+
description: 'OpenRouter search benchmark.'
24338+
enum:
24339+
- 'search_browsecomp'
24340+
- 'search_hle'
24341+
- 'search_dsqa'
24342+
- 'search_widesearch'
24343+
example: 'search_browsecomp'
24344+
type: 'string'
24345+
display_name:
24346+
description: 'Human-readable model name.'
24347+
example: 'GPT-4o'
24348+
type: 'string'
24349+
last_run_timestamp:
24350+
description: 'Timestamp of the newest qualifying run in the published configuration''s lane.'
24351+
example: '2026-07-28T13:38:18Z'
24352+
type: 'string'
24353+
model_permaslug:
24354+
description: 'Stable OpenRouter model identifier.'
24355+
example: 'openai/gpt-4o'
24356+
type: 'string'
24357+
primary_metric:
24358+
description: 'Identifies the meaning of `primary_score`: `f1_by_item` for WideSearch, `accuracy` for all other search benchmarks.'
24359+
enum:
24360+
- 'accuracy'
24361+
- 'f1_by_item'
24362+
example: 'accuracy'
24363+
type: 'string'
24364+
primary_score:
24365+
description: 'The benchmark''s headline score from 0 to 1. Its meaning is identified by `primary_metric`: item-weighted F1 for WideSearch or strict accuracy for the other search benchmarks. Higher is better.'
24366+
example: 0.72
24367+
format: 'double'
24368+
type: 'number'
24369+
run_config:
24370+
$ref: '#/components/schemas/UnifiedBenchmarksSearchRunConfig'
24371+
search_engine:
24372+
description: 'Search engine the published configuration used.'
24373+
example: 'exa'
24374+
type: 'string'
24375+
search_surface:
24376+
description: 'Request surface the published configuration went through.'
24377+
enum:
24378+
- 'server-tool'
24379+
- 'plugin'
24380+
example: 'server-tool'
24381+
type: 'string'
24382+
source:
24383+
description: 'Benchmark source discriminator.'
24384+
enum:
24385+
- 'openrouter'
24386+
type: 'string'
24387+
total_tasks:
24388+
description: 'Tasks evaluated across the published configuration''s runs.'
24389+
example: 100
24390+
type: 'integer'
24391+
required:
24392+
- 'source'
24393+
- 'model_permaslug'
24394+
- 'display_name'
24395+
- 'benchmark_type'
24396+
- 'primary_metric'
24397+
- 'primary_score'
24398+
- 'total_tasks'
24399+
- 'avg_cost_per_task'
24400+
- 'avg_latency_per_task_ms'
24401+
- 'search_engine'
24402+
- 'search_surface'
24403+
- 'last_run_timestamp'
24404+
type: 'object'
24405+
UnifiedBenchmarksSearchRunConfig:
24406+
description: 'Published lane configuration, included only when include_run_config=true. Only the agent turn count, reasoning effort, and temperature are exposed; other harness settings are intentionally not part of the public contract.'
24407+
properties:
24408+
max_agent_turns:
24409+
description: 'Agent-turn count for the published lane, or null for plugin lanes.'
24410+
example: 25
24411+
type:
24412+
- 'integer'
24413+
- 'null'
24414+
reasoning_effort:
24415+
description: 'Reasoning effort configured for the published lane, or null when omitted.'
24416+
example: 'high'
24417+
type:
24418+
- 'string'
24419+
- 'null'
24420+
temperature:
24421+
description: 'Sampling temperature configured for the published lane, or null when omitted.'
24422+
example: 0.2
24423+
format: 'double'
24424+
type:
24425+
- 'number'
24426+
- 'null'
24427+
required:
24428+
- 'max_agent_turns'
24429+
- 'reasoning_effort'
24430+
- 'temperature'
24431+
type: 'object'
2432524432
UnprocessableEntityResponse:
2432624433
description: 'Unprocessable Entity - Semantic validation failure'
2432724434
example:
@@ -27158,7 +27265,7 @@ paths:
2715827265
x-speakeasy-name-override: 'createAuthCode'
2715927266
/benchmarks:
2716027267
get:
27161-
description: 'Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter''s own tau-bench and GPQA evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.'
27268+
description: 'Unified benchmark endpoint that aggregates scores from multiple benchmark sources (Artificial Analysis, Design Arena, and OpenRouter''s own tau-bench, GPQA, and web-search evals). Filter by source to reproduce the exact shapes from the legacy per-source endpoints, or use task_type to find models suited for specific workloads. Use task_type=search (or a search_* benchmark_type) for OpenRouter''s search benchmarks, which publish each model''s highest-scoring eligible evaluation configuration with same-configuration runs combined by task-weighted mean. Authenticate with any valid OpenRouter API key. Rate-limited to 30 requests/minute per key and 500 requests/day per account.'
2716227269
operationId: 'getBenchmarks'
2716327270
parameters:
2716427271
- description: 'Benchmark source to query. Determines the shape of the returned items. When omitted, returns results from all sources.'
@@ -27173,18 +27280,62 @@ paths:
2717327280
- 'openrouter'
2717427281
example: 'artificial-analysis'
2717527282
type: 'string'
27176-
- description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.'
27283+
- description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.'
2717727284
in: 'query'
2717827285
name: 'task_type'
2717927286
required: false
2718027287
schema:
27181-
description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category.'
27288+
description: 'Filter results by task type. For Artificial Analysis, maps to the corresponding index. For Design Arena, maps to the matching category. `search` returns OpenRouter search benchmark results only.'
2718227289
enum:
2718327290
- 'coding'
2718427291
- 'intelligence'
2718527292
- 'agentic'
27293+
- 'search'
2718627294
example: 'coding'
2718727295
type: 'string'
27296+
- description: 'Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources'' items as they are.'
27297+
in: 'query'
27298+
name: 'benchmark_type'
27299+
required: false
27300+
schema:
27301+
description: 'Return results for one exact OpenRouter benchmark. A `search_*` value narrows the response to search results only; a classic value narrows the OpenRouter items and leaves other sources'' items as they are.'
27302+
enum:
27303+
- 'gpqa_diamond'
27304+
- 'tau_bench_verified_airline'
27305+
- 'search_browsecomp'
27306+
- 'search_hle'
27307+
- 'search_dsqa'
27308+
- 'search_widesearch'
27309+
example: 'search_widesearch'
27310+
type: 'string'
27311+
- description: 'Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract.'
27312+
in: 'query'
27313+
name: 'include_run_config'
27314+
required: false
27315+
schema:
27316+
default: false
27317+
description: 'Search benchmarks only: include the published lane configuration whitelist in each search item. Defaults to false. The whitelist is limited to agent turn count, reasoning effort, and temperature so future harness configuration changes do not change the public contract.'
27318+
example: true
27319+
type: 'boolean'
27320+
- description: 'OpenRouter search benchmarks only: filter by the search engine used.'
27321+
in: 'query'
27322+
name: 'search_engine'
27323+
required: false
27324+
schema:
27325+
description: 'OpenRouter search benchmarks only: filter by the search engine used.'
27326+
example: 'exa'
27327+
type: 'string'
27328+
- description: 'OpenRouter search benchmarks only: filter by the request surface the lane ran on.'
27329+
in: 'query'
27330+
name: 'search_surface'
27331+
required: false
27332+
schema:
27333+
description: 'OpenRouter search benchmarks only: filter by the request surface the lane ran on.'
27334+
enum:
27335+
- 'server-tool'
27336+
- 'plugin'
27337+
example: 'server-tool'
27338+
type: 'string'
2718827339
- description: 'Design Arena only: arena to query. Defaults to `models` when source is `design-arena`.'
2718927340
in: 'query'
2719027341
name: 'arena'

0 commit comments

Comments
 (0)