diff --git a/.speakeasy/gen.lock b/.speakeasy/gen.lock index b9228b68..5327a917 100644 --- a/.speakeasy/gen.lock +++ b/.speakeasy/gen.lock @@ -1,19 +1,19 @@ lockVersion: 2.0.0 id: c48cf606-fb42-4a45-9c23-8f0555307828 management: - docChecksum: c1f2d60eed7113249b0f829e5fc2bd03 + docChecksum: af6a87267db41808daddec25eed1036b docVersion: 1.0.0 speakeasyVersion: 1.787.0 generationVersion: 2.914.0 - releaseVersion: 1.1.107 - configChecksum: 232fc9d45fd4cf93406edb1411316d1e + releaseVersion: 1.1.108 + configChecksum: c5291e06c972b15bcc783797e47ed3a7 repoURL: https://github.com/OpenRouterTeam/python-sdk.git installationURL: https://github.com/OpenRouterTeam/python-sdk.git published: true persistentEdits: - generation_id: 1ae76d6d-7232-43d8-94c4-17052a779dda - pristine_commit_hash: a707e3bc34d7fc33344ed915721b63a49e5d28aa - pristine_tree_hash: 5226c982e2bacb889a0e9ddfa8aabca4d2f1776a + generation_id: 31fb5f49-d381-49b5-8704-0a6840f1157c + pristine_commit_hash: 2b2af55dbc4e0f4d4f1f32b0945ad20ce32065da + pristine_tree_hash: 3650cac6dece255627852dfda822d5c4629cde3c features: python: acceptHeaders: 3.0.0 @@ -1848,6 +1848,10 @@ trackedFiles: id: eb40c458515d last_write_checksum: sha1:b85ea056036bbdbbcd4e0adea9dcd9a7dc695c51 pristine_git_object: b876816efb61e12af7f25bf72f0447adbfbd810c + docs/components/embeddings.mdx: + id: 71ed622a2f4d + last_write_checksum: sha1:df72f6151943cfe8e1299569ab30fbbd3067b468 + pristine_git_object: dafa20864dfb01d8b9e3db798aa1dab473331c34 docs/components/endpointinfo.mdx: id: 6dcdfc7379a9 last_write_checksum: sha1:2c5fc2144a44e919bda2929fd9dbb10f64641ced @@ -2336,6 +2340,10 @@ trackedFiles: id: dd9998bc9093 last_write_checksum: sha1:a7d7c52d0a7c5128a40fd9f1524e8b6f372a9061 pristine_git_object: df5a9a53d2ea20bb548913c902b0bffdf2dc5d15 + docs/components/imagegeneration.mdx: + id: fd23e84c1023 + last_write_checksum: sha1:bb149a39160f98e6679617250ec3abbcd1853a6b + pristine_git_object: 55213d2296a67affb0bb82d4686082e559444520 docs/components/imagegenerationproviderpreferences.mdx: id: 96067cea4257 last_write_checksum: sha1:63f1caa2d78a1e1d32c1222e8275155417e7e87f @@ -4260,6 +4268,10 @@ trackedFiles: id: 5ed0945b3483 last_write_checksum: sha1:75469331593bced83bfe98408caa584b939c95bc pristine_git_object: 86ea312944a50501fd307f41c471b2462adeeb5e + docs/components/perflast30mbyworkload.mdx: + id: bdcf1999b086 + last_write_checksum: sha1:1390768d24471859dc830f0dbd892085aceab9c5 + pristine_git_object: a75ac7be6f20c9b564a2fc7bc23d8efc4057bac1 docs/components/perrequestlimits.mdx: id: a503880c4645 last_write_checksum: sha1:a971ac638f0c8885577b1165909c8d5ef6b77e1f @@ -4430,8 +4442,8 @@ trackedFiles: pristine_git_object: 5aa8e90a6761d61aa63204bc3aa73f4481449d3e docs/components/publicendpoint.mdx: id: ec4843ef5d86 - last_write_checksum: sha1:cc81fa2e66c0f573baccacd6459fa6291cab6e06 - pristine_git_object: 9b1b04fa64839f8a7ef14593d9dd4080003257f1 + last_write_checksum: sha1:23aab4f17e54311ce27a585548fef94c997def7c + pristine_git_object: 78fc661608b214ddacd52d640660ea7e27cf3c73 docs/components/publicpricing.mdx: id: 61fb18b7ff94 last_write_checksum: sha1:e241420249fcbd1bf515b0f5e84934de0b207948 @@ -4668,6 +4680,10 @@ trackedFiles: id: fd80c94f61ca last_write_checksum: sha1:9df63b87a79a3ccb889870640b0d658ea57436fe pristine_git_object: 01acc3f8d0a451cf6ca5d1c0da90995c79f8f727 + docs/components/rerank.mdx: + id: b011fb4a30ac + last_write_checksum: sha1:95f802db6ffc672d1e266f2e5ce17a5f183ae13d + pristine_git_object: 171e9c32dd65407644e98cc8ce421b9c69376f29 docs/components/resetinterval.mdx: id: ef4f6969df8e last_write_checksum: sha1:e47284974689070a1dd9c82195e4588fe9ffe17c @@ -5044,6 +5060,10 @@ trackedFiles: id: 512e44d2c601 last_write_checksum: sha1:fa643d79e608574ac0210dbdebb293cae1d0af33 pristine_git_object: 972fa7f69f80fae904b94b8a10bbed2fa7bbf1c1 + docs/components/stt.mdx: + id: 2f508094a4f9 + last_write_checksum: sha1:9fc321c80bf682f4eae2def394b0cf01cfae2feb + pristine_git_object: 5ebe2d199734dd2729a8247420ae7e106a90e65a docs/components/sttinputaudio.mdx: id: f4fc56cf1641 last_write_checksum: sha1:816a73a74abbff305ca8405d04870595d525b7bd @@ -5184,6 +5204,10 @@ trackedFiles: id: 061069626d8c last_write_checksum: sha1:c2fc0a4e7fe5c8817e26afe1a646e3e9593d3a86 pristine_git_object: 04c08c1d74e51e7902360962f7bda67b70746feb + docs/components/textgeneration.mdx: + id: 216f7cbe5e8d + last_write_checksum: sha1:313027e4eb6dd5141d035e55ee7429983ab8d66a + pristine_git_object: 83129fca047441e3fbcda675010679cd79efdf90 docs/components/thinking.mdx: id: 8d7292f65441 last_write_checksum: sha1:747fc54885c54d140e7c398d5fd5de79606ce248 @@ -5316,6 +5340,10 @@ trackedFiles: id: d338e8ec18bd last_write_checksum: sha1:c8bc50c016758d48320643ce2965fedba9b1e3f1 pristine_git_object: 6b46b4c151e73cf9efeb4bcbfb0343258a0ed9d5 + docs/components/tts.mdx: + id: f0121f2441bc + last_write_checksum: sha1:02a8f53aa86f14b11a7d2f7eccfa516f428b4a5f + pristine_git_object: 3a4751890ad320101c3f9d67501b848c60a9eea0 docs/components/turnrange.mdx: id: d4f863eecbe8 last_write_checksum: sha1:3cc0bd9a3336447a3ba36639b4af2dff530d33ac @@ -5560,6 +5588,10 @@ trackedFiles: id: 0bdb46a190ad last_write_checksum: sha1:4d2b3619dece58c555b2bed3078a2a5fc6036168 pristine_git_object: 8369333a20b7b67cc70ff68129abc4f236c7ce9b + docs/components/unknown.mdx: + id: b79daf1c00cb + last_write_checksum: sha1:d62323da84114a3f07997a4e7fd9075697e8741d + pristine_git_object: 77c502f0a45d2c4c5aec04a8bfb65f22fd5989f2 docs/components/unprocessableentityresponseerrordata.mdx: id: 7171837ab497 last_write_checksum: sha1:fe645e7a874320f871b2ab92dd9cd945b316f688 @@ -5652,6 +5684,10 @@ trackedFiles: id: 8dacec626dda last_write_checksum: sha1:77a815930df06dc0c0534ad059e49368be46d94d pristine_git_object: 7dedd74d072b742dc84e8afafa3e2ae47dbded10 + docs/components/videogeneration.mdx: + id: 37932f97d574 + last_write_checksum: sha1:d75be772b421373f470e9019896355f728ed5e64 + pristine_git_object: 0f7cc13e29aa234d72930b1a95cf56cdaf836d4d docs/components/videogenerationrequest.mdx: id: b92ddbd15ee4 last_write_checksum: sha1:691ba3554e13181bbb7e5e591d2095df3d2c4eb5 @@ -7402,8 +7438,8 @@ trackedFiles: pristine_git_object: 3e38f1a929f7d6b1d6de74604aa87e3d8f010544 pyproject.toml: id: 5d07e7d72637 - last_write_checksum: sha1:5183913ccc38fe502032fb3a8ddbee0b04008766 - pristine_git_object: 347e295a033bfbaa7a5815e631e0aa8c4af56228 + last_write_checksum: sha1:74d35a6ebb778bd5a34d3db9e655220d5f646347 + pristine_git_object: 7295271514483cecc8a3a0374bfa69cc78cd579b scripts/prepare_readme.py: id: e0c5957a6035 last_write_checksum: sha1:77f44b60b98bc126557ec27391f91dfba764bb54 @@ -7430,8 +7466,8 @@ trackedFiles: pristine_git_object: 86713cfea633e09d33b3d4e65281071fe20e6137 src/openrouter/_version.py: id: d8d15ad6c586 - last_write_checksum: sha1:482494cd3cffec58b42093c7ea7474a17d9473af - pristine_git_object: 66f547c50e336bcf7055324ab17ab7ae2946dc27 + last_write_checksum: sha1:3ecf47a55d18d20fe4c9239caef15359433ec433 + pristine_git_object: 15a2a234f44ca90a855307e3aa4dfeeaded83d63 src/openrouter/analytics.py: id: cb406b5aaabb last_write_checksum: sha1:1e0004d8d1d5d797e2b54cd1cabeb7f9489d08e9 @@ -7470,8 +7506,8 @@ trackedFiles: pristine_git_object: ad3d247954547814054c01989a2dff3d12b3e4e1 src/openrouter/components/__init__.py: id: 81754e97b3f4 - last_write_checksum: sha1:182f20946f7207fdb8152dfe53a9428b2c5b09ff - pristine_git_object: b4138417059fa5b2de853e18bd2578f0b59cd6ed + last_write_checksum: sha1:e8a8f19183c0ef7bc204355bbd29e849e8914a9a + pristine_git_object: 8749d732a97344d90c347cd84a77662e76da5785 src/openrouter/components/aabenchmarkentry.py: id: e2e0f0b48c82 last_write_checksum: sha1:fab4d9a24d2cea937bb749d46c5f83941e99d65c @@ -9310,8 +9346,8 @@ trackedFiles: pristine_git_object: 1798ae9dd8c2e4b629bb99cc5c643827895e8231 src/openrouter/components/publicendpoint.py: id: 848aa2ef9129 - last_write_checksum: sha1:6ca6cfb03234dfd54a121e7b2d5a1fc371e09ea7 - pristine_git_object: 182e3034b8dd4fb9308382fc6ed7eafdc3400abb + last_write_checksum: sha1:859e0d6fb0db27400d0ce70949dfb1b08ac47bed + pristine_git_object: 2b6ad0a0dbfebf6be9e7c4eaff03b99da5e2f890 src/openrouter/components/publicpricing.py: id: 96d115d83cc5 last_write_checksum: sha1:fd8c320ac83b282eaf2405045329b55076d4311a @@ -12477,3 +12513,7 @@ examples: "503": application/json: {"error": {"code": 503, "message": "Service temporarily unavailable"}} examplesVersion: 1.0.2 +releaseNotes: | + ## Python SDK Changes: + * `open_router.endpoints.list_zdr_endpoints()`: `response.data[].perf_last_30m_by_workload` **Added** + * `open_router.endpoints.list()`: `response.data.endpoints[].perf_last_30m_by_workload` **Added** diff --git a/.speakeasy/gen.yaml b/.speakeasy/gen.yaml index 434e5513..b306e164 100644 --- a/.speakeasy/gen.yaml +++ b/.speakeasy/gen.yaml @@ -36,7 +36,7 @@ generation: documentation: mintlify preApplyUnionDiscriminators: true python: - version: 1.1.107 + version: 1.1.108 additionalDependencies: dev: {} main: {} diff --git a/.speakeasy/out.openapi.yaml b/.speakeasy/out.openapi.yaml index bbbe8da3..aa93d5c1 100644 --- a/.speakeasy/out.openapi.yaml +++ b/.speakeasy/out.openapi.yaml @@ -21800,6 +21800,19 @@ components: model_id: 'openai/gpt-4' model_name: 'GPT-4' name: 'OpenAI: GPT-4' + perf_last_30m_by_workload: + text_generation: + latency: + p50: 250 + p75: 350 + p90: 480 + p99: 850 + request_count: 1000 + throughput: + p50: 45.2 + p75: 38.5 + p90: 28.3 + p99: 15.1 pricing: completion: '0.00006' image: '0' @@ -21850,6 +21863,171 @@ components: type: 'string' name: type: 'string' + perf_last_30m_by_workload: + additionalProperties: false + description: 'Endpoint performance over the last 30 minutes, keyed by the kind of request served (e.g. `text_generation`, `image_generation`). Additive to the legacy singular latency and throughput fields; image and video generation report end-to-end latency. Only visible when authenticated with an API key or cookie.' + properties: + embeddings: + properties: + latency: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.' + request_count: + description: 'Total requests admitted for this workload in the window.' + type: + - 'integer' + - 'null' + throughput: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.' + required: + - 'latency' + - 'throughput' + - 'request_count' + type: 'object' + image_generation: + properties: + latency: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.' + request_count: + description: 'Total requests admitted for this workload in the window.' + type: + - 'integer' + - 'null' + throughput: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.' + required: + - 'latency' + - 'throughput' + - 'request_count' + type: 'object' + rerank: + properties: + latency: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.' + request_count: + description: 'Total requests admitted for this workload in the window.' + type: + - 'integer' + - 'null' + throughput: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.' + required: + - 'latency' + - 'throughput' + - 'request_count' + type: 'object' + stt: + properties: + latency: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.' + request_count: + description: 'Total requests admitted for this workload in the window.' + type: + - 'integer' + - 'null' + throughput: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.' + required: + - 'latency' + - 'throughput' + - 'request_count' + type: 'object' + text_generation: + properties: + latency: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.' + request_count: + description: 'Total requests admitted for this workload in the window.' + type: + - 'integer' + - 'null' + throughput: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.' + required: + - 'latency' + - 'throughput' + - 'request_count' + type: 'object' + tts: + properties: + latency: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.' + request_count: + description: 'Total requests admitted for this workload in the window.' + type: + - 'integer' + - 'null' + throughput: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.' + required: + - 'latency' + - 'throughput' + - 'request_count' + type: 'object' + unknown: + properties: + latency: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.' + request_count: + description: 'Total requests admitted for this workload in the window.' + type: + - 'integer' + - 'null' + throughput: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.' + required: + - 'latency' + - 'throughput' + - 'request_count' + type: 'object' + video_generation: + properties: + latency: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Latency percentiles in milliseconds for this workload. Image and video generation report full end-to-end generation time, because their raw latency only measures request acknowledgement; every other workload reports time to first token or result.' + request_count: + description: 'Total requests admitted for this workload in the window.' + type: + - 'integer' + - 'null' + throughput: + allOf: + - $ref: '#/components/schemas/PercentileStats' + - description: 'Throughput percentiles in tokens per second. Only meaningful for text generation; null for workloads without token throughput.' + required: + - 'latency' + - 'throughput' + - 'request_count' + type: 'object' + type: 'object' pricing: properties: audio: diff --git a/.speakeasy/workflow.lock b/.speakeasy/workflow.lock index 2bfd5eef..149ec2fb 100644 --- a/.speakeasy/workflow.lock +++ b/.speakeasy/workflow.lock @@ -2,8 +2,8 @@ speakeasyVersion: 1.787.0 sources: OpenRouter API: sourceNamespace: open-router-chat-completions-api - sourceRevisionDigest: sha256:80157beda13bce6b09e5697bc96e1fe917fcf45e6bd81e72cab6af1cdd4d8174 - sourceBlobDigest: sha256:d8fac9b853e10a8c2e3a863459d66f628e1587ba75b00c2157be666f6a9d45fc + sourceRevisionDigest: sha256:34e5e838344be18bcb3c3144f1e2f79a701e94aaf267b817a7744f586d9de7e6 + sourceBlobDigest: sha256:3d9db05740a823d2891feb40f8872542afdd376d49b77912884f535d96cf993b tags: - latest - 1.0.0 @@ -11,10 +11,10 @@ targets: open-router: source: OpenRouter API sourceNamespace: open-router-chat-completions-api - sourceRevisionDigest: sha256:80157beda13bce6b09e5697bc96e1fe917fcf45e6bd81e72cab6af1cdd4d8174 - sourceBlobDigest: sha256:d8fac9b853e10a8c2e3a863459d66f628e1587ba75b00c2157be666f6a9d45fc + sourceRevisionDigest: sha256:34e5e838344be18bcb3c3144f1e2f79a701e94aaf267b817a7744f586d9de7e6 + sourceBlobDigest: sha256:3d9db05740a823d2891feb40f8872542afdd376d49b77912884f535d96cf993b codeSamplesNamespace: open-router-python-code-samples - codeSamplesRevisionDigest: sha256:f23abdb69df57476ebd61364aff9943bd5af74ea192f36ac04785a374aab7d81 + codeSamplesRevisionDigest: sha256:5a10ed36d297d4d33a9f3baf40bb17ac2104fd2429615ccd7dab19f997b5ca10 workflow: workflowVersion: 1.0.0 speakeasyVersion: 1.787.0 diff --git a/RELEASES.md b/RELEASES.md index 9795f24a..a19f712d 100644 --- a/RELEASES.md +++ b/RELEASES.md @@ -1859,4 +1859,14 @@ Based on: ### Generated - [python v1.1.107] . ### Releases -- [PyPI v1.1.107] https://pypi.org/project/openrouter/1.1.107 - . \ No newline at end of file +- [PyPI v1.1.107] https://pypi.org/project/openrouter/1.1.107 - . + +## 2026-09-01 01:41:43 +### Changes +Based on: +- OpenAPI Doc +- Speakeasy CLI 1.787.0 (2.914.0) https://github.com/speakeasy-api/speakeasy +### Generated +- [python v1.1.108] . +### Releases +- [PyPI v1.1.108] https://pypi.org/project/openrouter/1.1.108 - . \ No newline at end of file diff --git a/docs/components/embeddings.mdx b/docs/components/embeddings.mdx new file mode 100644 index 00000000..dafa2086 --- /dev/null +++ b/docs/components/embeddings.mdx @@ -0,0 +1,11 @@ +--- +title: "Embeddings" +--- + +## Fields + +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | +| `latency` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | +| `request_count` | *Nullable[int]* | :heavy_check_mark: | Total requests admitted for this workload in the window. | | +| `throughput` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | \ No newline at end of file diff --git a/docs/components/imagegeneration.mdx b/docs/components/imagegeneration.mdx new file mode 100644 index 00000000..55213d22 --- /dev/null +++ b/docs/components/imagegeneration.mdx @@ -0,0 +1,11 @@ +--- +title: "ImageGeneration" +--- + +## Fields + +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | +| `latency` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | +| `request_count` | *Nullable[int]* | :heavy_check_mark: | Total requests admitted for this workload in the window. | | +| `throughput` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | \ No newline at end of file diff --git a/docs/components/perflast30mbyworkload.mdx b/docs/components/perflast30mbyworkload.mdx new file mode 100644 index 00000000..a75ac7be --- /dev/null +++ b/docs/components/perflast30mbyworkload.mdx @@ -0,0 +1,19 @@ +--- +title: "PerfLast30mByWorkload" +--- + +Endpoint performance over the last 30 minutes, keyed by the kind of request served (e.g. `text_generation`, `image_generation`). Additive to the legacy singular latency and throughput fields; image and video generation report end-to-end latency. Only visible when authenticated with an API key or cookie. + + +## Fields + +| Field | Type | Required | Description | +| ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | +| `embeddings` | [Optional[components.Embeddings]](../components/embeddings.mdx) | :heavy_minus_sign: | N/A | +| `image_generation` | [Optional[components.ImageGeneration]](../components/imagegeneration.mdx) | :heavy_minus_sign: | N/A | +| `rerank` | [Optional[components.Rerank]](../components/rerank.mdx) | :heavy_minus_sign: | N/A | +| `stt` | [Optional[components.STT]](../components/stt.mdx) | :heavy_minus_sign: | N/A | +| `text_generation` | [Optional[components.TextGeneration]](../components/textgeneration.mdx) | :heavy_minus_sign: | N/A | +| `tts` | [Optional[components.TTS]](../components/tts.mdx) | :heavy_minus_sign: | N/A | +| `unknown` | [Optional[components.Unknown]](../components/unknown.mdx) | :heavy_minus_sign: | N/A | +| `video_generation` | [Optional[components.VideoGeneration]](../components/videogeneration.mdx) | :heavy_minus_sign: | N/A | \ No newline at end of file diff --git a/docs/components/publicendpoint.mdx b/docs/components/publicendpoint.mdx index 9b1b04fa..78fc6616 100644 --- a/docs/components/publicendpoint.mdx +++ b/docs/components/publicendpoint.mdx @@ -7,25 +7,26 @@ Information about a specific model endpoint ## Fields -| Field | Type | Required | Description | Example | -| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| `context_length` | *int* | :heavy_check_mark: | N/A | | -| `latency_last_30m` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | Latency percentiles in milliseconds over the last 30 minutes. Latency measures time to first token. Only visible when authenticated with an API key or cookie; returns null for unauthenticated requests. | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | -| `max_completion_tokens` | *Nullable[int]* | :heavy_check_mark: | Maximum completion tokens for this endpoint. Input and output tokens share the context window, so the effective maximum output for a request is further limited by the context remaining after input tokens. | | -| `max_prompt_tokens` | *Nullable[int]* | :heavy_check_mark: | N/A | | -| `model_id` | *str* | :heavy_check_mark: | The unique identifier for the model (permaslug) | openai/gpt-4 | -| `model_name` | *str* | :heavy_check_mark: | N/A | | -| `name` | *str* | :heavy_check_mark: | N/A | | -| `pricing` | [components.Pricing](../components/pricing.mdx) | :heavy_check_mark: | N/A | | -| `provider_name` | [components.ProviderName](../components/providername.mdx) | :heavy_check_mark: | N/A | OpenAI | -| `quantization` | [Nullable[components.Quantization]](../components/quantization.mdx) | :heavy_check_mark: | N/A | fp16 | -| `status` | [Optional[components.EndpointStatus]](../components/endpointstatus.mdx) | :heavy_minus_sign: | N/A | 0 | -| `supported_parameters` | List[[components.Parameter](../components/parameter.mdx)] | :heavy_check_mark: | N/A | | -| `supports_implicit_caching` | *bool* | :heavy_check_mark: | N/A | | -| `supports_tool_choice` | [components.ToolChoiceSupport](../components/toolchoicesupport.mdx) | :heavy_check_mark: | Per-variant `tool_choice` support. `tool_choice` in `supported_parameters` only says the parameter is accepted; these flags say which of its values passed testing. | \{
"auto": true,
"function": true,
"none": true,
"required": true
} | -| `supports_voice_cloning` | *Optional[bool]* | :heavy_minus_sign: | Whether this TTS endpoint accepts inline reference audio (`input_references`) for stateless voice cloning. Requests carrying reference audio are only routed to endpoints where this is true. | | -| `tag` | *str* | :heavy_check_mark: | N/A | | -| `throughput_last_30m` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | -| `uptime_last_1d` | *Nullable[float]* | :heavy_check_mark: | Uptime percentage over the last 1 day, calculated as successful requests / (successful + error requests) * 100. Rate-limited requests are excluded. Returns null if insufficient data. | | -| `uptime_last_30m` | *Nullable[float]* | :heavy_check_mark: | N/A | | -| `uptime_last_5m` | *Nullable[float]* | :heavy_check_mark: | Uptime percentage over the last 5 minutes, calculated as successful requests / (successful + error requests) * 100. Rate-limited requests are excluded. Returns null if insufficient data. | | \ No newline at end of file +| Field | Type | Required | Description | Example | +| ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `context_length` | *int* | :heavy_check_mark: | N/A | | +| `latency_last_30m` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | Latency percentiles in milliseconds over the last 30 minutes. Latency measures time to first token. Only visible when authenticated with an API key or cookie; returns null for unauthenticated requests. | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | +| `max_completion_tokens` | *Nullable[int]* | :heavy_check_mark: | Maximum completion tokens for this endpoint. Input and output tokens share the context window, so the effective maximum output for a request is further limited by the context remaining after input tokens. | | +| `max_prompt_tokens` | *Nullable[int]* | :heavy_check_mark: | N/A | | +| `model_id` | *str* | :heavy_check_mark: | The unique identifier for the model (permaslug) | openai/gpt-4 | +| `model_name` | *str* | :heavy_check_mark: | N/A | | +| `name` | *str* | :heavy_check_mark: | N/A | | +| `perf_last_30m_by_workload` | [Optional[components.PerfLast30mByWorkload]](../components/perflast30mbyworkload.mdx) | :heavy_minus_sign: | Endpoint performance over the last 30 minutes, keyed by the kind of request served (e.g. `text_generation`, `image_generation`). Additive to the legacy singular latency and throughput fields; image and video generation report end-to-end latency. Only visible when authenticated with an API key or cookie. | | +| `pricing` | [components.Pricing](../components/pricing.mdx) | :heavy_check_mark: | N/A | | +| `provider_name` | [components.ProviderName](../components/providername.mdx) | :heavy_check_mark: | N/A | OpenAI | +| `quantization` | [Nullable[components.Quantization]](../components/quantization.mdx) | :heavy_check_mark: | N/A | fp16 | +| `status` | [Optional[components.EndpointStatus]](../components/endpointstatus.mdx) | :heavy_minus_sign: | N/A | 0 | +| `supported_parameters` | List[[components.Parameter](../components/parameter.mdx)] | :heavy_check_mark: | N/A | | +| `supports_implicit_caching` | *bool* | :heavy_check_mark: | N/A | | +| `supports_tool_choice` | [components.ToolChoiceSupport](../components/toolchoicesupport.mdx) | :heavy_check_mark: | Per-variant `tool_choice` support. `tool_choice` in `supported_parameters` only says the parameter is accepted; these flags say which of its values passed testing. | \{
"auto": true,
"function": true,
"none": true,
"required": true
} | +| `supports_voice_cloning` | *Optional[bool]* | :heavy_minus_sign: | Whether this TTS endpoint accepts inline reference audio (`input_references`) for stateless voice cloning. Requests carrying reference audio are only routed to endpoints where this is true. | | +| `tag` | *str* | :heavy_check_mark: | N/A | | +| `throughput_last_30m` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | +| `uptime_last_1d` | *Nullable[float]* | :heavy_check_mark: | Uptime percentage over the last 1 day, calculated as successful requests / (successful + error requests) * 100. Rate-limited requests are excluded. Returns null if insufficient data. | | +| `uptime_last_30m` | *Nullable[float]* | :heavy_check_mark: | N/A | | +| `uptime_last_5m` | *Nullable[float]* | :heavy_check_mark: | Uptime percentage over the last 5 minutes, calculated as successful requests / (successful + error requests) * 100. Rate-limited requests are excluded. Returns null if insufficient data. | | \ No newline at end of file diff --git a/docs/components/rerank.mdx b/docs/components/rerank.mdx new file mode 100644 index 00000000..171e9c32 --- /dev/null +++ b/docs/components/rerank.mdx @@ -0,0 +1,11 @@ +--- +title: "Rerank" +--- + +## Fields + +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | +| `latency` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | +| `request_count` | *Nullable[int]* | :heavy_check_mark: | Total requests admitted for this workload in the window. | | +| `throughput` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | \ No newline at end of file diff --git a/docs/components/stt.mdx b/docs/components/stt.mdx new file mode 100644 index 00000000..5ebe2d19 --- /dev/null +++ b/docs/components/stt.mdx @@ -0,0 +1,11 @@ +--- +title: "STT" +--- + +## Fields + +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | +| `latency` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | +| `request_count` | *Nullable[int]* | :heavy_check_mark: | Total requests admitted for this workload in the window. | | +| `throughput` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | \ No newline at end of file diff --git a/docs/components/textgeneration.mdx b/docs/components/textgeneration.mdx new file mode 100644 index 00000000..83129fca --- /dev/null +++ b/docs/components/textgeneration.mdx @@ -0,0 +1,11 @@ +--- +title: "TextGeneration" +--- + +## Fields + +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | +| `latency` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | +| `request_count` | *Nullable[int]* | :heavy_check_mark: | Total requests admitted for this workload in the window. | | +| `throughput` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | \ No newline at end of file diff --git a/docs/components/tts.mdx b/docs/components/tts.mdx new file mode 100644 index 00000000..3a475189 --- /dev/null +++ b/docs/components/tts.mdx @@ -0,0 +1,11 @@ +--- +title: "TTS" +--- + +## Fields + +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | +| `latency` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | +| `request_count` | *Nullable[int]* | :heavy_check_mark: | Total requests admitted for this workload in the window. | | +| `throughput` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | \ No newline at end of file diff --git a/docs/components/unknown.mdx b/docs/components/unknown.mdx new file mode 100644 index 00000000..77c502f0 --- /dev/null +++ b/docs/components/unknown.mdx @@ -0,0 +1,11 @@ +--- +title: "Unknown" +--- + +## Fields + +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | +| `latency` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | +| `request_count` | *Nullable[int]* | :heavy_check_mark: | Total requests admitted for this workload in the window. | | +| `throughput` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | \ No newline at end of file diff --git a/docs/components/videogeneration.mdx b/docs/components/videogeneration.mdx new file mode 100644 index 00000000..0f7cc13e --- /dev/null +++ b/docs/components/videogeneration.mdx @@ -0,0 +1,11 @@ +--- +title: "VideoGeneration" +--- + +## Fields + +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | ------------------------------------------------------------------------ | +| `latency` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | +| `request_count` | *Nullable[int]* | :heavy_check_mark: | Total requests admitted for this workload in the window. | | +| `throughput` | [Nullable[components.PercentileStats]](../components/percentilestats.mdx) | :heavy_check_mark: | N/A | \{
"p50": 25.5,
"p75": 35.2,
"p90": 48.7,
"p99": 85.3
} | \ No newline at end of file diff --git a/pyproject.toml b/pyproject.toml index 347e295a..72952715 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "openrouter" -version = "1.1.107" +version = "1.1.108" description = "Official Python Client SDK for OpenRouter." authors = [{ name = "OpenRouter" },] readme = "README-PYPI.md" diff --git a/src/openrouter/_version.py b/src/openrouter/_version.py index 66f547c5..15a2a234 100644 --- a/src/openrouter/_version.py +++ b/src/openrouter/_version.py @@ -3,10 +3,10 @@ import importlib.metadata __title__: str = "openrouter" -__version__: str = "1.1.107" +__version__: str = "1.1.108" __openapi_doc_version__: str = "1.0.0" __gen_version__: str = "2.914.0" -__user_agent__: str = "speakeasy-sdk/python 1.1.107 2.914.0 1.0.0 openrouter" +__user_agent__: str = "speakeasy-sdk/python 1.1.108 2.914.0 1.0.0 openrouter" try: if __package__ is not None: diff --git a/src/openrouter/components/__init__.py b/src/openrouter/components/__init__.py index b4138417..8749d732 100644 --- a/src/openrouter/components/__init__.py +++ b/src/openrouter/components/__init__.py @@ -2517,10 +2517,28 @@ ProviderSortConfigTypedDict, ) from .publicendpoint import ( + Embeddings, + EmbeddingsTypedDict, + ImageGeneration, + ImageGenerationTypedDict, + PerfLast30mByWorkload, + PerfLast30mByWorkloadTypedDict, Pricing, PricingTypedDict, PublicEndpoint, PublicEndpointTypedDict, + Rerank, + RerankTypedDict, + STT, + STTTypedDict, + TTS, + TTSTypedDict, + TextGeneration, + TextGenerationTypedDict, + Unknown, + UnknownTypedDict, + VideoGeneration, + VideoGenerationTypedDict, ) from .publicpricing import PublicPricing, PublicPricingTypedDict from .quantization import Quantization @@ -3877,6 +3895,8 @@ "EditCompact20260112TypedDict", "EditTypeInputTokens", "EditTypedDict", + "Embeddings", + "EmbeddingsTypedDict", "EndpointInfo", "EndpointInfoTypedDict", "EndpointStatus", @@ -4087,6 +4107,7 @@ "ImageGenTextChunkEventPhase", "ImageGenTextChunkEventType", "ImageGenTextChunkEventTypedDict", + "ImageGeneration", "ImageGenerationProviderPreferences", "ImageGenerationProviderPreferencesIgnore", "ImageGenerationProviderPreferencesIgnoreTypedDict", @@ -4124,6 +4145,7 @@ "ImageGenerationServerToolType", "ImageGenerationServerToolTypedDict", "ImageGenerationStatus", + "ImageGenerationTypedDict", "ImageGenerationUsage", "ImageGenerationUsageCompletionTokensDetails", "ImageGenerationUsageCompletionTokensDetailsTypedDict", @@ -4834,6 +4856,8 @@ "PercentileStatsTypedDict", "PercentileThroughputCutoffs", "PercentileThroughputCutoffsTypedDict", + "PerfLast30mByWorkload", + "PerfLast30mByWorkloadTypedDict", "PipelineStage", "PipelineStageType", "PipelineStageTypedDict", @@ -4988,6 +5012,8 @@ "RequireApprovalTypedDict", "RequireApprovalUnion", "RequireApprovalUnionTypedDict", + "Rerank", + "RerankTypedDict", "ResetInterval", "Response", "ResponseFormat", @@ -5024,6 +5050,7 @@ "RoutingStrategy", "Rule", "RuleTypedDict", + "STT", "STTInputAudio", "STTInputAudioTypedDict", "STTRequest", @@ -5036,6 +5063,7 @@ "STTSegment", "STTSegmentTypedDict", "STTTimestampGranularity", + "STTTypedDict", "STTUsage", "STTUsageTypedDict", "STTWord", @@ -5179,6 +5207,8 @@ "Syntax", "System", "SystemTypedDict", + "TTS", + "TTSTypedDict", "TaskBudget", "TaskBudgetTypedDict", "TaskClassificationItem", @@ -5199,6 +5229,8 @@ "TextDoneEventTypedDict", "TextExtendedConfig", "TextExtendedConfigTypedDict", + "TextGeneration", + "TextGenerationTypedDict", "Thinking", "ThinkingAdaptive", "ThinkingAdaptiveTypedDict", @@ -5334,6 +5366,7 @@ "UniqueInsight", "UniqueInsightTypedDict", "Unit", + "Unknown", "UnknownAction", "UnknownApplyPatchCallOperation", "UnknownBaseInputsContent1", @@ -5361,6 +5394,7 @@ "UnknownOutputMessageItemContent", "UnknownReasoningDetailUnion", "UnknownStreamEvents", + "UnknownTypedDict", "UnprocessableEntityResponseErrorData", "UnprocessableEntityResponseErrorDataTypedDict", "UpdateBYOKKeyRequest", @@ -5400,6 +5434,7 @@ "Variables", "VariablesTypedDict", "Verbosity", + "VideoGeneration", "VideoGenerationRequest", "VideoGenerationRequestAspectRatio", "VideoGenerationRequestOptions", @@ -5411,6 +5446,7 @@ "VideoGenerationResponse", "VideoGenerationResponseStatus", "VideoGenerationResponseTypedDict", + "VideoGenerationTypedDict", "VideoGenerationUsage", "VideoGenerationUsageTypedDict", "VideoModel", @@ -7338,10 +7374,28 @@ "Partition": ".providersortconfig", "ProviderSortConfig": ".providersortconfig", "ProviderSortConfigTypedDict": ".providersortconfig", + "Embeddings": ".publicendpoint", + "EmbeddingsTypedDict": ".publicendpoint", + "ImageGeneration": ".publicendpoint", + "ImageGenerationTypedDict": ".publicendpoint", + "PerfLast30mByWorkload": ".publicendpoint", + "PerfLast30mByWorkloadTypedDict": ".publicendpoint", "Pricing": ".publicendpoint", "PricingTypedDict": ".publicendpoint", "PublicEndpoint": ".publicendpoint", "PublicEndpointTypedDict": ".publicendpoint", + "Rerank": ".publicendpoint", + "RerankTypedDict": ".publicendpoint", + "STT": ".publicendpoint", + "STTTypedDict": ".publicendpoint", + "TTS": ".publicendpoint", + "TTSTypedDict": ".publicendpoint", + "TextGeneration": ".publicendpoint", + "TextGenerationTypedDict": ".publicendpoint", + "Unknown": ".publicendpoint", + "UnknownTypedDict": ".publicendpoint", + "VideoGeneration": ".publicendpoint", + "VideoGenerationTypedDict": ".publicendpoint", "PublicPricing": ".publicpricing", "PublicPricingTypedDict": ".publicpricing", "Quantization": ".quantization", diff --git a/src/openrouter/components/publicendpoint.py b/src/openrouter/components/publicendpoint.py index 182e3034..2b6ad0a0 100644 --- a/src/openrouter/components/publicendpoint.py +++ b/src/openrouter/components/publicendpoint.py @@ -14,6 +14,306 @@ from typing_extensions import NotRequired, TypedDict +class EmbeddingsTypedDict(TypedDict): + latency: Nullable[PercentileStatsTypedDict] + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + throughput: Nullable[PercentileStatsTypedDict] + + +class Embeddings(BaseModel): + latency: Nullable[PercentileStats] + + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + + throughput: Nullable[PercentileStats] + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + m[k] = val + + return m + + +class ImageGenerationTypedDict(TypedDict): + latency: Nullable[PercentileStatsTypedDict] + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + throughput: Nullable[PercentileStatsTypedDict] + + +class ImageGeneration(BaseModel): + latency: Nullable[PercentileStats] + + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + + throughput: Nullable[PercentileStats] + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + m[k] = val + + return m + + +class RerankTypedDict(TypedDict): + latency: Nullable[PercentileStatsTypedDict] + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + throughput: Nullable[PercentileStatsTypedDict] + + +class Rerank(BaseModel): + latency: Nullable[PercentileStats] + + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + + throughput: Nullable[PercentileStats] + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + m[k] = val + + return m + + +class STTTypedDict(TypedDict): + latency: Nullable[PercentileStatsTypedDict] + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + throughput: Nullable[PercentileStatsTypedDict] + + +class STT(BaseModel): + latency: Nullable[PercentileStats] + + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + + throughput: Nullable[PercentileStats] + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + m[k] = val + + return m + + +class TextGenerationTypedDict(TypedDict): + latency: Nullable[PercentileStatsTypedDict] + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + throughput: Nullable[PercentileStatsTypedDict] + + +class TextGeneration(BaseModel): + latency: Nullable[PercentileStats] + + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + + throughput: Nullable[PercentileStats] + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + m[k] = val + + return m + + +class TTSTypedDict(TypedDict): + latency: Nullable[PercentileStatsTypedDict] + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + throughput: Nullable[PercentileStatsTypedDict] + + +class TTS(BaseModel): + latency: Nullable[PercentileStats] + + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + + throughput: Nullable[PercentileStats] + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + m[k] = val + + return m + + +class UnknownTypedDict(TypedDict): + latency: Nullable[PercentileStatsTypedDict] + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + throughput: Nullable[PercentileStatsTypedDict] + + +class Unknown(BaseModel): + latency: Nullable[PercentileStats] + + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + + throughput: Nullable[PercentileStats] + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + m[k] = val + + return m + + +class VideoGenerationTypedDict(TypedDict): + latency: Nullable[PercentileStatsTypedDict] + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + throughput: Nullable[PercentileStatsTypedDict] + + +class VideoGeneration(BaseModel): + latency: Nullable[PercentileStats] + + request_count: Nullable[int] + r"""Total requests admitted for this workload in the window.""" + + throughput: Nullable[PercentileStats] + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + m[k] = val + + return m + + +class PerfLast30mByWorkloadTypedDict(TypedDict): + r"""Endpoint performance over the last 30 minutes, keyed by the kind of request served (e.g. `text_generation`, `image_generation`). Additive to the legacy singular latency and throughput fields; image and video generation report end-to-end latency. Only visible when authenticated with an API key or cookie.""" + + embeddings: NotRequired[EmbeddingsTypedDict] + image_generation: NotRequired[ImageGenerationTypedDict] + rerank: NotRequired[RerankTypedDict] + stt: NotRequired[STTTypedDict] + text_generation: NotRequired[TextGenerationTypedDict] + tts: NotRequired[TTSTypedDict] + unknown: NotRequired[UnknownTypedDict] + video_generation: NotRequired[VideoGenerationTypedDict] + + +class PerfLast30mByWorkload(BaseModel): + r"""Endpoint performance over the last 30 minutes, keyed by the kind of request served (e.g. `text_generation`, `image_generation`). Additive to the legacy singular latency and throughput fields; image and video generation report end-to-end latency. Only visible when authenticated with an API key or cookie.""" + + embeddings: Optional[Embeddings] = None + + image_generation: Optional[ImageGeneration] = None + + rerank: Optional[Rerank] = None + + stt: Optional[STT] = None + + text_generation: Optional[TextGeneration] = None + + tts: Optional[TTS] = None + + unknown: Optional[Unknown] = None + + video_generation: Optional[VideoGeneration] = None + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + optional_fields = set( + [ + "embeddings", + "image_generation", + "rerank", + "stt", + "text_generation", + "tts", + "unknown", + "video_generation", + ] + ) + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + if val is not None or k not in optional_fields: + m[k] = val + + return m + + class PricingTypedDict(TypedDict): completion: str r"""Price in USD per token for completion (output) generation""" @@ -159,6 +459,8 @@ class PublicEndpointTypedDict(TypedDict): uptime_last_30m: Nullable[float] uptime_last_5m: Nullable[float] r"""Uptime percentage over the last 5 minutes, calculated as successful requests / (successful + error requests) * 100. Rate-limited requests are excluded. Returns null if insufficient data.""" + perf_last_30m_by_workload: NotRequired[PerfLast30mByWorkloadTypedDict] + r"""Endpoint performance over the last 30 minutes, keyed by the kind of request served (e.g. `text_generation`, `image_generation`). Additive to the legacy singular latency and throughput fields; image and video generation report end-to-end latency. Only visible when authenticated with an API key or cookie.""" status: NotRequired[EndpointStatus] supports_voice_cloning: NotRequired[bool] r"""Whether this TTS endpoint accepts inline reference audio (`input_references`) for stateless voice cloning. Requests carrying reference audio are only routed to endpoints where this is true.""" @@ -209,6 +511,9 @@ class PublicEndpoint(BaseModel): uptime_last_5m: Nullable[float] r"""Uptime percentage over the last 5 minutes, calculated as successful requests / (successful + error requests) * 100. Rate-limited requests are excluded. Returns null if insufficient data.""" + perf_last_30m_by_workload: Optional[PerfLast30mByWorkload] = None + r"""Endpoint performance over the last 30 minutes, keyed by the kind of request served (e.g. `text_generation`, `image_generation`). Additive to the legacy singular latency and throughput fields; image and video generation report end-to-end latency. Only visible when authenticated with an API key or cookie.""" + status: Optional[EndpointStatus] = None supports_voice_cloning: Optional[bool] = False @@ -216,7 +521,9 @@ class PublicEndpoint(BaseModel): @model_serializer(mode="wrap") def serialize_model(self, handler): - optional_fields = set(["status", "supports_voice_cloning"]) + optional_fields = set( + ["perf_last_30m_by_workload", "status", "supports_voice_cloning"] + ) nullable_fields = set( [ "latency_last_30m", diff --git a/uv.lock b/uv.lock index 05e1e5b6..6c5d6dfa 100644 --- a/uv.lock +++ b/uv.lock @@ -213,7 +213,7 @@ wheels = [ [[package]] name = "openrouter" -version = "1.1.107" +version = "1.1.108" source = { editable = "." } dependencies = [ { name = "httpcore" },