From c263d82cf21407955e75315ae1f768f41e4d42b9 Mon Sep 17 00:00:00 2001 From: speakeasybot Date: Fri, 25 Sep 2026 19:43:59 +0000 Subject: [PATCH 1/2] ## Go SDK Changes: * `OpenRouter.Tts.CreateSpeech()`: * `request.Request.InputReferences[]` **Changed** * `OpenRouter.Endpoints.ListZdrEndpoints()`: `response.Data[]` **Changed** * `OpenRouter.Endpoints.List()`: `response.Data.Endpoints[]` **Changed** --- .speakeasy/gen.lock | 92 +++++++++++-------- .speakeasy/gen.yaml | 2 +- .speakeasy/out.openapi.yaml | 63 +++++++++++-- .speakeasy/workflow.lock | 10 +- RELEASES.md | 12 ++- docs/models/components/publicendpoint.mdx | 2 + .../components/speechinputreference.mdx | 10 +- .../components/speechinputreferenceaudio.mdx | 2 +- .../speechinputreferenceaudioinput.mdx | 9 +- .../components/speechinputreferenceimage.mdx | 13 +++ .../speechinputreferenceimageinput.mdx | 12 +++ .../speechinputreferenceimagetype.mdx | 20 ++++ .../components/speechinputreferencetext.mdx | 10 +- docs/models/components/speechrequest.mdx | 24 ++--- models/components/publicendpoint.go | 32 +++++-- models/components/speechinputreference.go | 29 +++++- .../components/speechinputreferenceaudio.go | 2 +- .../speechinputreferenceaudioinput.go | 17 +++- .../components/speechinputreferenceimage.go | 64 +++++++++++++ .../speechinputreferenceimageinput.go | 31 +++++++ models/components/speechinputreferencetext.go | 4 +- models/components/speechrequest.go | 2 +- openrouter.go | 18 +--- 23 files changed, 374 insertions(+), 106 deletions(-) create mode 100644 docs/models/components/speechinputreferenceimage.mdx create mode 100644 docs/models/components/speechinputreferenceimageinput.mdx create mode 100644 docs/models/components/speechinputreferenceimagetype.mdx create mode 100644 models/components/speechinputreferenceimage.go create mode 100644 models/components/speechinputreferenceimageinput.go diff --git a/.speakeasy/gen.lock b/.speakeasy/gen.lock index 3500c43e..5d6ef17a 100644 --- a/.speakeasy/gen.lock +++ b/.speakeasy/gen.lock @@ -1,18 +1,18 @@ lockVersion: 2.0.0 id: 9eea3e30-2722-4060-973b-57b7f1f75dc0 management: - docChecksum: 76847dddaf4195bb5389098ecd64c911 + docChecksum: 7789087cc61c124e2e0ea1977f73f712 docVersion: 1.0.0 speakeasyVersion: 1.787.0 generationVersion: 2.914.0 - releaseVersion: 0.8.27 - configChecksum: ec18e8b7f0be26ec826bc9bdd004f47e + releaseVersion: 0.8.28 + configChecksum: 8bfa4eaa1285bdfc4936a9a33479c5b7 repoURL: https://github.com/OpenRouterTeam/go-sdk.git installationURL: https://github.com/OpenRouterTeam/go-sdk persistentEdits: - generation_id: dde7b26a-2dfe-402f-8710-1f99a3dddadb - pristine_commit_hash: ec7672cfeb8d02cf1ec82e270065df53745d1ac2 - pristine_tree_hash: 8dfa0b1fb374b87a80dca82275353857cd54c843 + generation_id: e32d550b-4e47-49fc-bc84-7dcceaaec832 + pristine_commit_hash: 673cad2273a902a7aad162d0e3b233d5fa005861 + pristine_tree_hash: b25a5fbbcda74d2e2de106d95eb01e711adcb060 features: go: acceptHeaders: 2.81.2 @@ -4964,8 +4964,8 @@ trackedFiles: pristine_git_object: 0b1066907451e465095ec04fcf71cbf4e6019a38 docs/models/components/publicendpoint.mdx: id: 63d6ef18c590 - last_write_checksum: sha1:2b763e2218899451969deb00331d839b8748cc4a - pristine_git_object: 77bda5791f91b261283aa50ad5e54b8fb3e944ac + last_write_checksum: sha1:84aa58570c2fa81e017dff4233e5798b759ba3d7 + pristine_git_object: 70ca038b1dc2d40804401fcb39b90a7ba77f386b docs/models/components/publicpricing.mdx: id: c23229563aba last_write_checksum: sha1:bfc74c17989b9ad338568c72e1b78a86f9a39232 @@ -5452,32 +5452,44 @@ trackedFiles: pristine_git_object: e2787b1e7be4600fe347c8d90077895fb86ff3a5 docs/models/components/speechinputreference.mdx: id: c9a45ff446df - last_write_checksum: sha1:6b318eed9fd5de580dafe916fa7bb5403df8a6d5 - pristine_git_object: 3c918c413a58aebb4ee402e921eb5a79d1945f9f + last_write_checksum: sha1:e9b61b3061d8b5cf043c384a7362af3208558144 + pristine_git_object: fc519a750b5e20416efc881c35f4f0066de7a393 docs/models/components/speechinputreferenceaudio.mdx: id: b4e20c639be3 - last_write_checksum: sha1:b85edf8e0d0ff33129aa8148fa44913a5784433e - pristine_git_object: e52e761ac3fb4585faf11e37b92edd1de121dbab + last_write_checksum: sha1:417e60dd820907f078137c6cdbb6ad882c129ae8 + pristine_git_object: c2bbb82bace8fe0d42337f3cd0d9d87f69e63550 docs/models/components/speechinputreferenceaudioinput.mdx: id: d153391b6207 - last_write_checksum: sha1:3a2d89f72314e97e97051828ba3eef291ca66909 - pristine_git_object: 9896ac9a006671e9a9615e5796d3c04834e742e2 + last_write_checksum: sha1:5dc63d159b6c04b9d90162fb748a026d129ca3b6 + pristine_git_object: 1dcc1498a1e88a52831c8c5351edc372815b56a8 docs/models/components/speechinputreferenceaudiotype.mdx: id: 7b2fdf1f469f last_write_checksum: sha1:813639ff290657da9ec1e6be7f253cb8b4454215 pristine_git_object: f189c672b74e77e0eb2b1362f894d086174023c8 + docs/models/components/speechinputreferenceimage.mdx: + id: 6ede4c48391d + last_write_checksum: sha1:4e65fa80f5912f65448ed5c387fcfb75f970596f + pristine_git_object: a50429cdf1c526a1762c51c8f45e10772282ce39 + docs/models/components/speechinputreferenceimageinput.mdx: + id: 02b13656ab4a + last_write_checksum: sha1:7f9482f1d7d75b98022aa893a0c37cedd15e8b1c + pristine_git_object: 5af0f2d241c7457890550429b2fc84e8e0d218a2 + docs/models/components/speechinputreferenceimagetype.mdx: + id: 4da4ae255d1a + last_write_checksum: sha1:936318622baf7db646dad7197b66fdbd1f998ed6 + pristine_git_object: 4f732841d1c7f38432049401ad0b92467b2b8745 docs/models/components/speechinputreferencetext.mdx: id: df0d7f7523b2 - last_write_checksum: sha1:2ed32228b8d0dc2f7658e2eb6b3fb85ea3cd8ba1 - pristine_git_object: 919b6cb8a7e99d11fd3af9b972ff17e1d907513f + last_write_checksum: sha1:8918ba36c5756e454f2d968daabc389a1c615830 + pristine_git_object: c8ab60319cd88ba6817f44191657e5ce7c3be101 docs/models/components/speechinputreferencetexttype.mdx: id: fc7d503a8e34 last_write_checksum: sha1:4cf4bf648ac7a69a281925ad790940d263747544 pristine_git_object: d11bcca08713060dd9e5b41ad311978fcabddfbf docs/models/components/speechrequest.mdx: id: 572bb1b07290 - last_write_checksum: sha1:56b1479240139cdf24604f38dcf28584a7512c09 - pristine_git_object: ba5782436cfa3e39be40608a549fa5ada0ab2278 + last_write_checksum: sha1:d0111ebee077afcc88e88c106b10d0ffd7efe56e + pristine_git_object: 03bd63abed61b9e03665702d75c3d29b164af47d docs/models/components/speechrequestprovider.mdx: id: 70ddc64fe615 last_write_checksum: sha1:0b345acd1029103dca5e9dbe7e3480cad15266d7 @@ -9876,8 +9888,8 @@ trackedFiles: pristine_git_object: 1ef96ca8132f8e9c7657a7e7064dc44c720a046a models/components/publicendpoint.go: id: "32378212e204" - last_write_checksum: sha1:8ad5f2b7c1f02b6d496a3748dc038bf93d8bb6b6 - pristine_git_object: 4435f33435814c6a37305333fe9c88a7eb4b7e83 + last_write_checksum: sha1:be2c7f4c44ef4222f5f13704ae76d15ce92ab5d5 + pristine_git_object: bb11f4a28dbed30c21a635c434e5c53bbb20fc88 models/components/publicpricing.go: id: 8596f6c62439 last_write_checksum: sha1:27958df794cc170d479c41ce42c79839ce3118b3 @@ -10116,24 +10128,32 @@ trackedFiles: pristine_git_object: 696bbae97193bc399544e814a3e9d99a24522909 models/components/speechinputreference.go: id: 8b017b7a2135 - last_write_checksum: sha1:915119c639d516b210bffff9be313582b13b13a5 - pristine_git_object: 43f57bafa4bc9e7eca380fea8326783b078fd44c + last_write_checksum: sha1:f292ef383c3eb20059c817ca6988deb9da90df19 + pristine_git_object: 0447cba9fffc93491b946e0273787dde49341de7 models/components/speechinputreferenceaudio.go: id: a7dd0b13a05b - last_write_checksum: sha1:df07caf6da980c041c6fb5c18c0370097190a4e1 - pristine_git_object: 643f74859da5072bc20ba2dd77bf20fa989f412a + last_write_checksum: sha1:fa461e523ce63f69872e8b97a43423585f498d1e + pristine_git_object: e83f7b82aa553c74d8ec972bf839502550ef204d models/components/speechinputreferenceaudioinput.go: id: 36ea0aa95c87 - last_write_checksum: sha1:dcba790bbb7b46d468fce335c95d5a6ce9617ccc - pristine_git_object: 738c1a9e9f54465459343b39343ef951026cbcef + last_write_checksum: sha1:83be13c9461e6d5d35b8c0ec3af0c06741cda7ac + pristine_git_object: faf947a2ec79353acfc6c95b3fd0a432f8148453 + models/components/speechinputreferenceimage.go: + id: cf0ef0a9bd93 + last_write_checksum: sha1:c8f623747db0561009bcbe113f8a97066b6669e9 + pristine_git_object: 9d1ef4d0121ea3901d5eb4d58927ceb98c0cb345 + models/components/speechinputreferenceimageinput.go: + id: 8c14ad7e5815 + last_write_checksum: sha1:15e213b06d6384ef1c1811d519a8ff58463cb979 + pristine_git_object: 8209feefc59a2ee47b445a2316af55826781ed08 models/components/speechinputreferencetext.go: id: da29d62e1fb5 - last_write_checksum: sha1:4e95a077b5d17341c96cdbfa9090dc850b92ce5a - pristine_git_object: 0b13751c25b0c24645772f69b4a034eb6738b95c + last_write_checksum: sha1:024e77a5c8e4448457667171b11a166e67899d98 + pristine_git_object: 282892fcabc303c6f383535c88d640ecd1bf35be models/components/speechrequest.go: id: c5f83d3285eb - last_write_checksum: sha1:1ce6448b22fb24b8fff7cb4fa256c97abfe8e8b6 - pristine_git_object: a366d643b798f593d2806cb217ddb6dd96e477a7 + last_write_checksum: sha1:599dbe517a8be002f7ae7a12c911aaffd6ae254b + pristine_git_object: e50e827a82fd4fe04967bf53fbda14d27764b94c models/components/stopservertoolswhencondition.go: id: 6c12d61377dc last_write_checksum: sha1:600037f2c92bb83b31cc6469a5ce9a169ee82b09 @@ -11100,8 +11120,8 @@ trackedFiles: pristine_git_object: 5e3a2cdf8cb4ed4f5378b712293cba9c82074f16 openrouter.go: id: 207ad004b774 - last_write_checksum: sha1:a4fa7f2e5c6a04fcaf2f42a157dfaaff9c733842 - pristine_git_object: b7af4d8bb416299e80569f78a9cdc3f2d51088aa + last_write_checksum: sha1:08dc9b94f57a09be1d68653f30eed473e1ee94a9 + pristine_git_object: 97325b43ed405ff173c188a0a7e8ca284344f2ec optionalnullable/optionalnullable.go: id: ce60f259ead3 last_write_checksum: sha1:d6aff1a420c31e025ea21d46cf056e305ae76fe8 @@ -11395,7 +11415,7 @@ examples: slug: "" responses: "200": - application/json: {"data": {"architecture": {"input_modalities": ["text"], "instruct_type": "chatml", "modality": "text->text", "output_modalities": ["text"], "tokenizer": "GPT"}, "created": 1692901234, "description": "GPT-4 is a large multimodal model that can solve difficult problems with greater accuracy.", "endpoints": [{"context_length": 8192, "latency_last_30m": {"p50": 0.25, "p75": 0.35, "p90": 0.48, "p99": 0.85}, "max_completion_tokens": 4096, "max_prompt_tokens": 8192, "model_id": "openai/gpt-4", "model_name": "GPT-4", "name": "OpenAI: GPT-4", "pricing": {"completion": "0.00006", "prompt": "0.00003"}, "provider_name": "OpenAI", "quantization": "fp16", "supported_parameters": ["temperature", "top_p", "max_tokens"], "supports_implicit_caching": true, "supports_tool_choice": {"auto": true, "function": true, "none": true, "required": true}, "supports_voice_cloning": false, "tag": "openai", "throughput_last_30m": {"p50": 45.2, "p75": 38.5, "p90": 28.3, "p99": 15.1}, "uptime_last_1d": 99.8, "uptime_last_30m": 99.5, "uptime_last_5m": 100}], "id": "openai/gpt-4", "name": "GPT-4"}} + application/json: {"data": {"architecture": {"input_modalities": ["text"], "instruct_type": "chatml", "modality": "text->text", "output_modalities": ["text"], "tokenizer": "GPT"}, "created": 1692901234, "description": "GPT-4 is a large multimodal model that can solve difficult problems with greater accuracy.", "endpoints": [{"context_length": 8192, "latency_last_30m": {"p50": 0.25, "p75": 0.35, "p90": 0.48, "p99": 0.85}, "max_completion_tokens": 4096, "max_prompt_tokens": 8192, "model_id": "openai/gpt-4", "model_name": "GPT-4", "name": "OpenAI: GPT-4", "pricing": {"completion": "0.00006", "prompt": "0.00003"}, "provider_name": "OpenAI", "quantization": "fp16", "supported_parameters": ["temperature", "top_p", "max_tokens"], "supports_image_reference": false, "supports_implicit_caching": true, "supports_multiple_audio_references": false, "supports_tool_choice": {"auto": true, "function": true, "none": true, "required": true}, "supports_voice_cloning": false, "tag": "openai", "throughput_last_30m": {"p50": 45.2, "p75": 38.5, "p90": 28.3, "p99": 15.1}, "uptime_last_1d": 99.8, "uptime_last_30m": 99.5, "uptime_last_5m": 100}], "id": "openai/gpt-4", "name": "GPT-4"}} "404": application/json: {"error": {"code": 404, "message": "Resource not found"}} "500": @@ -11406,7 +11426,7 @@ examples: speakeasy-default-list-endpoints-zdr: responses: "200": - application/json: {"data": [{"context_length": 8192, "latency_last_30m": {"p50": 0.25, "p75": 0.35, "p90": 0.48, "p99": 0.85}, "max_completion_tokens": 4096, "max_prompt_tokens": 8192, "model_id": "openai/gpt-4", "model_name": "GPT-4", "name": "OpenAI: GPT-4", "pricing": {"completion": "0.00006", "prompt": "0.00003"}, "provider_name": "OpenAI", "quantization": "fp16", "supported_parameters": ["temperature", "top_p", "max_tokens"], "supports_implicit_caching": true, "supports_tool_choice": {"auto": true, "function": true, "none": true, "required": true}, "supports_voice_cloning": false, "tag": "openai", "throughput_last_30m": {"p50": 45.2, "p75": 38.5, "p90": 28.3, "p99": 15.1}, "uptime_last_1d": 99.8, "uptime_last_30m": 99.5, "uptime_last_5m": 100}]} + application/json: {"data": [{"context_length": 8192, "latency_last_30m": {"p50": 0.25, "p75": 0.35, "p90": 0.48, "p99": 0.85}, "max_completion_tokens": 4096, "max_prompt_tokens": 8192, "model_id": "openai/gpt-4", "model_name": "GPT-4", "name": "OpenAI: GPT-4", "pricing": {"completion": "0.00006", "prompt": "0.00003"}, "provider_name": "OpenAI", "quantization": "fp16", "supported_parameters": ["temperature", "top_p", "max_tokens"], "supports_image_reference": false, "supports_implicit_caching": true, "supports_multiple_audio_references": false, "supports_tool_choice": {"auto": true, "function": true, "none": true, "required": true}, "supports_voice_cloning": false, "tag": "openai", "throughput_last_30m": {"p50": 45.2, "p75": 38.5, "p90": 28.3, "p99": 15.1}, "uptime_last_1d": 99.8, "uptime_last_30m": 99.5, "uptime_last_5m": 100}]} "500": application/json: {"error": {"code": 500, "message": "Internal Server Error"}} "403": @@ -14429,6 +14449,4 @@ examples: "500": application/json: {"error": {"code": 404, "message": "Intern not found"}} examplesVersion: 1.0.2 -releaseNotes: | - ## Go SDK Changes: - * `OpenRouter.Interns.GetInternDaemonAccess()`: `error.status[429]` **Removed** (Breaking ⚠️) +releaseNotes: "## Go SDK Changes:\n* `OpenRouter.Tts.CreateSpeech()`: \n * `request.Request.InputReferences[]` **Changed**\n* `OpenRouter.Endpoints.ListZdrEndpoints()`: `response.Data[]` **Changed**\n* `OpenRouter.Endpoints.List()`: `response.Data.Endpoints[]` **Changed**\n" diff --git a/.speakeasy/gen.yaml b/.speakeasy/gen.yaml index 0cd91043..ddb9d6a7 100644 --- a/.speakeasy/gen.yaml +++ b/.speakeasy/gen.yaml @@ -36,7 +36,7 @@ generation: documentation: mintlify preApplyUnionDiscriminators: true go: - version: 0.8.27 + version: 0.8.28 additionalDependencies: {} baseErrorName: OpenRouterError clientServerStatusCodesAsErrors: true diff --git a/.speakeasy/out.openapi.yaml b/.speakeasy/out.openapi.yaml index 3a3773c2..35eabb17 100644 --- a/.speakeasy/out.openapi.yaml +++ b/.speakeasy/out.openapi.yaml @@ -23960,7 +23960,9 @@ components: - 'temperature' - 'top_p' - 'max_tokens' + supports_image_reference: false supports_implicit_caching: true + supports_multiple_audio_references: false supports_tool_choice: auto: true function: true @@ -24252,8 +24254,16 @@ components: items: $ref: '#/components/schemas/Parameter' type: 'array' + supports_image_reference: + default: false + description: 'Whether this TTS endpoint accepts an `image_url` reference describing the desired voice. Requests carrying an image reference are only routed to endpoints where this is true.' + type: 'boolean' supports_implicit_caching: type: 'boolean' + supports_multiple_audio_references: + default: false + description: 'Whether this TTS endpoint accepts more than one `input_audio` reference clip per request. Requests carrying several clips are only routed to endpoints where this is true.' + type: 'boolean' supports_tool_choice: $ref: '#/components/schemas/ToolChoiceSupport' supports_voice_cloning: @@ -25918,17 +25928,19 @@ components: - $ref: '#/components/schemas/ContainerAutoEnvironment' - $ref: '#/components/schemas/ContainerReferenceEnvironment' SpeechInputReference: - description: 'Reference content part for stateless voice cloning' + description: 'Reference content part for stateless voice cloning or voice design' discriminator: mapping: + image_url: '#/components/schemas/SpeechInputReferenceImage' input_audio: '#/components/schemas/SpeechInputReferenceAudio' text: '#/components/schemas/SpeechInputReferenceText' propertyName: 'type' oneOf: - $ref: '#/components/schemas/SpeechInputReferenceAudio' - $ref: '#/components/schemas/SpeechInputReferenceText' + - $ref: '#/components/schemas/SpeechInputReferenceImage' SpeechInputReferenceAudio: - description: 'Reference audio input for stateless voice cloning' + description: 'Reference audio input for stateless voice cloning. Up to three parts per request; the Nth audio part is addressable from `input` as `@AudioN` on providers that support multiple references.' example: input_audio: data: 'data:audio/wav;base64,UklGRuQXDABXQVZF...' @@ -25948,7 +25960,7 @@ components: description: 'Reference audio input object' properties: data: - description: 'Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio).' + description: 'Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio). Exactly one of `data` or `url` is required.' example: 'data:audio/wav;base64,UklGRuQXDABXQVZF...' maxLength: 20971520 minLength: 1 @@ -25957,17 +25969,50 @@ components: description: 'Audio format of the reference audio (e.g., wav, mp3). Optional; most providers detect the format from the audio bytes.' example: 'wav' type: 'string' + url: + description: 'Public http(s) URL of the reference audio. OpenRouter downloads it (15 MiB max) and forwards the bytes, never the URL. Exactly one of `data` or `url` is required.' + example: 'https://example.com/reference.wav' + format: 'uri' + maxLength: 2048 + type: 'string' + type: 'object' + SpeechInputReferenceImage: + description: 'Reference image describing the desired voice. Cannot be combined with `input_audio` parts. Only routed to endpoints that support image references.' + example: + image_url: + url: 'data:image/png;base64,iVBORw0KGgo...' + type: 'image_url' + properties: + image_url: + $ref: '#/components/schemas/SpeechInputReferenceImageInput' + type: + enum: + - 'image_url' + type: 'string' required: - - 'data' + - 'type' + - 'image_url' + type: 'object' + SpeechInputReferenceImageInput: + description: 'Reference image input object' + properties: + url: + description: 'JPEG, PNG, or WebP reference image as a base64 data URI or a public http(s) URL. Remote images are downloaded (15 MiB max) and forwarded as bytes, never as the URL.' + example: 'data:image/png;base64,iVBORw0KGgo...' + maxLength: 20971520 + minLength: 1 + type: 'string' + required: + - 'url' type: 'object' SpeechInputReferenceText: - description: 'Transcript of the accompanying reference audio' + description: 'Transcript of an `input_audio` part' example: text: 'I used to rule the world.' type: 'text' properties: text: - description: 'Transcript of the accompanying reference audio.' + description: 'Transcript of an `input_audio` part. With a single clip it may appear before or after the clip; with multiple clips it must immediately follow the clip it transcribes.' example: 'I used to rule the world.' maxLength: 10000 type: 'string' @@ -25993,7 +26038,7 @@ components: example: 'Hello world' type: 'string' input_references: - description: 'Reference content for stateless voice cloning: one `input_audio` part carrying the voice sample, optionally accompanied by one `text` part with its transcript. Only routed to endpoints that support voice cloning.' + description: 'Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference.' example: - input_audio: data: 'data:audio/wav;base64,UklGRuQXDABXQVZF...' @@ -34107,7 +34152,9 @@ paths: - 'temperature' - 'top_p' - 'max_tokens' + supports_image_reference: false supports_implicit_caching: true + supports_multiple_audio_references: false supports_voice_cloning: false tag: 'openai' throughput_last_30m: @@ -34144,7 +34191,9 @@ paths: - 'temperature' - 'top_p' - 'max_tokens' + supports_image_reference: false supports_implicit_caching: true + supports_multiple_audio_references: false supports_voice_cloning: false tag: 'openai' throughput_last_30m: diff --git a/.speakeasy/workflow.lock b/.speakeasy/workflow.lock index 7f2425ec..5d030d24 100644 --- a/.speakeasy/workflow.lock +++ b/.speakeasy/workflow.lock @@ -2,8 +2,8 @@ speakeasyVersion: 1.787.0 sources: OpenRouter API: sourceNamespace: open-router-chat-completions-api - sourceRevisionDigest: sha256:10d80fa11ae336d9888d7f3df46d6dc9243c5045e32a3d18fad2806fee5e2f9f - sourceBlobDigest: sha256:1e0ded936cabba0ea8a5bdef6d43c48b790415ca4dad32509120e5011ffc776e + sourceRevisionDigest: sha256:4c2dbf9c234b26b24edfe5116142336eed02a40816d335ff118d1a9e4cc04793 + sourceBlobDigest: sha256:3fb0793c01d28ac5cabf0d0ba0f5be748962e398d5f88ae947c2b84a8a50753f tags: - latest - 1.0.0 @@ -11,10 +11,10 @@ targets: openrouter: source: OpenRouter API sourceNamespace: open-router-chat-completions-api - sourceRevisionDigest: sha256:10d80fa11ae336d9888d7f3df46d6dc9243c5045e32a3d18fad2806fee5e2f9f - sourceBlobDigest: sha256:1e0ded936cabba0ea8a5bdef6d43c48b790415ca4dad32509120e5011ffc776e + sourceRevisionDigest: sha256:4c2dbf9c234b26b24edfe5116142336eed02a40816d335ff118d1a9e4cc04793 + sourceBlobDigest: sha256:3fb0793c01d28ac5cabf0d0ba0f5be748962e398d5f88ae947c2b84a8a50753f codeSamplesNamespace: open-router-chat-completions-api-go-code-samples - codeSamplesRevisionDigest: sha256:1ff5b5e0e84b7eb334a7b3e309058fc79c47d80549b277e13f49bd2ffb41ea33 + codeSamplesRevisionDigest: sha256:f799a51fe3e3d344ab92b9cc12ba69be2584fc48c9864c348f216397f5bb60ca workflow: workflowVersion: 1.0.0 speakeasyVersion: 1.787.0 diff --git a/RELEASES.md b/RELEASES.md index d1f961f4..2202f003 100644 --- a/RELEASES.md +++ b/RELEASES.md @@ -2428,4 +2428,14 @@ Based on: ### Generated - [go v0.8.27] . ### Releases -- [Go v0.8.27] https://github.com/OpenRouterTeam/go-sdk/releases/tag/v0.8.27 - . \ No newline at end of file +- [Go v0.8.27] https://github.com/OpenRouterTeam/go-sdk/releases/tag/v0.8.27 - . + +## 2026-09-25 19:39:50 +### Changes +Based on: +- OpenAPI Doc +- Speakeasy CLI 1.787.0 (2.914.0) https://github.com/speakeasy-api/speakeasy +### Generated +- [go v0.8.28] . +### Releases +- [Go v0.8.28] https://github.com/OpenRouterTeam/go-sdk/releases/tag/v0.8.28 - . \ No newline at end of file diff --git a/docs/models/components/publicendpoint.mdx b/docs/models/components/publicendpoint.mdx index 77bda579..70ca038b 100644 --- a/docs/models/components/publicendpoint.mdx +++ b/docs/models/components/publicendpoint.mdx @@ -22,7 +22,9 @@ Information about a specific model endpoint | `Quantization` | [*components.Quantization](../../models/components/quantization.mdx) | :heavy_check_mark: | N/A | fp16 | | `Status` | [*components.EndpointStatus](../../models/components/endpointstatus.mdx) | :heavy_minus_sign: | N/A | 0 | | `SupportedParameters` | [][components.Parameter](../../models/components/parameter.mdx) | :heavy_check_mark: | N/A | | +| `SupportsImageReference` | `*bool` | :heavy_minus_sign: | Whether this TTS endpoint accepts an `image_url` reference describing the desired voice. Requests carrying an image reference are only routed to endpoints where this is true. | | | `SupportsImplicitCaching` | `bool` | :heavy_check_mark: | N/A | | +| `SupportsMultipleAudioReferences` | `*bool` | :heavy_minus_sign: | Whether this TTS endpoint accepts more than one `input_audio` reference clip per request. Requests carrying several clips are only routed to endpoints where this is true. | | | `SupportsToolChoice` | [components.ToolChoiceSupport](../../models/components/toolchoicesupport.mdx) | :heavy_check_mark: | Per-variant `tool_choice` support. `tool_choice` in `supported_parameters` only says the parameter is accepted; these flags say which of its values passed testing. | \{
"auto": true,
"function": true,
"none": true,
"required": true
} | | `SupportsVoiceCloning` | `*bool` | :heavy_minus_sign: | Whether this TTS endpoint accepts inline reference audio (`input_references`) for stateless voice cloning. Requests carrying reference audio are only routed to endpoints where this is true. | | | `Tag` | `string` | :heavy_check_mark: | N/A | | diff --git a/docs/models/components/speechinputreference.mdx b/docs/models/components/speechinputreference.mdx index 3c918c41..fc519a75 100644 --- a/docs/models/components/speechinputreference.mdx +++ b/docs/models/components/speechinputreference.mdx @@ -2,11 +2,17 @@ title: "SpeechInputReference" --- -Reference content part for stateless voice cloning +Reference content part for stateless voice cloning or voice design ## Supported Types +### SpeechInputReferenceImage + +```go +speechInputReference := components.CreateSpeechInputReferenceImageURL(components.SpeechInputReferenceImage{/* values here */}) +``` + ### SpeechInputReferenceAudio ```go @@ -25,6 +31,8 @@ Use the `Type` field to determine which variant is active, then access the corre ```go switch speechInputReference.Type { + case components.SpeechInputReferenceTypeImageURL: + // speechInputReference.SpeechInputReferenceImage is populated case components.SpeechInputReferenceTypeInputAudio: // speechInputReference.SpeechInputReferenceAudio is populated case components.SpeechInputReferenceTypeText: diff --git a/docs/models/components/speechinputreferenceaudio.mdx b/docs/models/components/speechinputreferenceaudio.mdx index e52e761a..c2bbb82b 100644 --- a/docs/models/components/speechinputreferenceaudio.mdx +++ b/docs/models/components/speechinputreferenceaudio.mdx @@ -2,7 +2,7 @@ title: "SpeechInputReferenceAudio" --- -Reference audio input for stateless voice cloning +Reference audio input for stateless voice cloning. Up to three parts per request; the Nth audio part is addressable from `input` as `@AudioN` on providers that support multiple references. ## Fields diff --git a/docs/models/components/speechinputreferenceaudioinput.mdx b/docs/models/components/speechinputreferenceaudioinput.mdx index 9896ac9a..1dcc1498 100644 --- a/docs/models/components/speechinputreferenceaudioinput.mdx +++ b/docs/models/components/speechinputreferenceaudioinput.mdx @@ -7,7 +7,8 @@ Reference audio input object ## Fields -| Field | Type | Required | Description | Example | -| ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `Data` | `string` | :heavy_check_mark: | Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio). | data:audio/wav;base64,UklGRuQXDABXQVZF... | -| `Format` | `*string` | :heavy_minus_sign: | Audio format of the reference audio (e.g., wav, mp3). Optional; most providers detect the format from the audio bytes. | wav | \ No newline at end of file +| Field | Type | Required | Description | Example | +| --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `Data` | `*string` | :heavy_minus_sign: | Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio). Exactly one of `data` or `url` is required. | data:audio/wav;base64,UklGRuQXDABXQVZF... | +| `Format` | `*string` | :heavy_minus_sign: | Audio format of the reference audio (e.g., wav, mp3). Optional; most providers detect the format from the audio bytes. | wav | +| `URL` | `*string` | :heavy_minus_sign: | Public http(s) URL of the reference audio. OpenRouter downloads it (15 MiB max) and forwards the bytes, never the URL. Exactly one of `data` or `url` is required. | https://example.com/reference.wav | \ No newline at end of file diff --git a/docs/models/components/speechinputreferenceimage.mdx b/docs/models/components/speechinputreferenceimage.mdx new file mode 100644 index 00000000..a50429cd --- /dev/null +++ b/docs/models/components/speechinputreferenceimage.mdx @@ -0,0 +1,13 @@ +--- +title: "SpeechInputReferenceImage" +--- + +Reference image describing the desired voice. Cannot be combined with `input_audio` parts. Only routed to endpoints that support image references. + + +## Fields + +| Field | Type | Required | Description | +| ------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------ | +| `ImageURL` | [components.SpeechInputReferenceImageInput](../../models/components/speechinputreferenceimageinput.mdx) | :heavy_check_mark: | Reference image input object | +| `Type` | [components.SpeechInputReferenceImageType](../../models/components/speechinputreferenceimagetype.mdx) | :heavy_check_mark: | N/A | \ No newline at end of file diff --git a/docs/models/components/speechinputreferenceimageinput.mdx b/docs/models/components/speechinputreferenceimageinput.mdx new file mode 100644 index 00000000..5af0f2d2 --- /dev/null +++ b/docs/models/components/speechinputreferenceimageinput.mdx @@ -0,0 +1,12 @@ +--- +title: "SpeechInputReferenceImageInput" +--- + +Reference image input object + + +## Fields + +| Field | Type | Required | Description | Example | +| -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `URL` | `string` | :heavy_check_mark: | JPEG, PNG, or WebP reference image as a base64 data URI or a public http(s) URL. Remote images are downloaded (15 MiB max) and forwarded as bytes, never as the URL. | data:image/png;base64,iVBORw0KGgo... | \ No newline at end of file diff --git a/docs/models/components/speechinputreferenceimagetype.mdx b/docs/models/components/speechinputreferenceimagetype.mdx new file mode 100644 index 00000000..4f732841 --- /dev/null +++ b/docs/models/components/speechinputreferenceimagetype.mdx @@ -0,0 +1,20 @@ +--- +title: "SpeechInputReferenceImageType" +--- + +## Example Usage + +```go +import ( + "github.com/OpenRouterTeam/go-sdk/models/components" +) + +value := components.SpeechInputReferenceImageTypeImageURL +``` + + +## Values + +| Name | Value | +| --------------------------------------- | --------------------------------------- | +| `SpeechInputReferenceImageTypeImageURL` | image_url | \ No newline at end of file diff --git a/docs/models/components/speechinputreferencetext.mdx b/docs/models/components/speechinputreferencetext.mdx index 919b6cb8..c8ab6031 100644 --- a/docs/models/components/speechinputreferencetext.mdx +++ b/docs/models/components/speechinputreferencetext.mdx @@ -2,12 +2,12 @@ title: "SpeechInputReferenceText" --- -Transcript of the accompanying reference audio +Transcript of an `input_audio` part ## Fields -| Field | Type | Required | Description | Example | -| -------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------- | -| `Text` | `string` | :heavy_check_mark: | Transcript of the accompanying reference audio. | I used to rule the world. | -| `Type` | [components.SpeechInputReferenceTextType](../../models/components/speechinputreferencetexttype.mdx) | :heavy_check_mark: | N/A | | \ No newline at end of file +| Field | Type | Required | Description | Example | +| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `Text` | `string` | :heavy_check_mark: | Transcript of an `input_audio` part. With a single clip it may appear before or after the clip; with multiple clips it must immediately follow the clip it transcribes. | I used to rule the world. | +| `Type` | [components.SpeechInputReferenceTextType](../../models/components/speechinputreferencetexttype.mdx) | :heavy_check_mark: | N/A | | \ No newline at end of file diff --git a/docs/models/components/speechrequest.mdx b/docs/models/components/speechrequest.mdx index ba578243..03bd63ab 100644 --- a/docs/models/components/speechrequest.mdx +++ b/docs/models/components/speechrequest.mdx @@ -7,15 +7,15 @@ Text-to-speech request input ## Fields -| Field | Type | Required | Description | Example | -| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `Input` | `string` | :heavy_check_mark: | Text to synthesize | Hello world | -| `InputReferences` | [][components.SpeechInputReference](../../models/components/speechinputreference.mdx) | :heavy_minus_sign: | Reference content for stateless voice cloning: one `input_audio` part carrying the voice sample, optionally accompanied by one `text` part with its transcript. Only routed to endpoints that support voice cloning. | [
\{
"input_audio": \{
"data": "data:audio/wav;base64,UklGRuQXDABXQVZF..."
},
"type": "input_audio"
},
\{
"text": "I used to rule the world.",
"type": "text"
}
] | -| `Model` | `string` | :heavy_check_mark: | TTS model identifier | mistralai/voxtral-mini-tts-2603 | -| `Provider` | [*components.SpeechRequestProvider](../../models/components/speechrequestprovider.mdx) | :heavy_minus_sign: | Provider-specific passthrough configuration | | -| `ResponseFormat` | [*components.SpeechRequestResponseFormat](../../models/components/speechrequestresponseformat.mdx) | :heavy_minus_sign: | Audio output format | pcm | -| `SessionID` | `*string` | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | session-1234 | -| `Speed` | `*float64` | :heavy_minus_sign: | Playback speed multiplier. Only used by models that support it (e.g. OpenAI TTS). Ignored by other providers. | 1 | -| `Trace` | [*components.TraceConfig](../../models/components/traceconfig.mdx) | :heavy_minus_sign: | Metadata for observability and tracing. Known keys (trace_id, trace_name, span_name, generation_name, parent_span_id) have special handling. Additional keys are passed through as custom metadata to configured broadcast destinations. | \{
"trace_id": "trace-abc123",
"trace_name": "my-app-trace"
} | -| `User` | `*string` | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | user-1234 | -| `Voice` | `*string` | :heavy_minus_sign: | Voice identifier (provider-specific). | en_paul_neutral | \ No newline at end of file +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `Input` | `string` | :heavy_check_mark: | Text to synthesize | Hello world | +| `InputReferences` | [][components.SpeechInputReference](../../models/components/speechinputreference.mdx) | :heavy_minus_sign: | Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference. | [
\{
"input_audio": \{
"data": "data:audio/wav;base64,UklGRuQXDABXQVZF..."
},
"type": "input_audio"
},
\{
"text": "I used to rule the world.",
"type": "text"
}
] | +| `Model` | `string` | :heavy_check_mark: | TTS model identifier | mistralai/voxtral-mini-tts-2603 | +| `Provider` | [*components.SpeechRequestProvider](../../models/components/speechrequestprovider.mdx) | :heavy_minus_sign: | Provider-specific passthrough configuration | | +| `ResponseFormat` | [*components.SpeechRequestResponseFormat](../../models/components/speechrequestresponseformat.mdx) | :heavy_minus_sign: | Audio output format | pcm | +| `SessionID` | `*string` | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | session-1234 | +| `Speed` | `*float64` | :heavy_minus_sign: | Playback speed multiplier. Only used by models that support it (e.g. OpenAI TTS). Ignored by other providers. | 1 | +| `Trace` | [*components.TraceConfig](../../models/components/traceconfig.mdx) | :heavy_minus_sign: | Metadata for observability and tracing. Known keys (trace_id, trace_name, span_name, generation_name, parent_span_id) have special handling. Additional keys are passed through as custom metadata to configured broadcast destinations. | \{
"trace_id": "trace-abc123",
"trace_name": "my-app-trace"
} | +| `User` | `*string` | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | user-1234 | +| `Voice` | `*string` | :heavy_minus_sign: | Voice identifier (provider-specific). | en_paul_neutral | \ No newline at end of file diff --git a/models/components/publicendpoint.go b/models/components/publicendpoint.go index 4435f334..bb11f4a2 100644 --- a/models/components/publicendpoint.go +++ b/models/components/publicendpoint.go @@ -519,13 +519,17 @@ type PublicEndpoint struct { ModelName string `json:"model_name"` Name string `json:"name"` // Endpoint performance over the last 30 minutes, keyed by the kind of request served (e.g. `text_generation`, `image_generation`). Additive to the legacy singular latency and throughput fields; image and video generation report end-to-end latency. Only visible when authenticated with an API key or cookie. - PerfLast30mByWorkload *PerfLast30mByWorkload `json:"perf_last_30m_by_workload,omitzero"` - Pricing Pricing `json:"pricing"` - ProviderName ProviderName `json:"provider_name"` - Quantization *Quantization `json:"quantization"` - Status *EndpointStatus `json:"status,omitzero"` - SupportedParameters []Parameter `json:"supported_parameters"` - SupportsImplicitCaching bool `json:"supports_implicit_caching"` + PerfLast30mByWorkload *PerfLast30mByWorkload `json:"perf_last_30m_by_workload,omitzero"` + Pricing Pricing `json:"pricing"` + ProviderName ProviderName `json:"provider_name"` + Quantization *Quantization `json:"quantization"` + Status *EndpointStatus `json:"status,omitzero"` + SupportedParameters []Parameter `json:"supported_parameters"` + // Whether this TTS endpoint accepts an `image_url` reference describing the desired voice. Requests carrying an image reference are only routed to endpoints where this is true. + SupportsImageReference *bool `default:"false" json:"supports_image_reference"` + SupportsImplicitCaching bool `json:"supports_implicit_caching"` + // Whether this TTS endpoint accepts more than one `input_audio` reference clip per request. Requests carrying several clips are only routed to endpoints where this is true. + SupportsMultipleAudioReferences *bool `default:"false" json:"supports_multiple_audio_references"` // Per-variant `tool_choice` support. `tool_choice` in `supported_parameters` only says the parameter is accepted; these flags say which of its values passed testing. SupportsToolChoice ToolChoiceSupport `json:"supports_tool_choice"` // Whether this TTS endpoint accepts inline reference audio (`input_references`) for stateless voice cloning. Requests carrying reference audio are only routed to endpoints where this is true. @@ -641,6 +645,13 @@ func (p *PublicEndpoint) GetSupportedParameters() []Parameter { return p.SupportedParameters } +func (p *PublicEndpoint) GetSupportsImageReference() *bool { + if p == nil { + return nil + } + return p.SupportsImageReference +} + func (p *PublicEndpoint) GetSupportsImplicitCaching() bool { if p == nil { return false @@ -648,6 +659,13 @@ func (p *PublicEndpoint) GetSupportsImplicitCaching() bool { return p.SupportsImplicitCaching } +func (p *PublicEndpoint) GetSupportsMultipleAudioReferences() *bool { + if p == nil { + return nil + } + return p.SupportsMultipleAudioReferences +} + func (p *PublicEndpoint) GetSupportsToolChoice() ToolChoiceSupport { if p == nil { return ToolChoiceSupport{} diff --git a/models/components/speechinputreference.go b/models/components/speechinputreference.go index 43f57baf..0447cba9 100644 --- a/models/components/speechinputreference.go +++ b/models/components/speechinputreference.go @@ -12,18 +12,32 @@ import ( type SpeechInputReferenceType string const ( + SpeechInputReferenceTypeImageURL SpeechInputReferenceType = "image_url" SpeechInputReferenceTypeInputAudio SpeechInputReferenceType = "input_audio" SpeechInputReferenceTypeText SpeechInputReferenceType = "text" ) -// SpeechInputReference - Reference content part for stateless voice cloning +// SpeechInputReference - Reference content part for stateless voice cloning or voice design type SpeechInputReference struct { SpeechInputReferenceAudio *SpeechInputReferenceAudio `queryParam:"inline" union:"member"` SpeechInputReferenceText *SpeechInputReferenceText `queryParam:"inline" union:"member"` + SpeechInputReferenceImage *SpeechInputReferenceImage `queryParam:"inline" union:"member"` Type SpeechInputReferenceType } +func CreateSpeechInputReferenceImageURL(imageURL SpeechInputReferenceImage) SpeechInputReference { + typ := SpeechInputReferenceTypeImageURL + + typStr := SpeechInputReferenceImageType(typ) + imageURL.Type = typStr + + return SpeechInputReference{ + SpeechInputReferenceImage: &imageURL, + Type: typ, + } +} + func CreateSpeechInputReferenceInputAudio(inputAudio SpeechInputReferenceAudio) SpeechInputReference { typ := SpeechInputReferenceTypeInputAudio @@ -60,6 +74,15 @@ func (u *SpeechInputReference) UnmarshalJSON(data []byte) error { } switch dis.Type { + case "image_url": + speechInputReferenceImage := new(SpeechInputReferenceImage) + if err := utils.UnmarshalJSON(data, &speechInputReferenceImage, "", true, nil); err != nil { + return fmt.Errorf("could not unmarshal `%s` into expected (Type == image_url) type SpeechInputReferenceImage within SpeechInputReference: %w", string(data), err) + } + + u.SpeechInputReferenceImage = speechInputReferenceImage + u.Type = SpeechInputReferenceTypeImageURL + return nil case "input_audio": speechInputReferenceAudio := new(SpeechInputReferenceAudio) if err := utils.UnmarshalJSON(data, &speechInputReferenceAudio, "", true, nil); err != nil { @@ -92,5 +115,9 @@ func (u SpeechInputReference) MarshalJSON() ([]byte, error) { return utils.MarshalJSON(u.SpeechInputReferenceText, "", true) } + if u.SpeechInputReferenceImage != nil { + return utils.MarshalJSON(u.SpeechInputReferenceImage, "", true) + } + return nil, errors.New("could not marshal union type SpeechInputReference: all fields are null") } diff --git a/models/components/speechinputreferenceaudio.go b/models/components/speechinputreferenceaudio.go index 643f7485..e83f7b82 100644 --- a/models/components/speechinputreferenceaudio.go +++ b/models/components/speechinputreferenceaudio.go @@ -31,7 +31,7 @@ func (e *SpeechInputReferenceAudioType) UnmarshalJSON(data []byte) error { } } -// SpeechInputReferenceAudio - Reference audio input for stateless voice cloning +// SpeechInputReferenceAudio - Reference audio input for stateless voice cloning. Up to three parts per request; the Nth audio part is addressable from `input` as `@AudioN` on providers that support multiple references. type SpeechInputReferenceAudio struct { // Reference audio input object InputAudio SpeechInputReferenceAudioInput `json:"input_audio"` diff --git a/models/components/speechinputreferenceaudioinput.go b/models/components/speechinputreferenceaudioinput.go index 738c1a9e..faf947a2 100644 --- a/models/components/speechinputreferenceaudioinput.go +++ b/models/components/speechinputreferenceaudioinput.go @@ -8,10 +8,12 @@ import ( // SpeechInputReferenceAudioInput - Reference audio input object type SpeechInputReferenceAudioInput struct { - // Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio). - Data string `json:"data"` + // Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio). Exactly one of `data` or `url` is required. + Data *string `json:"data,omitzero"` // Audio format of the reference audio (e.g., wav, mp3). Optional; most providers detect the format from the audio bytes. Format *string `json:"format,omitzero"` + // Public http(s) URL of the reference audio. OpenRouter downloads it (15 MiB max) and forwards the bytes, never the URL. Exactly one of `data` or `url` is required. + URL *string `json:"url,omitzero"` } func (s SpeechInputReferenceAudioInput) MarshalJSON() ([]byte, error) { @@ -25,9 +27,9 @@ func (s *SpeechInputReferenceAudioInput) UnmarshalJSON(data []byte) error { return nil } -func (s *SpeechInputReferenceAudioInput) GetData() string { +func (s *SpeechInputReferenceAudioInput) GetData() *string { if s == nil { - return "" + return nil } return s.Data } @@ -38,3 +40,10 @@ func (s *SpeechInputReferenceAudioInput) GetFormat() *string { } return s.Format } + +func (s *SpeechInputReferenceAudioInput) GetURL() *string { + if s == nil { + return nil + } + return s.URL +} diff --git a/models/components/speechinputreferenceimage.go b/models/components/speechinputreferenceimage.go new file mode 100644 index 00000000..9d1ef4d0 --- /dev/null +++ b/models/components/speechinputreferenceimage.go @@ -0,0 +1,64 @@ +// Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT. + +package components + +import ( + "encoding/json" + "fmt" + "github.com/OpenRouterTeam/go-sdk/internal/utils" +) + +type SpeechInputReferenceImageType string + +const ( + SpeechInputReferenceImageTypeImageURL SpeechInputReferenceImageType = "image_url" +) + +func (e SpeechInputReferenceImageType) ToPointer() *SpeechInputReferenceImageType { + return &e +} +func (e *SpeechInputReferenceImageType) UnmarshalJSON(data []byte) error { + var v string + if err := json.Unmarshal(data, &v); err != nil { + return err + } + switch v { + case "image_url": + *e = SpeechInputReferenceImageType(v) + return nil + default: + return fmt.Errorf("invalid value for SpeechInputReferenceImageType: %v", v) + } +} + +// SpeechInputReferenceImage - Reference image describing the desired voice. Cannot be combined with `input_audio` parts. Only routed to endpoints that support image references. +type SpeechInputReferenceImage struct { + // Reference image input object + ImageURL SpeechInputReferenceImageInput `json:"image_url"` + Type SpeechInputReferenceImageType `json:"type"` +} + +func (s SpeechInputReferenceImage) MarshalJSON() ([]byte, error) { + return utils.MarshalJSON(s, "", false) +} + +func (s *SpeechInputReferenceImage) UnmarshalJSON(data []byte) error { + if err := utils.UnmarshalJSON(data, &s, "", false, nil); err != nil { + return err + } + return nil +} + +func (s *SpeechInputReferenceImage) GetImageURL() SpeechInputReferenceImageInput { + if s == nil { + return SpeechInputReferenceImageInput{} + } + return s.ImageURL +} + +func (s *SpeechInputReferenceImage) GetType() SpeechInputReferenceImageType { + if s == nil { + return SpeechInputReferenceImageType("") + } + return s.Type +} diff --git a/models/components/speechinputreferenceimageinput.go b/models/components/speechinputreferenceimageinput.go new file mode 100644 index 00000000..8209feef --- /dev/null +++ b/models/components/speechinputreferenceimageinput.go @@ -0,0 +1,31 @@ +// Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT. + +package components + +import ( + "github.com/OpenRouterTeam/go-sdk/internal/utils" +) + +// SpeechInputReferenceImageInput - Reference image input object +type SpeechInputReferenceImageInput struct { + // JPEG, PNG, or WebP reference image as a base64 data URI or a public http(s) URL. Remote images are downloaded (15 MiB max) and forwarded as bytes, never as the URL. + URL string `json:"url"` +} + +func (s SpeechInputReferenceImageInput) MarshalJSON() ([]byte, error) { + return utils.MarshalJSON(s, "", false) +} + +func (s *SpeechInputReferenceImageInput) UnmarshalJSON(data []byte) error { + if err := utils.UnmarshalJSON(data, &s, "", false, nil); err != nil { + return err + } + return nil +} + +func (s *SpeechInputReferenceImageInput) GetURL() string { + if s == nil { + return "" + } + return s.URL +} diff --git a/models/components/speechinputreferencetext.go b/models/components/speechinputreferencetext.go index 0b13751c..282892fc 100644 --- a/models/components/speechinputreferencetext.go +++ b/models/components/speechinputreferencetext.go @@ -31,9 +31,9 @@ func (e *SpeechInputReferenceTextType) UnmarshalJSON(data []byte) error { } } -// SpeechInputReferenceText - Transcript of the accompanying reference audio +// SpeechInputReferenceText - Transcript of an `input_audio` part type SpeechInputReferenceText struct { - // Transcript of the accompanying reference audio. + // Transcript of an `input_audio` part. With a single clip it may appear before or after the clip; with multiple clips it must immediately follow the clip it transcribes. Text string `json:"text"` Type SpeechInputReferenceTextType `json:"type"` } diff --git a/models/components/speechrequest.go b/models/components/speechrequest.go index a366d643..e50e827a 100644 --- a/models/components/speechrequest.go +++ b/models/components/speechrequest.go @@ -57,7 +57,7 @@ func (e *SpeechRequestResponseFormat) IsExact() bool { type SpeechRequest struct { // Text to synthesize Input string `json:"input"` - // Reference content for stateless voice cloning: one `input_audio` part carrying the voice sample, optionally accompanied by one `text` part with its transcript. Only routed to endpoints that support voice cloning. + // Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference. InputReferences []SpeechInputReference `json:"input_references,omitzero"` // TTS model identifier Model string `json:"model"` diff --git a/openrouter.go b/openrouter.go index b7af4d8b..97325b43 100644 --- a/openrouter.go +++ b/openrouter.go @@ -71,20 +71,6 @@ type OpenRouter struct { Benchmarks *Benchmarks // BYOK endpoints BYOK *BYOK - // Stream a chat completion with an intern - // Sends a prompt to one of your interns and streams the reply as OpenAI-compatible server-sent events ending with `[DONE]`. The run executes on the intern, which may pause to ask you something. It then streams one `openrouter.provide_input` tool call and finishes with `finish_reason: "tool_calls"`, and the run stays open on the intern. - // - // Every response, whether it ends with `stop`, `tool_calls` or `error`, is followed by a final chunk with empty `choices` that carries `session_id`, then `data: [DONE]`. That chunk carries the `usage` the intern reported for the run, after `stop` or `error`, and `null` when the intern reported none. After `tool_calls` its `usage` is `null` because the turn is not over. Read through `[DONE]`: the `session_id` you need to reply arrives after the `tool_calls` finish chunk. - // - // To answer, send a second request with the same `session_id`, the assistant message echoing that tool call, and a `tool` message whose `tool_call_id` is the tool call id and whose `content` is the answer. The answer is delivered to the run that asked and the stream continues from where it paused. A question stays open for its interaction deadline (5 minutes by default) and the run is cancelled when that passes. Rejected replies do not extend the deadline. - // - // Closing the connection after the `[DONE]` that follows `finish_reason: "tool_calls"` keeps the run alive. Disconnecting while a response is still streaming cancels the run. The stream writes a `: keepalive` comment whenever nothing else has been written for 30 seconds, so a disconnect is noticed within that interval even while the intern is silent. - // - // A run the intern ends while you are still connected, by cancellation or by a deadline, ends the stream with a `finish_reason: "error"` chunk carrying `410` and reason `run_ended`, then the final empty-`choices` chunk and `[DONE]`. That error reports only an ending the intern confirmed. A connection that breaks without that confirmation ends with reason `stream_severed`, and a client that has already disconnected is promised no final event. - // - // Set `approval_mode` to `manual` to have the intern ask before approval-bearing tools such as the shell. Omitted, the run self-drives and consents on your behalf. The mode belongs to the run started by that prompt and must be repeated on later prompts. - // - // Available to interns programme members. Callers outside the programme receive `404` for every path under `/api/v1/interns`. Chat *Chat // Task classification market-share endpoints Classifications *Classifications @@ -226,9 +212,9 @@ func WithTimeout(timeout time.Duration) SDKOption { // New creates a new instance of the SDK with the provided options func New(opts ...SDKOption) *OpenRouter { sdk := &OpenRouter{ - SDKVersion: "0.8.27", + SDKVersion: "0.8.28", sdkConfiguration: config.SDKConfiguration{ - UserAgent: "speakeasy-sdk/go 0.8.27 2.914.0 1.0.0 github.com/OpenRouterTeam/go-sdk", + UserAgent: "speakeasy-sdk/go 0.8.28 2.914.0 1.0.0 github.com/OpenRouterTeam/go-sdk", Globals: globals.Globals{}, ServerList: ServerList, }, From 40959496e9f1d2cc9768d95aaa24075de7ec96a3 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Fri, 25 Sep 2026 19:46:21 +0000 Subject: [PATCH 2/2] docs: sync install snippets to v0.8.28 --- OVERVIEW.md | 2 +- README.md | 2 +- doc.go | 2 +- example_test.go | 2 +- examples/README.md | 4 ++-- 5 files changed, 6 insertions(+), 6 deletions(-) diff --git a/OVERVIEW.md b/OVERVIEW.md index 8c820220..38c60bc8 100644 --- a/OVERVIEW.md +++ b/OVERVIEW.md @@ -117,7 +117,7 @@ go get github.com/OpenRouterTeam/go-sdk For beta releases, pin an explicit version: ```bash -go get github.com/OpenRouterTeam/go-sdk@v0.8.27 +go get github.com/OpenRouterTeam/go-sdk@v0.8.28 ``` **Requirements:** Go 1.25 or higher diff --git a/README.md b/README.md index 50fa020c..93bbb9a1 100644 --- a/README.md +++ b/README.md @@ -13,7 +13,7 @@ To learn more, see the [API Reference](https://openrouter.ai/docs/sdks/go/api-re > This SDK is in **beta**. Pin to a specific version to avoid unexpected breaking changes: > > ```bash -> go get github.com/OpenRouterTeam/go-sdk@v0.8.27 +> go get github.com/OpenRouterTeam/go-sdk@v0.8.28 > ``` diff --git a/doc.go b/doc.go index 93f184fe..df0a6741 100644 --- a/doc.go +++ b/doc.go @@ -6,7 +6,7 @@ provider selection, and unified billing. This SDK is in beta. Pin to a specific module version to avoid unexpected breaking changes: - go get github.com/OpenRouterTeam/go-sdk@v0.8.27 + go get github.com/OpenRouterTeam/go-sdk@v0.8.28 For full API documentation, visit: https://openrouter.ai/docs/client-sdks/go/overview diff --git a/example_test.go b/example_test.go index 4adc46be..fe1f8cb8 100644 --- a/example_test.go +++ b/example_test.go @@ -18,7 +18,7 @@ func ExampleNew() { openrouter.WithSecurity("your-api-key"), ) fmt.Println(sdk.SDKVersion) - // Output: 0.8.27 + // Output: 0.8.28 } // Example demonstrates basic usage of the OpenRouter SDK for chat completions. diff --git a/examples/README.md b/examples/README.md index 068aef1c..e4237b31 100644 --- a/examples/README.md +++ b/examples/README.md @@ -43,7 +43,7 @@ cd ../generation && go run . Each example pins a released SDK version in `go.mod`: ```go -require github.com/OpenRouterTeam/go-sdk v0.8.27 +require github.com/OpenRouterTeam/go-sdk v0.8.28 ``` This should match the version in [README.md](../README.md) and `.speakeasy/gen.lock` `releaseVersion`. CI runs `scripts/bump-examples.sh` after releases to keep these in sync. @@ -51,7 +51,7 @@ This should match the version in [README.md](../README.md) and `.speakeasy/gen.l To use a different version: ```bash -go get github.com/OpenRouterTeam/go-sdk@v0.8.27 +go get github.com/OpenRouterTeam/go-sdk@v0.8.28 go run . ```