diff --git a/.speakeasy/gen.lock b/.speakeasy/gen.lock index 860268ec8..5cae47c77 100644 --- a/.speakeasy/gen.lock +++ b/.speakeasy/gen.lock @@ -1,19 +1,19 @@ lockVersion: "" id: "" management: - docChecksum: 544f51110e99d7005f82f3949f348ea0 + docChecksum: 4162460b4e4c0ad3640d7ef614732117 docVersion: 1.0.0 speakeasyVersion: 1.787.0 generationVersion: 2.914.0 - releaseVersion: 1.3.28 - configChecksum: 4c6f8f74944865d2ac7f5c55657dbad9 + releaseVersion: 1.3.29 + configChecksum: 691cf13390450c7b27d1758ff9904045 repoURL: https://github.com/OpenRouterTeam/typescript-sdk.git installationURL: https://github.com/OpenRouterTeam/typescript-sdk published: true persistentEdits: - generation_id: febd10d5-46be-4c46-bda0-7e0520512880 - pristine_commit_hash: 9ac2ef016574a0560d500d75aa4b822613cd4726 - pristine_tree_hash: e251c7d6ab3e285aa3ce6742c55f8e92411289e6 + generation_id: 75a4fbcd-178d-4d65-9e31-a7648fa6de61 + pristine_commit_hash: 33f456ce6d73d27204af6f589637c84233c03d9b + pristine_tree_hash: dcbe4d107655a18be7144f87f0578944315c793e features: typescript: acceptHeaders: 2.81.2 @@ -11080,8 +11080,8 @@ trackedFiles: pristine_git_object: f39845ad4cd48709b0a95065779be54f1be55997 docs/models/publicendpoint.mdx: id: 298950f1060d - last_write_checksum: sha1:c8a941ca7b2ce1d1ee7962aed38793a197ca9f13 - pristine_git_object: 6a931fa03809aa9b35f8df8fe5af92709769b097 + last_write_checksum: sha1:6419531ade48974b983d7befdd3c448ba9530bf4 + pristine_git_object: b56a7d583a6e0a1ebfaba92d3ed8231a44e5bc28 docs/models/publicpricing.mdx: id: da7987d6141a last_write_checksum: sha1:49396adb88727e3fdaff06767101a5ef4479257f @@ -11955,20 +11955,28 @@ trackedFiles: deleted: true docs/models/speechinputreference.mdx: id: 185904811fc6 - last_write_checksum: sha1:4ad4ef214e73bb6e118d2866889eaaf5f8764211 - pristine_git_object: d95674028f67a020208128b7922319355889f3fd + last_write_checksum: sha1:dbf62d978e3b54f67edcb17963059946cb6e3034 + pristine_git_object: a03f5f3401eba71192d69dc880a7c376240b2725 docs/models/speechinputreferenceaudio.mdx: id: 396eede2c948 - last_write_checksum: sha1:b24b8cb685ec43a6eed6be8ba5351d88560aa8af - pristine_git_object: 260053ef6e55cdb1cd05af0f1179318d673b9456 + last_write_checksum: sha1:aac9e5eeafddf6f7b2d87cd9551b988a15431dae + pristine_git_object: 72f677f7bcaaf93302c6a1ff5f081e2e64565fe3 docs/models/speechinputreferenceaudioinput.mdx: id: 38c61bd6aa17 - last_write_checksum: sha1:d0639ffc05bb2da9266106afe4b95b795414f89c - pristine_git_object: cb044aa5e1618cc788634fa73d5f9b47758d0225 + last_write_checksum: sha1:af7d9e585a5d1400b80aa274ff66961a8c0fc3e4 + pristine_git_object: 7dc5c2cc4a3ed74470279348f73a2a1f7ce03037 + docs/models/speechinputreferenceimage.mdx: + id: 1230dca41222 + last_write_checksum: sha1:4ea7d8b18d598ac8dca810b1e8d5ae894e621a48 + pristine_git_object: 042996bad0e03d87255de61189ff5af4f55d25ac + docs/models/speechinputreferenceimageinput.mdx: + id: ff3c72c30311 + last_write_checksum: sha1:85e6ad83336c8be129d177bd7dbabf27c9676c8b + pristine_git_object: 871636012da4255b36cd292662ac145d97879314 docs/models/speechinputreferencetext.mdx: id: a8fcc50d87e3 - last_write_checksum: sha1:2b5ed0f197bb4fc925970abff3b90141c2f2ef10 - pristine_git_object: d282057cc9377b1ffac4acd9a580f5fe82249057 + last_write_checksum: sha1:2c053d13d524bc73506f692a88502091f3aa93b7 + pristine_git_object: 95f3c95b545dfaac21a509a9ee8ced2a069be741 docs/models/speechrequest.md: id: 1403855cf560 last_write_checksum: sha1:a5b9df15e47f688c711938f0760f0a5de6cfd7b5 @@ -11976,8 +11984,8 @@ trackedFiles: deleted: true docs/models/speechrequest.mdx: id: c9e3c03ff84e - last_write_checksum: sha1:3b688f1a97116c3f52e71ea2b9a4eb03eb39867a - pristine_git_object: 4acf9e895b7b28ef8db6d3217a8bb7f2599ca62e + last_write_checksum: sha1:45f14f2b4d74f3df613de2983d94739349396508 + pristine_git_object: e5647ce3ad6e014e37a66ae64fb9672775dff631 docs/models/speechrequestoptions.md: id: 2d3d36e99ad0 last_write_checksum: sha1:a5810a37b774c3e34bb1d943b64620bc31de2278 @@ -13858,12 +13866,12 @@ trackedFiles: pristine_git_object: 410efafd6a7f50d91ccb87131fedbe0c3d47e15a jsr.json: id: 7f6ab7767282 - last_write_checksum: sha1:a84b98446306e53f9f04431a32aafc1a7f19190d - pristine_git_object: b7cb50b39134ce807bbed9e1a9211fd179d1f510 + last_write_checksum: sha1:102ba02c930fcb4cc9ccbce36e0b164914c314e3 + pristine_git_object: a3abc6d47cc8be200d20aa141c9ad66aca4e7cb3 package.json: id: 7030d0b2f71b - last_write_checksum: sha1:feb9f09cd3ac7157f87970a12d120f26be1f214a - pristine_git_object: bb04a9b024f58dddfefd11b0ea188635195bb63f + last_write_checksum: sha1:b8ef604edeabb5d9243a401e136fcc8c8efc0180 + pristine_git_object: 565aa641c954336d780802a6a6c59846532fe5f6 src/core.ts: id: f431fdbcd144 last_write_checksum: sha1:5aa66b0b6a5964f3eea7f3098c2eb3c0ee9c0131 @@ -14382,8 +14390,8 @@ trackedFiles: pristine_git_object: a187e58707bdb726ca2aff74941efe7493422d4e src/lib/config.ts: id: 320761608fb3 - last_write_checksum: sha1:fc29463433712de9fe0c8a9fe00e66790503ecd4 - pristine_git_object: 23b5c39678dda1a2fbc4fdad617bbe7cea24d858 + last_write_checksum: sha1:98ab23fb46b946cfc5e61f10f6c6cd92c39ce97a + pristine_git_object: 07bcda2597e762b55cdd24d9f17a9c1b61483227 src/lib/dlv.ts: id: b1988214835a last_write_checksum: sha1:eaac763b22717206a6199104e0403ed17a4e2711 @@ -15784,8 +15792,8 @@ trackedFiles: pristine_git_object: 7abdfe096d3adff317db6837dc689c69db92e127 src/models/index.ts: id: f93644b0f37e - last_write_checksum: sha1:4efd1fde13b102e2ab46712c313193563d9f98cd - pristine_git_object: 72d92b2c5f06ed1994345dd09ae56b114d91d293 + last_write_checksum: sha1:970d0f9bfa1b668f99598a66d856a87c3c4d7740 + pristine_git_object: 930bfcc8382cd3b5df96c16c187db3335b7aa90f src/models/inputaudio.ts: id: 9bdc14c7565f last_write_checksum: sha1:2f8dc4c1d6a9c2927d9eb186e0ecedf7b81838e6 @@ -17140,8 +17148,8 @@ trackedFiles: pristine_git_object: 5960d3c3b97f3571ffdc2f633f503fdfc10ec88d src/models/publicendpoint.ts: id: 396ce3186017 - last_write_checksum: sha1:455804f56e64d434682cc3e49a7f939c613f64a1 - pristine_git_object: 7eb0e3b104b8cb336da9a7cac48dd14ca1daf667 + last_write_checksum: sha1:639fa91de62d1912ce709e32a7da43160f5e45a9 + pristine_git_object: b487e4ff11adcaed40ecb5089bf08ddf9cce8020 src/models/publicpricing.ts: id: 0a44a1c3bab5 last_write_checksum: sha1:bd478f319c40ea705b9705d3c73a9e4992a16877 @@ -17380,24 +17388,32 @@ trackedFiles: pristine_git_object: b7e831da0a7045eea3f87f9135d9e68af797df11 src/models/speechinputreference.ts: id: 25b19e4fa954 - last_write_checksum: sha1:3ecc9d1efba16becba862d3168838a139172cd77 - pristine_git_object: 25f478821f957fe0f07b2fded7d0533445605441 + last_write_checksum: sha1:4815d0bf1f8cb82c15020d02b475e6c7ec405203 + pristine_git_object: cf8f0de91dfa4d54fcd6537256e8517cdca70e52 src/models/speechinputreferenceaudio.ts: id: d7299510c0e7 - last_write_checksum: sha1:32c84c7f67b1f8ae068e73d03cdbd06a01d26bd5 - pristine_git_object: 798b33dab4254213969a730be682e9c8cf74cee7 + last_write_checksum: sha1:ba2635214a0c4af6b8294c2954b6c3a0ff87d4de + pristine_git_object: d22c73addae286fa5d86a9bb94d0e4cbbba02e56 src/models/speechinputreferenceaudioinput.ts: id: 7f6147039895 - last_write_checksum: sha1:7363e612401d8be5a04a7935251935792517186d - pristine_git_object: 04ff60eb78c2233ac73f12b426381dd6cfa34278 + last_write_checksum: sha1:3945dee32a21a83e86ddf902d0c127a08080dbce + pristine_git_object: 0efae28e8990a997fb30f216b564a2eb98dad940 + src/models/speechinputreferenceimage.ts: + id: ee7097a5c9df + last_write_checksum: sha1:9562f00e059837d30d027aa45f91242b3be52750 + pristine_git_object: 60fbd6a24bfffb2a3d9b0ca5e7f26b0e3e2f79ac + src/models/speechinputreferenceimageinput.ts: + id: 36100583e975 + last_write_checksum: sha1:26df4e569b83a32e612dcb89d0ffdf74b447d837 + pristine_git_object: db60a7d57e8c5e219bc4f9ab538b722cefdde057 src/models/speechinputreferencetext.ts: id: 88f12d1ebfa6 - last_write_checksum: sha1:80487a870d9c7e4c368b5aa084b1b15daf47864f - pristine_git_object: dd8e518866345a2b921986bff08878e424f3c69f + last_write_checksum: sha1:20c804f6ca2c37fea399e955491707f4f797d689 + pristine_git_object: 9736271f1eca14fc2be3d7281472b25fc79fec3e src/models/speechrequest.ts: id: 96d573d68420 - last_write_checksum: sha1:4c495f08e92920adb92e6edf454a80a7447addd3 - pristine_git_object: 8d8512c0e84c8fff8e07b6bd66af44f964552f02 + last_write_checksum: sha1:4e32364681988b66f4ea420e7ce902316a024971 + pristine_git_object: c70a2be33c926cf57a72595deea82a4c5c225717 src/models/stopservertoolswhencondition.ts: id: 3c90e3993bad last_write_checksum: sha1:b95847e6581593c6f1106971d0508acb2cce1106 @@ -18284,7 +18300,7 @@ examples: slug: "" responses: "200": - application/json: {"data": {"architecture": {"input_modalities": ["text"], "instruct_type": "chatml", "modality": "text->text", "output_modalities": ["text"], "tokenizer": "GPT"}, "created": 1692901234, "description": "GPT-4 is a large multimodal model that can solve difficult problems with greater accuracy.", "endpoints": [{"context_length": 8192, "latency_last_30m": {"p50": 0.25, "p75": 0.35, "p90": 0.48, "p99": 0.85}, "max_completion_tokens": 4096, "max_prompt_tokens": 8192, "model_id": "openai/gpt-4", "model_name": "GPT-4", "name": "OpenAI: GPT-4", "pricing": {"completion": "0.00006", "prompt": "0.00003"}, "provider_name": "OpenAI", "quantization": "fp16", "supported_parameters": ["temperature", "top_p", "max_tokens"], "supports_implicit_caching": true, "supports_tool_choice": {"auto": true, "function": true, "none": true, "required": true}, "supports_voice_cloning": false, "tag": "openai", "throughput_last_30m": {"p50": 45.2, "p75": 38.5, "p90": 28.3, "p99": 15.1}, "uptime_last_1d": 99.8, "uptime_last_30m": 99.5, "uptime_last_5m": 100}], "id": "openai/gpt-4", "name": "GPT-4"}} + application/json: {"data": {"architecture": {"input_modalities": ["text"], "instruct_type": "chatml", "modality": "text->text", "output_modalities": ["text"], "tokenizer": "GPT"}, "created": 1692901234, "description": "GPT-4 is a large multimodal model that can solve difficult problems with greater accuracy.", "endpoints": [{"context_length": 8192, "latency_last_30m": {"p50": 0.25, "p75": 0.35, "p90": 0.48, "p99": 0.85}, "max_completion_tokens": 4096, "max_prompt_tokens": 8192, "model_id": "openai/gpt-4", "model_name": "GPT-4", "name": "OpenAI: GPT-4", "pricing": {"completion": "0.00006", "prompt": "0.00003"}, "provider_name": "OpenAI", "quantization": "fp16", "supported_parameters": ["temperature", "top_p", "max_tokens"], "supports_image_reference": false, "supports_implicit_caching": true, "supports_multiple_audio_references": false, "supports_tool_choice": {"auto": true, "function": true, "none": true, "required": true}, "supports_voice_cloning": false, "tag": "openai", "throughput_last_30m": {"p50": 45.2, "p75": 38.5, "p90": 28.3, "p99": 15.1}, "uptime_last_1d": 99.8, "uptime_last_30m": 99.5, "uptime_last_5m": 100}], "id": "openai/gpt-4", "name": "GPT-4"}} "404": application/json: {"error": {"code": 404, "message": "Resource not found"}} "500": @@ -18295,7 +18311,7 @@ examples: speakeasy-default-list-endpoints-zdr: responses: "200": - application/json: {"data": [{"context_length": 8192, "latency_last_30m": {"p50": 0.25, "p75": 0.35, "p90": 0.48, "p99": 0.85}, "max_completion_tokens": 4096, "max_prompt_tokens": 8192, "model_id": "openai/gpt-4", "model_name": "GPT-4", "name": "OpenAI: GPT-4", "pricing": {"completion": "0.00006", "prompt": "0.00003"}, "provider_name": "OpenAI", "quantization": "fp16", "supported_parameters": ["temperature", "top_p", "max_tokens"], "supports_implicit_caching": true, "supports_tool_choice": {"auto": true, "function": true, "none": true, "required": true}, "supports_voice_cloning": false, "tag": "openai", "throughput_last_30m": {"p50": 45.2, "p75": 38.5, "p90": 28.3, "p99": 15.1}, "uptime_last_1d": 99.8, "uptime_last_30m": 99.5, "uptime_last_5m": 100}]} + application/json: {"data": [{"context_length": 8192, "latency_last_30m": {"p50": 0.25, "p75": 0.35, "p90": 0.48, "p99": 0.85}, "max_completion_tokens": 4096, "max_prompt_tokens": 8192, "model_id": "openai/gpt-4", "model_name": "GPT-4", "name": "OpenAI: GPT-4", "pricing": {"completion": "0.00006", "prompt": "0.00003"}, "provider_name": "OpenAI", "quantization": "fp16", "supported_parameters": ["temperature", "top_p", "max_tokens"], "supports_image_reference": false, "supports_implicit_caching": true, "supports_multiple_audio_references": false, "supports_tool_choice": {"auto": true, "function": true, "none": true, "required": true}, "supports_voice_cloning": false, "tag": "openai", "throughput_last_30m": {"p50": 45.2, "p75": 38.5, "p90": 28.3, "p99": 15.1}, "uptime_last_1d": 99.8, "uptime_last_30m": 99.5, "uptime_last_5m": 100}]} "500": application/json: {"error": {"code": 500, "message": "Internal Server Error"}} "403": @@ -21298,6 +21314,4 @@ examples: "500": application/json: {"error": {"code": 404, "message": "Intern not found"}} examplesVersion: 1.0.2 -releaseNotes: | - ## Typescript SDK Changes: - * `openrouter.interns.getInternDaemonAccess()`: `error.status[429]` **Removed** (Breaking ⚠️) +releaseNotes: "## Typescript SDK Changes:\n* `openrouter.tts.createSpeech()`: \n * `request.speechRequest.inputReferences[]` **Changed**\n* `openrouter.endpoints.listZdrEndpoints()`: `response.data[]` **Changed**\n* `openrouter.endpoints.list()`: `response.data.endpoints[]` **Changed**\n" diff --git a/.speakeasy/gen.yaml b/.speakeasy/gen.yaml index e6e9327d3..26e5dd314 100644 --- a/.speakeasy/gen.yaml +++ b/.speakeasy/gen.yaml @@ -37,7 +37,7 @@ generation: documentation: mintlify preApplyUnionDiscriminators: true typescript: - version: 1.3.28 + version: 1.3.29 acceptHeaderEnum: false additionalDependencies: dependencies: diff --git a/.speakeasy/out.openapi.yaml b/.speakeasy/out.openapi.yaml index df3c9168f..e24df630b 100644 --- a/.speakeasy/out.openapi.yaml +++ b/.speakeasy/out.openapi.yaml @@ -23983,7 +23983,9 @@ components: - 'temperature' - 'top_p' - 'max_tokens' + supports_image_reference: false supports_implicit_caching: true + supports_multiple_audio_references: false supports_tool_choice: auto: true function: true @@ -24275,8 +24277,16 @@ components: items: $ref: '#/components/schemas/Parameter' type: 'array' + supports_image_reference: + default: false + description: 'Whether this TTS endpoint accepts an `image_url` reference describing the desired voice. Requests carrying an image reference are only routed to endpoints where this is true.' + type: 'boolean' supports_implicit_caching: type: 'boolean' + supports_multiple_audio_references: + default: false + description: 'Whether this TTS endpoint accepts more than one `input_audio` reference clip per request. Requests carrying several clips are only routed to endpoints where this is true.' + type: 'boolean' supports_tool_choice: $ref: '#/components/schemas/ToolChoiceSupport' supports_voice_cloning: @@ -25941,17 +25951,19 @@ components: - $ref: '#/components/schemas/ContainerAutoEnvironment' - $ref: '#/components/schemas/ContainerReferenceEnvironment' SpeechInputReference: - description: 'Reference content part for stateless voice cloning' + description: 'Reference content part for stateless voice cloning or voice design' discriminator: mapping: + image_url: '#/components/schemas/SpeechInputReferenceImage' input_audio: '#/components/schemas/SpeechInputReferenceAudio' text: '#/components/schemas/SpeechInputReferenceText' propertyName: 'type' oneOf: - $ref: '#/components/schemas/SpeechInputReferenceAudio' - $ref: '#/components/schemas/SpeechInputReferenceText' + - $ref: '#/components/schemas/SpeechInputReferenceImage' SpeechInputReferenceAudio: - description: 'Reference audio input for stateless voice cloning' + description: 'Reference audio input for stateless voice cloning. Up to three parts per request; the Nth audio part is addressable from `input` as `@AudioN` on providers that support multiple references.' example: input_audio: data: 'data:audio/wav;base64,UklGRuQXDABXQVZF...' @@ -25971,7 +25983,7 @@ components: description: 'Reference audio input object' properties: data: - description: 'Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio).' + description: 'Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio). Exactly one of `data` or `url` is required.' example: 'data:audio/wav;base64,UklGRuQXDABXQVZF...' maxLength: 20971520 minLength: 1 @@ -25980,17 +25992,50 @@ components: description: 'Audio format of the reference audio (e.g., wav, mp3). Optional; most providers detect the format from the audio bytes.' example: 'wav' type: 'string' + url: + description: 'Public http(s) URL of the reference audio. OpenRouter downloads it (15 MiB max) and forwards the bytes, never the URL. Exactly one of `data` or `url` is required.' + example: 'https://example.com/reference.wav' + format: 'uri' + maxLength: 2048 + type: 'string' + type: 'object' + SpeechInputReferenceImage: + description: 'Reference image describing the desired voice. Cannot be combined with `input_audio` parts. Only routed to endpoints that support image references.' + example: + image_url: + url: 'data:image/png;base64,iVBORw0KGgo...' + type: 'image_url' + properties: + image_url: + $ref: '#/components/schemas/SpeechInputReferenceImageInput' + type: + enum: + - 'image_url' + type: 'string' required: - - 'data' + - 'type' + - 'image_url' + type: 'object' + SpeechInputReferenceImageInput: + description: 'Reference image input object' + properties: + url: + description: 'JPEG, PNG, or WebP reference image as a base64 data URI or a public http(s) URL. Remote images are downloaded (15 MiB max) and forwarded as bytes, never as the URL.' + example: 'data:image/png;base64,iVBORw0KGgo...' + maxLength: 20971520 + minLength: 1 + type: 'string' + required: + - 'url' type: 'object' SpeechInputReferenceText: - description: 'Transcript of the accompanying reference audio' + description: 'Transcript of an `input_audio` part' example: text: 'I used to rule the world.' type: 'text' properties: text: - description: 'Transcript of the accompanying reference audio.' + description: 'Transcript of an `input_audio` part. With a single clip it may appear before or after the clip; with multiple clips it must immediately follow the clip it transcribes.' example: 'I used to rule the world.' maxLength: 10000 type: 'string' @@ -26016,7 +26061,7 @@ components: example: 'Hello world' type: 'string' input_references: - description: 'Reference content for stateless voice cloning: one `input_audio` part carrying the voice sample, optionally accompanied by one `text` part with its transcript. Only routed to endpoints that support voice cloning.' + description: 'Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference.' example: - input_audio: data: 'data:audio/wav;base64,UklGRuQXDABXQVZF...' @@ -34228,7 +34273,9 @@ paths: - 'temperature' - 'top_p' - 'max_tokens' + supports_image_reference: false supports_implicit_caching: true + supports_multiple_audio_references: false supports_voice_cloning: false tag: 'openai' throughput_last_30m: @@ -34265,7 +34312,9 @@ paths: - 'temperature' - 'top_p' - 'max_tokens' + supports_image_reference: false supports_implicit_caching: true + supports_multiple_audio_references: false supports_voice_cloning: false tag: 'openai' throughput_last_30m: diff --git a/.speakeasy/workflow.lock b/.speakeasy/workflow.lock index 4c94de7a8..1a141dda9 100644 --- a/.speakeasy/workflow.lock +++ b/.speakeasy/workflow.lock @@ -2,8 +2,8 @@ speakeasyVersion: 1.787.0 sources: OpenRouter API: sourceNamespace: open-router-chat-completions-api - sourceRevisionDigest: sha256:acf5334f900397f46c8f6f0210ea9aae862a9477e6cd1917fe7a80ad67b9d979 - sourceBlobDigest: sha256:51a6ce7d852942fda55e59d75e5375a622741e93d33e291514f616fbfa926014 + sourceRevisionDigest: sha256:7a95ff3051101de85dd0b3ac09cad39c61622cfead6d13a21d455528e6fa608b + sourceBlobDigest: sha256:f894a0b4777222e7bbc223cc2ca6061ff83979edcb0cb148212130eca09c9bad tags: - latest - 1.0.0 @@ -11,10 +11,10 @@ targets: openrouter: source: OpenRouter API sourceNamespace: open-router-chat-completions-api - sourceRevisionDigest: sha256:acf5334f900397f46c8f6f0210ea9aae862a9477e6cd1917fe7a80ad67b9d979 - sourceBlobDigest: sha256:51a6ce7d852942fda55e59d75e5375a622741e93d33e291514f616fbfa926014 + sourceRevisionDigest: sha256:7a95ff3051101de85dd0b3ac09cad39c61622cfead6d13a21d455528e6fa608b + sourceBlobDigest: sha256:f894a0b4777222e7bbc223cc2ca6061ff83979edcb0cb148212130eca09c9bad codeSamplesNamespace: open-router-chat-completions-api-typescript-code-samples - codeSamplesRevisionDigest: sha256:9a9678c75ff4d97352ae552d1fa8828dd3ff0f7e120f3f0aed76610532ec5210 + codeSamplesRevisionDigest: sha256:c4007623380dcf6916f452db2244ffec65bebace120e52c563576675793b1b79 workflow: workflowVersion: 1.0.0 speakeasyVersion: 1.787.0 diff --git a/RELEASES.md b/RELEASES.md index 9f689e93a..5e09fdbc3 100644 --- a/RELEASES.md +++ b/RELEASES.md @@ -4066,4 +4066,14 @@ Based on: ### Generated - [typescript v1.3.28] . ### Releases -- [NPM v1.3.28] https://www.npmjs.com/package/@openrouter/sdk/v/1.3.28 - . \ No newline at end of file +- [NPM v1.3.28] https://www.npmjs.com/package/@openrouter/sdk/v/1.3.28 - . + +## 2026-09-25 19:39:40 +### Changes +Based on: +- OpenAPI Doc +- Speakeasy CLI 1.787.0 (2.914.0) https://github.com/speakeasy-api/speakeasy +### Generated +- [typescript v1.3.29] . +### Releases +- [NPM v1.3.29] https://www.npmjs.com/package/@openrouter/sdk/v/1.3.29 - . \ No newline at end of file diff --git a/docs/models/publicendpoint.mdx b/docs/models/publicendpoint.mdx index 6a931fa03..b56a7d583 100644 --- a/docs/models/publicendpoint.mdx +++ b/docs/models/publicendpoint.mdx @@ -70,7 +70,9 @@ let value: PublicEndpoint = { | `quantization` | [models.Quantization](../models/quantization.mdx) | :heavy_check_mark: | N/A | fp16 | | `status` | [models.EndpointStatus](../models/endpointstatus.mdx) | :heavy_minus_sign: | N/A | 0 | | `supportedParameters` | [models.Parameter](../models/parameter.mdx)[] | :heavy_check_mark: | N/A | | +| `supportsImageReference` | *boolean* | :heavy_minus_sign: | Whether this TTS endpoint accepts an `image_url` reference describing the desired voice. Requests carrying an image reference are only routed to endpoints where this is true. | | | `supportsImplicitCaching` | *boolean* | :heavy_check_mark: | N/A | | +| `supportsMultipleAudioReferences` | *boolean* | :heavy_minus_sign: | Whether this TTS endpoint accepts more than one `input_audio` reference clip per request. Requests carrying several clips are only routed to endpoints where this is true. | | | `supportsToolChoice` | [models.ToolChoiceSupport](../models/toolchoicesupport.mdx) | :heavy_check_mark: | Per-variant `tool_choice` support. `tool_choice` in `supported_parameters` only says the parameter is accepted; these flags say which of its values passed testing. | \{
"auto": true,
"function": true,
"none": true,
"required": true
} | | `supportsVoiceCloning` | *boolean* | :heavy_minus_sign: | Whether this TTS endpoint accepts inline reference audio (`input_references`) for stateless voice cloning. Requests carrying reference audio are only routed to endpoints where this is true. | | | `tag` | *string* | :heavy_check_mark: | N/A | | diff --git a/docs/models/speechinputreference.mdx b/docs/models/speechinputreference.mdx index d95674028..a03f5f340 100644 --- a/docs/models/speechinputreference.mdx +++ b/docs/models/speechinputreference.mdx @@ -2,18 +2,27 @@ title: "SpeechInputReference" --- -Reference content part for stateless voice cloning +Reference content part for stateless voice cloning or voice design ## Supported Types +### `models.SpeechInputReferenceImage` + +```typescript +const value: models.SpeechInputReferenceImage = { + imageUrl: { + url: "data:image/png;base64,iVBORw0KGgo...", + }, + type: "image_url", +}; +``` + ### `models.SpeechInputReferenceAudio` ```typescript const value: models.SpeechInputReferenceAudio = { - inputAudio: { - data: "data:audio/wav;base64,UklGRuQXDABXQVZF...", - }, + inputAudio: {}, type: "input_audio", }; ``` diff --git a/docs/models/speechinputreferenceaudio.mdx b/docs/models/speechinputreferenceaudio.mdx index 260053ef6..72f677f7b 100644 --- a/docs/models/speechinputreferenceaudio.mdx +++ b/docs/models/speechinputreferenceaudio.mdx @@ -2,7 +2,7 @@ title: "SpeechInputReferenceAudio" --- -Reference audio input for stateless voice cloning +Reference audio input for stateless voice cloning. Up to three parts per request; the Nth audio part is addressable from `input` as `@AudioN` on providers that support multiple references. ## Example Usage @@ -10,9 +10,7 @@ Reference audio input for stateless voice cloning import { SpeechInputReferenceAudio } from "@openrouter/sdk/models"; let value: SpeechInputReferenceAudio = { - inputAudio: { - data: "data:audio/wav;base64,UklGRuQXDABXQVZF...", - }, + inputAudio: {}, type: "input_audio", }; ``` diff --git a/docs/models/speechinputreferenceaudioinput.mdx b/docs/models/speechinputreferenceaudioinput.mdx index cb044aa5e..7dc5c2cc4 100644 --- a/docs/models/speechinputreferenceaudioinput.mdx +++ b/docs/models/speechinputreferenceaudioinput.mdx @@ -9,14 +9,13 @@ Reference audio input object ```typescript import { SpeechInputReferenceAudioInput } from "@openrouter/sdk/models"; -let value: SpeechInputReferenceAudioInput = { - data: "data:audio/wav;base64,UklGRuQXDABXQVZF...", -}; +let value: SpeechInputReferenceAudioInput = {}; ``` ## Fields -| Field | Type | Required | Description | Example | -| ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `data` | *string* | :heavy_check_mark: | Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio). | data:audio/wav;base64,UklGRuQXDABXQVZF... | -| `format` | *string* | :heavy_minus_sign: | Audio format of the reference audio (e.g., wav, mp3). Optional; most providers detect the format from the audio bytes. | wav | \ No newline at end of file +| Field | Type | Required | Description | Example | +| --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `data` | *string* | :heavy_minus_sign: | Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio). Exactly one of `data` or `url` is required. | data:audio/wav;base64,UklGRuQXDABXQVZF... | +| `format` | *string* | :heavy_minus_sign: | Audio format of the reference audio (e.g., wav, mp3). Optional; most providers detect the format from the audio bytes. | wav | +| `url` | *string* | :heavy_minus_sign: | Public http(s) URL of the reference audio. OpenRouter downloads it (15 MiB max) and forwards the bytes, never the URL. Exactly one of `data` or `url` is required. | https://example.com/reference.wav | \ No newline at end of file diff --git a/docs/models/speechinputreferenceimage.mdx b/docs/models/speechinputreferenceimage.mdx new file mode 100644 index 000000000..042996bad --- /dev/null +++ b/docs/models/speechinputreferenceimage.mdx @@ -0,0 +1,25 @@ +--- +title: "SpeechInputReferenceImage" +--- + +Reference image describing the desired voice. Cannot be combined with `input_audio` parts. Only routed to endpoints that support image references. + +## Example Usage + +```typescript +import { SpeechInputReferenceImage } from "@openrouter/sdk/models"; + +let value: SpeechInputReferenceImage = { + imageUrl: { + url: "data:image/png;base64,iVBORw0KGgo...", + }, + type: "image_url", +}; +``` + +## Fields + +| Field | Type | Required | Description | +| ------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------ | +| `imageUrl` | [models.SpeechInputReferenceImageInput](../models/speechinputreferenceimageinput.mdx) | :heavy_check_mark: | Reference image input object | +| `type` | *"image_url"* | :heavy_check_mark: | N/A | \ No newline at end of file diff --git a/docs/models/speechinputreferenceimageinput.mdx b/docs/models/speechinputreferenceimageinput.mdx new file mode 100644 index 000000000..871636012 --- /dev/null +++ b/docs/models/speechinputreferenceimageinput.mdx @@ -0,0 +1,21 @@ +--- +title: "SpeechInputReferenceImageInput" +--- + +Reference image input object + +## Example Usage + +```typescript +import { SpeechInputReferenceImageInput } from "@openrouter/sdk/models"; + +let value: SpeechInputReferenceImageInput = { + url: "data:image/png;base64,iVBORw0KGgo...", +}; +``` + +## Fields + +| Field | Type | Required | Description | Example | +| -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `url` | *string* | :heavy_check_mark: | JPEG, PNG, or WebP reference image as a base64 data URI or a public http(s) URL. Remote images are downloaded (15 MiB max) and forwarded as bytes, never as the URL. | data:image/png;base64,iVBORw0KGgo... | \ No newline at end of file diff --git a/docs/models/speechinputreferencetext.mdx b/docs/models/speechinputreferencetext.mdx index d282057cc..95f3c95b5 100644 --- a/docs/models/speechinputreferencetext.mdx +++ b/docs/models/speechinputreferencetext.mdx @@ -2,7 +2,7 @@ title: "SpeechInputReferenceText" --- -Transcript of the accompanying reference audio +Transcript of an `input_audio` part ## Example Usage @@ -17,7 +17,7 @@ let value: SpeechInputReferenceText = { ## Fields -| Field | Type | Required | Description | Example | -| ----------------------------------------------- | ----------------------------------------------- | ----------------------------------------------- | ----------------------------------------------- | ----------------------------------------------- | -| `text` | *string* | :heavy_check_mark: | Transcript of the accompanying reference audio. | I used to rule the world. | -| `type` | *"text"* | :heavy_check_mark: | N/A | | \ No newline at end of file +| Field | Type | Required | Description | Example | +| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `text` | *string* | :heavy_check_mark: | Transcript of an `input_audio` part. With a single clip it may appear before or after the clip; with multiple clips it must immediately follow the clip it transcribes. | I used to rule the world. | +| `type` | *"text"* | :heavy_check_mark: | N/A | | \ No newline at end of file diff --git a/docs/models/speechrequest.mdx b/docs/models/speechrequest.mdx index 4acf9e895..e5647ce3a 100644 --- a/docs/models/speechrequest.mdx +++ b/docs/models/speechrequest.mdx @@ -17,15 +17,15 @@ let value: SpeechRequest = { ## Fields -| Field | Type | Required | Description | Example | -| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `input` | *string* | :heavy_check_mark: | Text to synthesize | Hello world | -| `inputReferences` | *models.SpeechInputReference*[] | :heavy_minus_sign: | Reference content for stateless voice cloning: one `input_audio` part carrying the voice sample, optionally accompanied by one `text` part with its transcript. Only routed to endpoints that support voice cloning. | [
\{
"input_audio": \{
"data": "data:audio/wav;base64,UklGRuQXDABXQVZF..."
},
"type": "input_audio"
},
\{
"text": "I used to rule the world.",
"type": "text"
}
] | -| `model` | *string* | :heavy_check_mark: | TTS model identifier | mistralai/voxtral-mini-tts-2603 | -| `provider` | [models.SpeechRequestProvider](../models/speechrequestprovider.mdx) | :heavy_minus_sign: | Provider-specific passthrough configuration | | -| `responseFormat` | [models.SpeechRequestResponseFormat](../models/speechrequestresponseformat.mdx) | :heavy_minus_sign: | Audio output format | pcm | -| `sessionId` | *string* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | session-1234 | -| `speed` | *number* | :heavy_minus_sign: | Playback speed multiplier. Only used by models that support it (e.g. OpenAI TTS). Ignored by other providers. | 1 | -| `trace` | [models.TraceConfig](../models/traceconfig.mdx) | :heavy_minus_sign: | Metadata for observability and tracing. Known keys (trace_id, trace_name, span_name, generation_name, parent_span_id) have special handling. Additional keys are passed through as custom metadata to configured broadcast destinations. | \{
"trace_id": "trace-abc123",
"trace_name": "my-app-trace"
} | -| `user` | *string* | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | user-1234 | -| `voice` | *string* | :heavy_minus_sign: | Voice identifier (provider-specific). | en_paul_neutral | \ No newline at end of file +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `input` | *string* | :heavy_check_mark: | Text to synthesize | Hello world | +| `inputReferences` | *models.SpeechInputReference*[] | :heavy_minus_sign: | Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference. | [
\{
"input_audio": \{
"data": "data:audio/wav;base64,UklGRuQXDABXQVZF..."
},
"type": "input_audio"
},
\{
"text": "I used to rule the world.",
"type": "text"
}
] | +| `model` | *string* | :heavy_check_mark: | TTS model identifier | mistralai/voxtral-mini-tts-2603 | +| `provider` | [models.SpeechRequestProvider](../models/speechrequestprovider.mdx) | :heavy_minus_sign: | Provider-specific passthrough configuration | | +| `responseFormat` | [models.SpeechRequestResponseFormat](../models/speechrequestresponseformat.mdx) | :heavy_minus_sign: | Audio output format | pcm | +| `sessionId` | *string* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | session-1234 | +| `speed` | *number* | :heavy_minus_sign: | Playback speed multiplier. Only used by models that support it (e.g. OpenAI TTS). Ignored by other providers. | 1 | +| `trace` | [models.TraceConfig](../models/traceconfig.mdx) | :heavy_minus_sign: | Metadata for observability and tracing. Known keys (trace_id, trace_name, span_name, generation_name, parent_span_id) have special handling. Additional keys are passed through as custom metadata to configured broadcast destinations. | \{
"trace_id": "trace-abc123",
"trace_name": "my-app-trace"
} | +| `user` | *string* | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | user-1234 | +| `voice` | *string* | :heavy_minus_sign: | Voice identifier (provider-specific). | en_paul_neutral | \ No newline at end of file diff --git a/jsr.json b/jsr.json index b7cb50b39..a3abc6d47 100644 --- a/jsr.json +++ b/jsr.json @@ -2,7 +2,7 @@ { "name": "@openrouter/sdk", - "version": "1.3.28", + "version": "1.3.29", "exports": { ".": "./src/index.ts", "./models/errors": "./src/models/errors/index.ts", diff --git a/package.json b/package.json index bb04a9b02..565aa641c 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@openrouter/sdk", - "version": "1.3.28", + "version": "1.3.29", "author": "OpenRouter", "description": "The OpenRouter TypeScript SDK is a type-safe toolkit for building AI applications with access to 400+ language models through a unified API.", "keywords": [ @@ -21,8 +21,8 @@ "license": "Apache-2.0", "packageManager": "pnpm@10.22.0", "publishConfig": { - "provenance": true, - "access": "public" + "access": "public", + "provenance": true }, "type": "module", "main": "./esm/index.js", @@ -74,14 +74,14 @@ "build": "tsc && node scripts/unbarrel-imports.js", "prepublishOnly": "npm run build", "prepare": "npm run build", - "test": "vitest --run --project unit", "test:e2e": "vitest --run --project e2e", - "typecheck:transit": "exit 0", - "compile": "tsc && node scripts/unbarrel-imports.js", - "postinstall": "node scripts/check-types.js || true", "test:transit": "exit 0", "test:watch": "vitest --watch --project unit", - "typecheck": "tsc --noEmit" + "typecheck:transit": "exit 0", + "test": "vitest --run --project unit", + "typecheck": "tsc --noEmit", + "compile": "tsc && node scripts/unbarrel-imports.js", + "postinstall": "node scripts/check-types.js || true" }, "peerDependencies": { diff --git a/src/lib/config.ts b/src/lib/config.ts index cdcf32967..0e3c82c7c 100644 --- a/src/lib/config.ts +++ b/src/lib/config.ts @@ -79,7 +79,7 @@ export function serverURLFromOptions(options: SDKOptions): URL | null { export const SDK_METADATA = { language: "typescript", openapiDocVersion: "1.0.0", - sdkVersion: "1.3.28", + sdkVersion: "1.3.29", genVersion: "2.914.0", - userAgent: "speakeasy-sdk/typescript 1.3.28 2.914.0 1.0.0 @openrouter/sdk", + userAgent: "speakeasy-sdk/typescript 1.3.29 2.914.0 1.0.0 @openrouter/sdk", } as const; diff --git a/src/models/index.ts b/src/models/index.ts index 72d92b2c5..930bfcc83 100644 --- a/src/models/index.ts +++ b/src/models/index.ts @@ -588,6 +588,8 @@ export * from "./shellservertoolopenrouter.js"; export * from "./speechinputreference.js"; export * from "./speechinputreferenceaudio.js"; export * from "./speechinputreferenceaudioinput.js"; +export * from "./speechinputreferenceimage.js"; +export * from "./speechinputreferenceimageinput.js"; export * from "./speechinputreferencetext.js"; export * from "./speechrequest.js"; export * from "./stopservertoolswhencondition.js"; diff --git a/src/models/publicendpoint.ts b/src/models/publicendpoint.ts index 7eb0e3b10..b487e4ff1 100644 --- a/src/models/publicendpoint.ts +++ b/src/models/publicendpoint.ts @@ -220,7 +220,15 @@ export type PublicEndpoint = { quantization: Quantization | null; status?: EndpointStatus | undefined; supportedParameters: Array; + /** + * Whether this TTS endpoint accepts an `image_url` reference describing the desired voice. Requests carrying an image reference are only routed to endpoints where this is true. + */ + supportsImageReference: boolean; supportsImplicitCaching: boolean; + /** + * Whether this TTS endpoint accepts more than one `input_audio` reference clip per request. Requests carrying several clips are only routed to endpoints where this is true. + */ + supportsMultipleAudioReferences: boolean; /** * Per-variant `tool_choice` support. `tool_choice` in `supported_parameters` only says the parameter is accepted; these flags say which of its values passed testing. */ @@ -530,7 +538,9 @@ export const PublicEndpoint$inboundSchema: z.ZodType = quantization: z.nullable(Quantization$inboundSchema), status: EndpointStatus$inboundSchema.optional(), supported_parameters: z.array(Parameter$inboundSchema), + supports_image_reference: z.boolean().default(false), supports_implicit_caching: z.boolean(), + supports_multiple_audio_references: z.boolean().default(false), supports_tool_choice: ToolChoiceSupport$inboundSchema, supports_voice_cloning: z.boolean().default(false), tag: z.string(), @@ -549,7 +559,9 @@ export const PublicEndpoint$inboundSchema: z.ZodType = "perf_last_30m_by_workload": "perfLast30mByWorkload", "provider_name": "providerName", "supported_parameters": "supportedParameters", + "supports_image_reference": "supportsImageReference", "supports_implicit_caching": "supportsImplicitCaching", + "supports_multiple_audio_references": "supportsMultipleAudioReferences", "supports_tool_choice": "supportsToolChoice", "supports_voice_cloning": "supportsVoiceCloning", "throughput_last_30m": "throughputLast30m", diff --git a/src/models/speechinputreference.ts b/src/models/speechinputreference.ts index 25f478821..cf8f0de91 100644 --- a/src/models/speechinputreference.ts +++ b/src/models/speechinputreference.ts @@ -9,6 +9,11 @@ import { SpeechInputReferenceAudio$Outbound, SpeechInputReferenceAudio$outboundSchema, } from "./speechinputreferenceaudio.js"; +import { + SpeechInputReferenceImage, + SpeechInputReferenceImage$Outbound, + SpeechInputReferenceImage$outboundSchema, +} from "./speechinputreferenceimage.js"; import { SpeechInputReferenceText, SpeechInputReferenceText$Outbound, @@ -16,14 +21,16 @@ import { } from "./speechinputreferencetext.js"; /** - * Reference content part for stateless voice cloning + * Reference content part for stateless voice cloning or voice design */ export type SpeechInputReference = + | SpeechInputReferenceImage | SpeechInputReferenceAudio | SpeechInputReferenceText; /** @internal */ export type SpeechInputReference$Outbound = + | SpeechInputReferenceImage$Outbound | SpeechInputReferenceAudio$Outbound | SpeechInputReferenceText$Outbound; @@ -32,6 +39,7 @@ export const SpeechInputReference$outboundSchema: z.ZodType< SpeechInputReference$Outbound, SpeechInputReference > = z.union([ + SpeechInputReferenceImage$outboundSchema, SpeechInputReferenceAudio$outboundSchema, SpeechInputReferenceText$outboundSchema, ]); diff --git a/src/models/speechinputreferenceaudio.ts b/src/models/speechinputreferenceaudio.ts index 798b33dab..d22c73add 100644 --- a/src/models/speechinputreferenceaudio.ts +++ b/src/models/speechinputreferenceaudio.ts @@ -12,7 +12,7 @@ import { } from "./speechinputreferenceaudioinput.js"; /** - * Reference audio input for stateless voice cloning + * Reference audio input for stateless voice cloning. Up to three parts per request; the Nth audio part is addressable from `input` as `@AudioN` on providers that support multiple references. */ export type SpeechInputReferenceAudio = { /** diff --git a/src/models/speechinputreferenceaudioinput.ts b/src/models/speechinputreferenceaudioinput.ts index 04ff60eb7..0efae28e8 100644 --- a/src/models/speechinputreferenceaudioinput.ts +++ b/src/models/speechinputreferenceaudioinput.ts @@ -10,19 +10,24 @@ import * as z from "zod/v4"; */ export type SpeechInputReferenceAudioInput = { /** - * Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio). + * Base64-encoded reference audio (optionally a data URI). Supported audio formats are provider-specific. Limited to 20 MiB of base64 (15 MiB of decoded audio). Exactly one of `data` or `url` is required. */ - data: string; + data?: string | undefined; /** * Audio format of the reference audio (e.g., wav, mp3). Optional; most providers detect the format from the audio bytes. */ format?: string | undefined; + /** + * Public http(s) URL of the reference audio. OpenRouter downloads it (15 MiB max) and forwards the bytes, never the URL. Exactly one of `data` or `url` is required. + */ + url?: string | undefined; }; /** @internal */ export type SpeechInputReferenceAudioInput$Outbound = { - data: string; + data?: string | undefined; format?: string | undefined; + url?: string | undefined; }; /** @internal */ @@ -30,8 +35,9 @@ export const SpeechInputReferenceAudioInput$outboundSchema: z.ZodType< SpeechInputReferenceAudioInput$Outbound, SpeechInputReferenceAudioInput > = z.object({ - data: z.string(), + data: z.string().optional(), format: z.string().optional(), + url: z.string().optional(), }); export function speechInputReferenceAudioInputToJSON( diff --git a/src/models/speechinputreferenceimage.ts b/src/models/speechinputreferenceimage.ts new file mode 100644 index 000000000..60fbd6a24 --- /dev/null +++ b/src/models/speechinputreferenceimage.ts @@ -0,0 +1,50 @@ +/* + * Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT. + * @generated-id: ee7097a5c9df + */ + +import * as z from "zod/v4"; +import { remap as remap$ } from "../lib/primitives.js"; +import { + SpeechInputReferenceImageInput, + SpeechInputReferenceImageInput$Outbound, + SpeechInputReferenceImageInput$outboundSchema, +} from "./speechinputreferenceimageinput.js"; + +/** + * Reference image describing the desired voice. Cannot be combined with `input_audio` parts. Only routed to endpoints that support image references. + */ +export type SpeechInputReferenceImage = { + /** + * Reference image input object + */ + imageUrl: SpeechInputReferenceImageInput; + type: "image_url"; +}; + +/** @internal */ +export type SpeechInputReferenceImage$Outbound = { + image_url: SpeechInputReferenceImageInput$Outbound; + type: "image_url"; +}; + +/** @internal */ +export const SpeechInputReferenceImage$outboundSchema: z.ZodType< + SpeechInputReferenceImage$Outbound, + SpeechInputReferenceImage +> = z.object({ + imageUrl: SpeechInputReferenceImageInput$outboundSchema, + type: z.literal("image_url"), +}).transform((v) => { + return remap$(v, { + imageUrl: "image_url", + }); +}); + +export function speechInputReferenceImageToJSON( + speechInputReferenceImage: SpeechInputReferenceImage, +): string { + return JSON.stringify( + SpeechInputReferenceImage$outboundSchema.parse(speechInputReferenceImage), + ); +} diff --git a/src/models/speechinputreferenceimageinput.ts b/src/models/speechinputreferenceimageinput.ts new file mode 100644 index 000000000..db60a7d57 --- /dev/null +++ b/src/models/speechinputreferenceimageinput.ts @@ -0,0 +1,39 @@ +/* + * Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT. + * @generated-id: 36100583e975 + */ + +import * as z from "zod/v4"; + +/** + * Reference image input object + */ +export type SpeechInputReferenceImageInput = { + /** + * JPEG, PNG, or WebP reference image as a base64 data URI or a public http(s) URL. Remote images are downloaded (15 MiB max) and forwarded as bytes, never as the URL. + */ + url: string; +}; + +/** @internal */ +export type SpeechInputReferenceImageInput$Outbound = { + url: string; +}; + +/** @internal */ +export const SpeechInputReferenceImageInput$outboundSchema: z.ZodType< + SpeechInputReferenceImageInput$Outbound, + SpeechInputReferenceImageInput +> = z.object({ + url: z.string(), +}); + +export function speechInputReferenceImageInputToJSON( + speechInputReferenceImageInput: SpeechInputReferenceImageInput, +): string { + return JSON.stringify( + SpeechInputReferenceImageInput$outboundSchema.parse( + speechInputReferenceImageInput, + ), + ); +} diff --git a/src/models/speechinputreferencetext.ts b/src/models/speechinputreferencetext.ts index dd8e51886..9736271f1 100644 --- a/src/models/speechinputreferencetext.ts +++ b/src/models/speechinputreferencetext.ts @@ -6,11 +6,11 @@ import * as z from "zod/v4"; /** - * Transcript of the accompanying reference audio + * Transcript of an `input_audio` part */ export type SpeechInputReferenceText = { /** - * Transcript of the accompanying reference audio. + * Transcript of an `input_audio` part. With a single clip it may appear before or after the clip; with multiple clips it must immediately follow the clip it transcribes. */ text: string; type: "text"; diff --git a/src/models/speechrequest.ts b/src/models/speechrequest.ts index 8d8512c0e..c70a2be33 100644 --- a/src/models/speechrequest.ts +++ b/src/models/speechrequest.ts @@ -56,7 +56,7 @@ export type SpeechRequest = { */ input: string; /** - * Reference content for stateless voice cloning: one `input_audio` part carrying the voice sample, optionally accompanied by one `text` part with its transcript. Only routed to endpoints that support voice cloning. + * Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference. */ inputReferences?: Array | undefined; /**