diff --git a/.speakeasy/gen.lock b/.speakeasy/gen.lock
index 1a68ec12..8f88644f 100644
--- a/.speakeasy/gen.lock
+++ b/.speakeasy/gen.lock
@@ -1,19 +1,19 @@
lockVersion: 2.0.0
id: c48cf606-fb42-4a45-9c23-8f0555307828
management:
- docChecksum: ca478c9fe977377c5080ec03921023a8
+ docChecksum: d1b96abffbe70fe9d18a0e68456bf0e3
docVersion: 1.0.0
speakeasyVersion: 1.787.0
generationVersion: 2.914.0
- releaseVersion: 1.3.26
- configChecksum: f1f6fdb944dd84de88b781a775f1b193
+ releaseVersion: 1.3.27
+ configChecksum: 8e1aa111f5eedd150d7163d092dbc596
repoURL: https://github.com/OpenRouterTeam/python-sdk.git
installationURL: https://github.com/OpenRouterTeam/python-sdk.git
published: true
persistentEdits:
- generation_id: 7e99d85c-0497-46d2-8f32-30b6816934f7
- pristine_commit_hash: 6559d85090ca32fc4de97e741687929da41137e2
- pristine_tree_hash: 3d570f3d755633ea2ee0eb370a1ed7571858d615
+ generation_id: aea6cfba-febf-4944-8c59-559d33ca5796
+ pristine_commit_hash: ce959dc03b41e9baed4c595c1d79f62117ec79cd
+ pristine_tree_hash: 8f75aaeffc4d498b2b0b09bd5f19af073549fd86
features:
python:
acceptHeaders: 3.0.0
@@ -6778,6 +6778,10 @@ trackedFiles:
id: 77262fe83449
last_write_checksum: sha1:9f290862f832b501a96456226ff1f18431f74c50
pristine_git_object: 5d617cac37f7b364f4dd0640246e97c34351d824
+ docs/components/speechinput.mdx:
+ id: 663248d32888
+ last_write_checksum: sha1:a8126cfb9764e1cd97fd896dedc4f40f7ab441bb
+ pristine_git_object: 948ee8455aede1f95245b7bfe33d04c293576edd
docs/components/speechinputreference.mdx:
id: eb5af16936c7
last_write_checksum: sha1:ab545b03da2161ae1ad449714b6861cd55ae66df
@@ -6816,8 +6820,8 @@ trackedFiles:
pristine_git_object: 34f4f07fb3008df83646fa3d88ce04009274246a
docs/components/speechrequest.mdx:
id: 06e81b0433f6
- last_write_checksum: sha1:b8cafb4ce14ea3983bd792eeaf95be0dbf4c8c3e
- pristine_git_object: 1ffadd62bda287f96c35e4b83a7408f3c0fc117b
+ last_write_checksum: sha1:ecac6988ab27bf129f774f3982e36744ec613218
+ pristine_git_object: c538a0061028cbaae17210def2258da7316b07e7
docs/components/speechrequestdatacollection.mdx:
id: 6c5938568888
last_write_checksum: sha1:89a08b4c4de66c1ef7676792a38747e030ed6729
@@ -6830,6 +6834,10 @@ trackedFiles:
id: bd09f77a4a90
last_write_checksum: sha1:1b87242f33c6d22a08dd085d0ef41335ec2e9b0c
pristine_git_object: 9af8a5aa6a7c26f353f552d369b87debb045b7cd
+ docs/components/speechturn.mdx:
+ id: ca90a4443e6e
+ last_write_checksum: sha1:4a63a0b46eb3051c86e4f5ed20b8d4534d5e7720
+ pristine_git_object: db37aa4e7bdc88e9216fad12c5b8e197e4d87e42
docs/components/speed.mdx:
id: 4fd1f2823924
last_write_checksum: sha1:c6815d115b55b9409052f9d1d1d16bb1d7197763
@@ -9948,8 +9956,8 @@ trackedFiles:
pristine_git_object: 1dd30f0b898f2a842d1bb585b03d887be90e7845
docs/sdks/tts/README.mdx:
id: cd1132543884
- last_write_checksum: sha1:c79acf8b375dfd535a3f5948da5f65446397c49a
- pristine_git_object: 2a3eda6822076c9de566cbe56bbe5cfdc691cf9d
+ last_write_checksum: sha1:cb9303bf791d8d9784be2815fae45f4b09f82221
+ pristine_git_object: b96e9d3ee1f40340dcd002a717920f7576b9fe3f
docs/sdks/vault/README.mdx:
id: 3738c6722acd
last_write_checksum: sha1:12acdfed4c2a32488d98b6091fd8c392af516494
@@ -9968,8 +9976,8 @@ trackedFiles:
pristine_git_object: 3e38f1a929f7d6b1d6de74604aa87e3d8f010544
pyproject.toml:
id: 5d07e7d72637
- last_write_checksum: sha1:178fb5ebf2d5e2e9357df50216ef837c9c45174b
- pristine_git_object: 112fa9a986c4f28337593041fbe9378906ee2670
+ last_write_checksum: sha1:984fcd4d70ebc92db6939d878585d9fe80107f49
+ pristine_git_object: 87f61eaba85b92fd2eea6d10bbc228f35dbb386c
scripts/prepare_readme.py:
id: e0c5957a6035
last_write_checksum: sha1:77f44b60b98bc126557ec27391f91dfba764bb54
@@ -9996,8 +10004,8 @@ trackedFiles:
pristine_git_object: 86713cfea633e09d33b3d4e65281071fe20e6137
src/openrouter/_version.py:
id: d8d15ad6c586
- last_write_checksum: sha1:6e671a7634a6396b6de4bed0c69bd623b0ebbd90
- pristine_git_object: 235db3aed8d14857805c9f976a5bc57a154113b3
+ last_write_checksum: sha1:8a2efb73d913c481db11b1295cf358c8a81e7e71
+ pristine_git_object: 550534f0e7ae6c001bba40e7a172e4735106f628
src/openrouter/alpha.py:
id: 306c4d93308d
last_write_checksum: sha1:30f55a360f41376ab194b9ea725fe4e001a5ae1a
@@ -10044,8 +10052,8 @@ trackedFiles:
pristine_git_object: ad3d247954547814054c01989a2dff3d12b3e4e1
src/openrouter/components/__init__.py:
id: 81754e97b3f4
- last_write_checksum: sha1:b4332657c5562071bfc8753410542b6376d6ae61
- pristine_git_object: 2e45778d420de470fb82a0280811b373f74a8274
+ last_write_checksum: sha1:5929f0aad1def97fb47d220701cd8ad626c32cdf
+ pristine_git_object: 92a29ef91468729ce876f26d46881cdbbf845c0d
src/openrouter/components/aabenchmarkentry.py:
id: e2e0f0b48c82
last_write_checksum: sha1:fab4d9a24d2cea937bb749d46c5f83941e99d65c
@@ -12902,6 +12910,10 @@ trackedFiles:
id: 6d6e8d7d80ad
last_write_checksum: sha1:d065afdd505a8303f6ba8e268cef42a3e9ea39fb
pristine_git_object: 418953ba7366a6015934bf8795178d79adb03b89
+ src/openrouter/components/speechinput.py:
+ id: 14e46bea4b32
+ last_write_checksum: sha1:4f2acbfbb0b8df5d8d2d3916f9a081a6143fdf45
+ pristine_git_object: 676f38928a4745ddb215cb8275c401e6fb5ebc25
src/openrouter/components/speechinputreference.py:
id: 8e2b63f83354
last_write_checksum: sha1:b2ebc68fae64644f8bfca9b555ca7d2a7b46c9e1
@@ -12928,8 +12940,12 @@ trackedFiles:
pristine_git_object: 4633e5d7658c500aa903ce8caa8ef86cc97b78eb
src/openrouter/components/speechrequest.py:
id: 2a9400167112
- last_write_checksum: sha1:282f155b9d0b09e015d3dda0a73abe8f42e44dbc
- pristine_git_object: 5a62dc470dde417e458ab488dd92c385dcc59459
+ last_write_checksum: sha1:49cbabd40418cc50b15e7eef19926ce55be23a4e
+ pristine_git_object: ad116652ee8ead569fb0939c29c6966bb397717e
+ src/openrouter/components/speechturn.py:
+ id: 121e0a940f91
+ last_write_checksum: sha1:31bdba00e027b81de8a3872d48e694475cf02e7a
+ pristine_git_object: d3e0e4be8c2b3fcc3427657d9c6eb162031b118a
src/openrouter/components/stopservertoolswhencondition.py:
id: 2deeda4209ac
last_write_checksum: sha1:581e0ee62776d42bf598b9f68b998faabfecd3ce
@@ -14244,8 +14260,8 @@ trackedFiles:
pristine_git_object: 701fa8597ff85f41a60736ade55fcb32f9ec7125
src/openrouter/tts.py:
id: 5055d4b95f1d
- last_write_checksum: sha1:e771b211ff92fd56d79f2c6ce05fa43211cfdbd8
- pristine_git_object: 6da4bd907cfe243374782b9523bccfad7071976d
+ last_write_checksum: sha1:9826552c18767897c4dc6c3cd8c2ce41da40a344
+ pristine_git_object: 5b1bf2a61bfd3d476284465192603aa7d90adfd7
src/openrouter/types/__init__.py:
id: 5eab536205b7
last_write_checksum: sha1:f9ad14217f832e74f594285960125add50324be9
@@ -18266,7 +18282,4 @@ examples:
examplesVersion: 1.0.2
releaseNotes: |
## Python SDK Changes:
- * `open_router.api_keys.list()`: `response.data[].last_used_at` **Added**
- * `open_router.api_keys.create()`: `response.data.last_used_at` **Added**
- * `open_router.api_keys.get()`: `response.data.last_used_at` **Added**
- * `open_router.api_keys.update()`: `response.data.last_used_at` **Added**
+ * `open_router.tts.create_speech()`: `request` **Changed** (Breaking ⚠️)
diff --git a/.speakeasy/gen.yaml b/.speakeasy/gen.yaml
index dd19de0f..f7b70146 100644
--- a/.speakeasy/gen.yaml
+++ b/.speakeasy/gen.yaml
@@ -36,7 +36,7 @@ generation:
documentation: mintlify
preApplyUnionDiscriminators: true
python:
- version: 1.3.26
+ version: 1.3.27
additionalDependencies:
dev: {}
main: {}
diff --git a/.speakeasy/out.openapi.yaml b/.speakeasy/out.openapi.yaml
index adc005eb..c819c46f 100644
--- a/.speakeasy/out.openapi.yaml
+++ b/.speakeasy/out.openapi.yaml
@@ -28813,6 +28813,15 @@ components:
oneOf:
- $ref: '#/components/schemas/ContainerAutoEnvironment'
- $ref: '#/components/schemas/ContainerReferenceEnvironment'
+ SpeechInput:
+ anyOf:
+ - type: 'string'
+ - items:
+ $ref: '#/components/schemas/SpeechTurn'
+ minItems: 1
+ type: 'array'
+ description: 'Text to synthesize, or a list of turns for multi-speaker input. Each turn has its own text, voice, and instructions. Multi-speaker input is currently supported by Gemini TTS models only.'
+ example: 'Hello world'
SpeechInputReference:
description: 'Reference content part for stateless voice cloning or voice design'
discriminator:
@@ -28920,9 +28929,7 @@ components:
voice: 'en_paul_neutral'
properties:
input:
- description: 'Text to synthesize'
- example: 'Hello world'
- type: 'string'
+ $ref: '#/components/schemas/SpeechInput'
input_references:
description: 'Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference.'
example:
@@ -28934,6 +28941,10 @@ components:
items:
$ref: '#/components/schemas/SpeechInputReference'
type: 'array'
+ instructions:
+ description: 'Delivery instructions for the whole request, such as tone, pacing, or emotion. Supported by OpenAI gpt-4o-mini-tts and Gemini TTS models. Ignored by other providers.'
+ example: 'Speak in a warm and friendly tone.'
+ type: 'string'
model:
description: 'TTS model identifier'
example: 'mistralai/voxtral-mini-tts-2603'
@@ -28997,6 +29008,24 @@ components:
- 'model'
- 'input'
type: 'object'
+ SpeechTurn:
+ properties:
+ instructions:
+ description: 'Delivery instructions for this turn, such as tone, pacing, or emotion. Overrides the top-level `instructions`.'
+ example: 'whispering'
+ type: 'string'
+ text:
+ description: 'Text to synthesize'
+ example: 'Hi Jane.'
+ type: 'string'
+ voice:
+ description: 'Voice for this turn. Defaults to the top-level `voice`.'
+ example: 'Kore'
+ minLength: 1
+ type: 'string'
+ required:
+ - 'text'
+ type: 'object'
StopServerToolsWhen:
description: 'Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.'
example:
diff --git a/.speakeasy/workflow.lock b/.speakeasy/workflow.lock
index 91a78bd4..ee10e56d 100644
--- a/.speakeasy/workflow.lock
+++ b/.speakeasy/workflow.lock
@@ -2,8 +2,8 @@ speakeasyVersion: 1.787.0
sources:
OpenRouter API:
sourceNamespace: open-router-chat-completions-api
- sourceRevisionDigest: sha256:d24d16338614e6343ad2fa1f47f408d2f8c3eba071a3d8a472544748a0aa6620
- sourceBlobDigest: sha256:da7dea9cc7d4012055e43f82b5e8dbf0324eda4990d380fc4968ae2e93bc5acd
+ sourceRevisionDigest: sha256:aa4988cf1340b9b4c45d0f7d34c1d80884016b955e99381c9aea52f38e355aa8
+ sourceBlobDigest: sha256:4828d7d1d7a9927d3268b7bede7018480aa9c23f2a8231ae6bac8eb6dc9bc811
tags:
- latest
- 1.0.0
@@ -11,10 +11,10 @@ targets:
open-router:
source: OpenRouter API
sourceNamespace: open-router-chat-completions-api
- sourceRevisionDigest: sha256:d24d16338614e6343ad2fa1f47f408d2f8c3eba071a3d8a472544748a0aa6620
- sourceBlobDigest: sha256:da7dea9cc7d4012055e43f82b5e8dbf0324eda4990d380fc4968ae2e93bc5acd
+ sourceRevisionDigest: sha256:aa4988cf1340b9b4c45d0f7d34c1d80884016b955e99381c9aea52f38e355aa8
+ sourceBlobDigest: sha256:4828d7d1d7a9927d3268b7bede7018480aa9c23f2a8231ae6bac8eb6dc9bc811
codeSamplesNamespace: open-router-python-code-samples
- codeSamplesRevisionDigest: sha256:3f73f08c00f988c928fcea61faba174a810f86938b84a2f3db39fb026966729c
+ codeSamplesRevisionDigest: sha256:0aafecfc542610744e021605d5923576663463046651088762e8cf5b1ab55d95
workflow:
workflowVersion: 1.0.0
speakeasyVersion: 1.787.0
diff --git a/RELEASES.md b/RELEASES.md
index 00111650..2bab81e4 100644
--- a/RELEASES.md
+++ b/RELEASES.md
@@ -3029,4 +3029,14 @@ Based on:
### Generated
- [python v1.3.26] .
### Releases
-- [PyPI v1.3.26] https://pypi.org/project/openrouter/1.3.26 - .
\ No newline at end of file
+- [PyPI v1.3.26] https://pypi.org/project/openrouter/1.3.26 - .
+
+## 2026-10-06 21:16:10
+### Changes
+Based on:
+- OpenAPI Doc
+- Speakeasy CLI 1.787.0 (2.914.0) https://github.com/speakeasy-api/speakeasy
+### Generated
+- [python v1.3.27] .
+### Releases
+- [PyPI v1.3.27] https://pypi.org/project/openrouter/1.3.27 - .
\ No newline at end of file
diff --git a/docs/components/speechinput.mdx b/docs/components/speechinput.mdx
new file mode 100644
index 00000000..948ee845
--- /dev/null
+++ b/docs/components/speechinput.mdx
@@ -0,0 +1,21 @@
+---
+title: "SpeechInput"
+---
+
+Text to synthesize, or a list of turns for multi-speaker input. Each turn has its own text, voice, and instructions. Multi-speaker input is currently supported by Gemini TTS models only.
+
+
+## Supported Types
+
+### `str`
+
+```python
+value: str = /* values here */
+```
+
+### `List[components.SpeechTurn]`
+
+```python
+value: List[components.SpeechTurn] = /* values here */
+```
+
diff --git a/docs/components/speechrequest.mdx b/docs/components/speechrequest.mdx
index 982d2af1..67b7322c 100644
--- a/docs/components/speechrequest.mdx
+++ b/docs/components/speechrequest.mdx
@@ -9,8 +9,9 @@ Text-to-speech request input
| Field | Type | Required | Description | Example |
| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| `input` | *str* | ✅ | Text to synthesize | Hello world |
+| `input` | [components.SpeechInput](../components/speechinput.mdx) | ✅ | Text to synthesize, or a list of turns for multi-speaker input. Each turn has its own text, voice, and instructions. Multi-speaker input is currently supported by Gemini TTS models only. | Hello world |
| `input_references` | List[[components.SpeechInputReference](../components/speechinputreference.mdx)] | ➖ | Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference. | [
\{
"input_audio": \{
"data": "data:audio/wav;base64,UklGRuQXDABXQVZF..."
},
"type": "input_audio"
},
\{
"text": "I used to rule the world.",
"type": "text"
}
] |
+| `instructions` | *Optional[str]* | ➖ | Delivery instructions for the whole request, such as tone, pacing, or emotion. Supported by OpenAI gpt-4o-mini-tts and Gemini TTS models. Ignored by other providers. | Speak in a warm and friendly tone. |
| `model` | *str* | ✅ | TTS model identifier | mistralai/voxtral-mini-tts-2603 |
| `provider` | [Optional[components.SpeechRequestProvider]](../components/speechrequestprovider.mdx) | ➖ | Provider configuration: data policy routing preferences (`zdr`, `data_collection`) and provider-specific passthrough options | |
| `response_format` | [Optional[components.SpeechRequestResponseFormat]](../components/speechrequestresponseformat.mdx) | ➖ | Audio output format | pcm |
diff --git a/docs/components/speechturn.mdx b/docs/components/speechturn.mdx
new file mode 100644
index 00000000..2297b61a
--- /dev/null
+++ b/docs/components/speechturn.mdx
@@ -0,0 +1,11 @@
+---
+title: "SpeechTurn"
+---
+
+## Fields
+
+| Field | Type | Required | Description | Example |
+| -------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------- |
+| `instructions` | *Optional[str]* | ➖ | Delivery instructions for this turn, such as tone, pacing, or emotion. Overrides the top-level `instructions`. | whispering |
+| `text` | *str* | ✅ | Text to synthesize | Hi Jane. |
+| `voice` | *Optional[str]* | ➖ | Voice for this turn. Defaults to the top-level `voice`. | Kore |
\ No newline at end of file
diff --git a/docs/sdks/tts/README.mdx b/docs/sdks/tts/README.mdx
index 7c05bbc5..f459eacf 100644
--- a/docs/sdks/tts/README.mdx
+++ b/docs/sdks/tts/README.mdx
@@ -40,12 +40,13 @@ with OpenRouter(
| Parameter | Type | Required | Description | Example |
| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
-| `input` | *str* | ✅ | Text to synthesize | Hello world |
+| `input` | [components.SpeechInput](../../components/speechinput.mdx) | ✅ | Text to synthesize, or a list of turns for multi-speaker input. Each turn has its own text, voice, and instructions. Multi-speaker input is currently supported by Gemini TTS models only. | Hello world |
| `model` | *str* | ✅ | TTS model identifier | mistralai/voxtral-mini-tts-2603 |
| `http_referer` | *Optional[str]* | ➖ | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| |
| `x_open_router_title` | *Optional[str]* | ➖ | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| |
| `x_open_router_categories` | *Optional[str]* | ➖ | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| |
| `input_references` | List[[components.SpeechInputReference](../../components/speechinputreference.mdx)] | ➖ | Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference. | [
\{
"input_audio": \{
"data": "data:audio/wav;base64,UklGRuQXDABXQVZF..."
},
"type": "input_audio"
},
\{
"text": "I used to rule the world.",
"type": "text"
}
] |
+| `instructions` | *Optional[str]* | ➖ | Delivery instructions for the whole request, such as tone, pacing, or emotion. Supported by OpenAI gpt-4o-mini-tts and Gemini TTS models. Ignored by other providers. | Speak in a warm and friendly tone. |
| `provider` | [Optional[components.SpeechRequestProvider]](../../components/speechrequestprovider.mdx) | ➖ | Provider configuration: data policy routing preferences (`zdr`, `data_collection`) and provider-specific passthrough options | |
| `response_format` | [Optional[components.SpeechRequestResponseFormat]](../../components/speechrequestresponseformat.mdx) | ➖ | Audio output format | pcm |
| `session_id` | *Optional[str]* | ➖ | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | session-1234 |
diff --git a/pyproject.toml b/pyproject.toml
index 112fa9a9..87f61eab 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,6 +1,6 @@
[project]
name = "openrouter"
-version = "1.3.26"
+version = "1.3.27"
description = "Official Python Client SDK for OpenRouter."
authors = [{ name = "OpenRouter" },]
readme = "README-PYPI.md"
diff --git a/src/openrouter/_version.py b/src/openrouter/_version.py
index 235db3ae..550534f0 100644
--- a/src/openrouter/_version.py
+++ b/src/openrouter/_version.py
@@ -3,10 +3,10 @@
import importlib.metadata
__title__: str = "openrouter"
-__version__: str = "1.3.26"
+__version__: str = "1.3.27"
__openapi_doc_version__: str = "1.0.0"
__gen_version__: str = "2.914.0"
-__user_agent__: str = "speakeasy-sdk/python 1.3.26 2.914.0 1.0.0 openrouter"
+__user_agent__: str = "speakeasy-sdk/python 1.3.27 2.914.0 1.0.0 openrouter"
try:
if __package__ is not None:
diff --git a/src/openrouter/components/__init__.py b/src/openrouter/components/__init__.py
index 2e45778d..92a29ef9 100644
--- a/src/openrouter/components/__init__.py
+++ b/src/openrouter/components/__init__.py
@@ -3827,6 +3827,7 @@
ShellServerToolEnvironment,
ShellServerToolEnvironmentTypedDict,
)
+ from .speechinput import SpeechInput, SpeechInputTypedDict
from .speechinputreference import (
SpeechInputReference,
SpeechInputReferenceTypedDict,
@@ -3862,6 +3863,7 @@
SpeechRequestResponseFormat,
SpeechRequestTypedDict,
)
+ from .speechturn import SpeechTurn, SpeechTurnTypedDict
from .stopservertoolswhencondition import (
StopServerToolsWhenCondition,
StopServerToolsWhenConditionTypedDict,
@@ -7019,6 +7021,7 @@
"SourceContent",
"SourceContentTypedDict",
"SourceType",
+ "SpeechInput",
"SpeechInputReference",
"SpeechInputReferenceAudio",
"SpeechInputReferenceAudioInput",
@@ -7034,12 +7037,15 @@
"SpeechInputReferenceTextType",
"SpeechInputReferenceTextTypedDict",
"SpeechInputReferenceTypedDict",
+ "SpeechInputTypedDict",
"SpeechRequest",
"SpeechRequestDataCollection",
"SpeechRequestProvider",
"SpeechRequestProviderTypedDict",
"SpeechRequestResponseFormat",
"SpeechRequestTypedDict",
+ "SpeechTurn",
+ "SpeechTurnTypedDict",
"Speed",
"Stance",
"StanceTypedDict",
@@ -10290,6 +10296,8 @@
"ShellServerToolEngine": ".shellservertoolengine",
"ShellServerToolEnvironment": ".shellservertoolenvironment",
"ShellServerToolEnvironmentTypedDict": ".shellservertoolenvironment",
+ "SpeechInput": ".speechinput",
+ "SpeechInputTypedDict": ".speechinput",
"SpeechInputReference": ".speechinputreference",
"SpeechInputReferenceTypedDict": ".speechinputreference",
"SpeechInputReferenceAudio": ".speechinputreferenceaudio",
@@ -10311,6 +10319,8 @@
"SpeechRequestProviderTypedDict": ".speechrequest",
"SpeechRequestResponseFormat": ".speechrequest",
"SpeechRequestTypedDict": ".speechrequest",
+ "SpeechTurn": ".speechturn",
+ "SpeechTurnTypedDict": ".speechturn",
"StopServerToolsWhenCondition": ".stopservertoolswhencondition",
"StopServerToolsWhenConditionTypedDict": ".stopservertoolswhencondition",
"StopServerToolsWhenFinishReasonIs": ".stopservertoolswhenfinishreasonis",
diff --git a/src/openrouter/components/speechinput.py b/src/openrouter/components/speechinput.py
new file mode 100644
index 00000000..676f3892
--- /dev/null
+++ b/src/openrouter/components/speechinput.py
@@ -0,0 +1,16 @@
+"""Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT."""
+
+from __future__ import annotations
+from .speechturn import SpeechTurn, SpeechTurnTypedDict
+from typing import List, Union
+from typing_extensions import TypeAliasType
+
+
+SpeechInputTypedDict = TypeAliasType(
+ "SpeechInputTypedDict", Union[str, List[SpeechTurnTypedDict]]
+)
+r"""Text to synthesize, or a list of turns for multi-speaker input. Each turn has its own text, voice, and instructions. Multi-speaker input is currently supported by Gemini TTS models only."""
+
+
+SpeechInput = TypeAliasType("SpeechInput", Union[str, List[SpeechTurn]])
+r"""Text to synthesize, or a list of turns for multi-speaker input. Each turn has its own text, voice, and instructions. Multi-speaker input is currently supported by Gemini TTS models only."""
diff --git a/src/openrouter/components/speechrequest.py b/src/openrouter/components/speechrequest.py
index 5a62dc47..ad116652 100644
--- a/src/openrouter/components/speechrequest.py
+++ b/src/openrouter/components/speechrequest.py
@@ -2,6 +2,7 @@
from __future__ import annotations
from .provideroptions import ProviderOptions, ProviderOptionsTypedDict
+from .speechinput import SpeechInput, SpeechInputTypedDict
from .speechinputreference import SpeechInputReference, SpeechInputReferenceTypedDict
from .traceconfig import TraceConfig, TraceConfigTypedDict
from openrouter.types import (
@@ -101,12 +102,14 @@ def serialize_model(self, handler):
class SpeechRequestTypedDict(TypedDict):
r"""Text-to-speech request input"""
- input: str
- r"""Text to synthesize"""
+ input: SpeechInputTypedDict
+ r"""Text to synthesize, or a list of turns for multi-speaker input. Each turn has its own text, voice, and instructions. Multi-speaker input is currently supported by Gemini TTS models only."""
model: str
r"""TTS model identifier"""
input_references: NotRequired[List[SpeechInputReferenceTypedDict]]
r"""Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference."""
+ instructions: NotRequired[str]
+ r"""Delivery instructions for the whole request, such as tone, pacing, or emotion. Supported by OpenAI gpt-4o-mini-tts and Gemini TTS models. Ignored by other providers."""
provider: NotRequired[SpeechRequestProviderTypedDict]
r"""Provider configuration: data policy routing preferences (`zdr`, `data_collection`) and provider-specific passthrough options"""
response_format: NotRequired[SpeechRequestResponseFormat]
@@ -126,8 +129,8 @@ class SpeechRequestTypedDict(TypedDict):
class SpeechRequest(BaseModel):
r"""Text-to-speech request input"""
- input: str
- r"""Text to synthesize"""
+ input: SpeechInput
+ r"""Text to synthesize, or a list of turns for multi-speaker input. Each turn has its own text, voice, and instructions. Multi-speaker input is currently supported by Gemini TTS models only."""
model: str
r"""TTS model identifier"""
@@ -135,6 +138,9 @@ class SpeechRequest(BaseModel):
input_references: Optional[List[SpeechInputReference]] = None
r"""Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference."""
+ instructions: Optional[str] = None
+ r"""Delivery instructions for the whole request, such as tone, pacing, or emotion. Supported by OpenAI gpt-4o-mini-tts and Gemini TTS models. Ignored by other providers."""
+
provider: Optional[SpeechRequestProvider] = None
r"""Provider configuration: data policy routing preferences (`zdr`, `data_collection`) and provider-specific passthrough options"""
@@ -161,6 +167,7 @@ def serialize_model(self, handler):
optional_fields = set(
[
"input_references",
+ "instructions",
"provider",
"response_format",
"session_id",
diff --git a/src/openrouter/components/speechturn.py b/src/openrouter/components/speechturn.py
new file mode 100644
index 00000000..d3e0e4be
--- /dev/null
+++ b/src/openrouter/components/speechturn.py
@@ -0,0 +1,43 @@
+"""Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT."""
+
+from __future__ import annotations
+from openrouter.types import BaseModel, UNSET_SENTINEL
+from pydantic import model_serializer
+from typing import Optional
+from typing_extensions import NotRequired, TypedDict
+
+
+class SpeechTurnTypedDict(TypedDict):
+ text: str
+ r"""Text to synthesize"""
+ instructions: NotRequired[str]
+ r"""Delivery instructions for this turn, such as tone, pacing, or emotion. Overrides the top-level `instructions`."""
+ voice: NotRequired[str]
+ r"""Voice for this turn. Defaults to the top-level `voice`."""
+
+
+class SpeechTurn(BaseModel):
+ text: str
+ r"""Text to synthesize"""
+
+ instructions: Optional[str] = None
+ r"""Delivery instructions for this turn, such as tone, pacing, or emotion. Overrides the top-level `instructions`."""
+
+ voice: Optional[str] = None
+ r"""Voice for this turn. Defaults to the top-level `voice`."""
+
+ @model_serializer(mode="wrap")
+ def serialize_model(self, handler):
+ optional_fields = set(["instructions", "voice"])
+ serialized = handler(self)
+ m = {}
+
+ for n, f in type(self).model_fields.items():
+ k = f.alias or n
+ val = serialized.get(k, serialized.get(n))
+
+ if val != UNSET_SENTINEL:
+ if val is not None or k not in optional_fields:
+ m[k] = val
+
+ return m
diff --git a/src/openrouter/tts.py b/src/openrouter/tts.py
index 6da4bd90..5b1bf2a6 100644
--- a/src/openrouter/tts.py
+++ b/src/openrouter/tts.py
@@ -16,7 +16,7 @@ class TTS(BaseSDK):
def create_speech(
self,
*,
- input: str,
+ input: Union[components.SpeechInput, components.SpeechInputTypedDict],
model: str,
http_referer: Optional[str] = None,
x_open_router_title: Optional[str] = None,
@@ -27,6 +27,7 @@ def create_speech(
Iterable[components.SpeechInputReferenceTypedDict],
]
] = None,
+ instructions: Optional[str] = None,
provider: Optional[
Union[
components.SpeechRequestProvider,
@@ -50,7 +51,7 @@ def create_speech(
Synthesizes audio from the input text. Returns a raw audio bytestream in the requested format (e.g. mp3, pcm, wav).
- :param input: Text to synthesize
+ :param input: Text to synthesize, or a list of turns for multi-speaker input. Each turn has its own text, voice, and instructions. Multi-speaker input is currently supported by Gemini TTS models only.
:param model: TTS model identifier
:param http_referer: The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
@@ -60,6 +61,7 @@ def create_speech(
:param x_open_router_categories: Comma-separated list of app categories (e.g. \"cli-agent,cloud-agent\"). Used for marketplace rankings.
:param input_references: Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference.
+ :param instructions: Delivery instructions for the whole request, such as tone, pacing, or emotion. Supported by OpenAI gpt-4o-mini-tts and Gemini TTS models. Ignored by other providers.
:param provider: Provider configuration: data policy routing preferences (`zdr`, `data_collection`) and provider-specific passthrough options
:param response_format: Audio output format
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
@@ -87,10 +89,11 @@ def create_speech(
x_open_router_title=x_open_router_title,
x_open_router_categories=x_open_router_categories,
speech_request=components.SpeechRequest(
- input=input,
+ input=utils.get_pydantic_model(input, components.SpeechInput),
input_references=utils.get_pydantic_model(
input_references, Optional[List[components.SpeechInputReference]]
),
+ instructions=instructions,
model=model,
provider=utils.get_pydantic_model(
provider, Optional[components.SpeechRequestProvider]
@@ -269,7 +272,7 @@ def create_speech(
async def create_speech_async(
self,
*,
- input: str,
+ input: Union[components.SpeechInput, components.SpeechInputTypedDict],
model: str,
http_referer: Optional[str] = None,
x_open_router_title: Optional[str] = None,
@@ -280,6 +283,7 @@ async def create_speech_async(
Iterable[components.SpeechInputReferenceTypedDict],
]
] = None,
+ instructions: Optional[str] = None,
provider: Optional[
Union[
components.SpeechRequestProvider,
@@ -303,7 +307,7 @@ async def create_speech_async(
Synthesizes audio from the input text. Returns a raw audio bytestream in the requested format (e.g. mp3, pcm, wav).
- :param input: Text to synthesize
+ :param input: Text to synthesize, or a list of turns for multi-speaker input. Each turn has its own text, voice, and instructions. Multi-speaker input is currently supported by Gemini TTS models only.
:param model: TTS model identifier
:param http_referer: The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
@@ -313,6 +317,7 @@ async def create_speech_async(
:param x_open_router_categories: Comma-separated list of app categories (e.g. \"cli-agent,cloud-agent\"). Used for marketplace rankings.
:param input_references: Reference content for stateless voice cloning or voice design. Audio mode: one to three `input_audio` parts, each optionally paired with a `text` part carrying its transcript (a single clip accepts its transcript before or after it; with multiple clips each transcript immediately follows its clip); only routed to endpoints that support voice cloning (and multiple references when more than one part is sent). Image mode: exactly one `image_url` part; only routed to endpoints that support image references. The two modes cannot be mixed. An empty array is treated as no reference.
+ :param instructions: Delivery instructions for the whole request, such as tone, pacing, or emotion. Supported by OpenAI gpt-4o-mini-tts and Gemini TTS models. Ignored by other providers.
:param provider: Provider configuration: data policy routing preferences (`zdr`, `data_collection`) and provider-specific passthrough options
:param response_format: Audio output format
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
@@ -340,10 +345,11 @@ async def create_speech_async(
x_open_router_title=x_open_router_title,
x_open_router_categories=x_open_router_categories,
speech_request=components.SpeechRequest(
- input=input,
+ input=utils.get_pydantic_model(input, components.SpeechInput),
input_references=utils.get_pydantic_model(
input_references, Optional[List[components.SpeechInputReference]]
),
+ instructions=instructions,
model=model,
provider=utils.get_pydantic_model(
provider, Optional[components.SpeechRequestProvider]
diff --git a/uv.lock b/uv.lock
index cdc856ea..2f4e8623 100644
--- a/uv.lock
+++ b/uv.lock
@@ -213,7 +213,7 @@ wheels = [
[[package]]
name = "openrouter"
-version = "1.3.26"
+version = "1.3.27"
source = { editable = "." }
dependencies = [
{ name = "httpcore" },