diff --git a/.speakeasy/gen.lock b/.speakeasy/gen.lock
index d230f16a..f1e4cb50 100644
--- a/.speakeasy/gen.lock
+++ b/.speakeasy/gen.lock
@@ -1,19 +1,19 @@
lockVersion: 2.0.0
id: c48cf606-fb42-4a45-9c23-8f0555307828
management:
- docChecksum: a9795ca524d72817a1dfc7d052af6535
+ docChecksum: 802f771745200ec789ec40b84cde6c16
docVersion: 1.0.0
speakeasyVersion: 1.787.0
generationVersion: 2.914.0
- releaseVersion: 1.3.6
- configChecksum: 189bc993faf2640dc049feb3dd575398
+ releaseVersion: 1.3.7
+ configChecksum: 4f3beb0365ac0e410ce9346ead2bc186
repoURL: https://github.com/OpenRouterTeam/python-sdk.git
installationURL: https://github.com/OpenRouterTeam/python-sdk.git
published: true
persistentEdits:
- generation_id: a8d2066d-821b-4c4a-9b68-e6a90abbdf10
- pristine_commit_hash: 4d9c4591e1c2739cd52212023a3cf0479cfd85a9
- pristine_tree_hash: be6292cecd56d6ee419c2c5de058124aa857adde
+ generation_id: a76b44ea-974e-4345-a9c0-a0cb4de6b2c6
+ pristine_commit_hash: 10e7143eb49447f9ba97d237e81cbfa63e99c955
+ pristine_tree_hash: 2755b9e7af9258a5a5ace0e1ac9b760849618af2
features:
python:
acceptHeaders: 3.0.0
@@ -1796,8 +1796,8 @@ trackedFiles:
pristine_git_object: 220be03bd674a257c2064c0455104959b3974a35
docs/components/chatrequest.mdx:
id: 055812be14d8
- last_write_checksum: sha1:e9f36dfa751e7821cb39e6f447066e300324f2f8
- pristine_git_object: e9cddf12eb787653a2911c203b30a97be2f36ab3
+ last_write_checksum: sha1:f52eef7f2f9d526f627739cc63b76ed40e38acc1
+ pristine_git_object: 84d57986843d56e93e8b016d1680842101470a11
docs/components/chatrequesteffort.mdx:
id: 703c73252ebd
last_write_checksum: sha1:243510f188071a3f317dc06a07fbcfcedd3cf821
@@ -1816,8 +1816,8 @@ trackedFiles:
pristine_git_object: 648e271855d957af7003d1dd3607116ab6f71ea3
docs/components/chatrequestservicetier.mdx:
id: 3037649ffa34
- last_write_checksum: sha1:4300697ab8232181443e9dd1b403c534f448cefa
- pristine_git_object: cb8cba525e66e5515ea69a5217d1b20809250003
+ last_write_checksum: sha1:3db6cc3bc4222d261b7975c5a4e2257d6e0aa40e
+ pristine_git_object: d215d672c97f91fcb38cd5a5bed5c00703545161
docs/components/chatresult.mdx:
id: 1d855f0d207b
last_write_checksum: sha1:70644649b0eef0c4edb2e7dd2d534847530e68c6
@@ -6208,16 +6208,16 @@ trackedFiles:
pristine_git_object: c5eb0598512c67cb4dcb72b9ef93c9a027a1a989
docs/components/responsesrequest.mdx:
id: 0dbcef40a4b1
- last_write_checksum: sha1:769a00c73703af1d080294617106dbde100ce30e
- pristine_git_object: 61e284e3ec7cc02541b74e533300ba8072cc0e6c
+ last_write_checksum: sha1:170d607dc913444b20d3f68dfb35fce3e85902b1
+ pristine_git_object: 97e9c52af4c5b817dfcb7abc533d7ee25c77ca75
docs/components/responsesrequestplugin.mdx:
id: 4d550c41c85f
last_write_checksum: sha1:6b8983d000776727ef4ae1b8f878c5116c983255
pristine_git_object: 70990bd403bd20926c3c8202fd61e6ea6268eff0
docs/components/responsesrequestservicetier.mdx:
id: 5000ee5983ff
- last_write_checksum: sha1:3b17d5eb36a9a41d84fd38fc7478d9368975438a
- pristine_git_object: 61ea7fe276229819beff171a25ee0438233af94f
+ last_write_checksum: sha1:833cacf1f124cb94079d80eb80a894cd51f12ba4
+ pristine_git_object: e4c42ad571a892ca402551f858cca27b366ab4f5
docs/components/responsesrequesttoolfunction.mdx:
id: dd0c817d15e6
last_write_checksum: sha1:4b00fb255e66f178316a613edf7185a93db88b58
@@ -6244,8 +6244,8 @@ trackedFiles:
pristine_git_object: 2609a9af043ce5570970252f3ebd5df12d693be3
docs/components/routedservicetier.mdx:
id: 1ecc7cdd925e
- last_write_checksum: sha1:538e8fc60dab1d88f2b735682161cc8ab6101d5d
- pristine_git_object: 0c826312b0c506004b51e7dbcaceb82639046687
+ last_write_checksum: sha1:674d22f78f85c9d89408f6e791023fd6d1f36289
+ pristine_git_object: acaa2fd7cdb610f0541def55869bd6386028b152
docs/components/routerattempt.mdx:
id: 372598a145ed
last_write_checksum: sha1:51d8aee604f0093ea6df2df0fd73f02241190f70
@@ -9352,16 +9352,16 @@ trackedFiles:
pristine_git_object: 5026ac2a283240480026d80d2f77d5528ae8bdc8
docs/sdks/betaresponses/README.mdx:
id: 8a1e987c9840
- last_write_checksum: sha1:c7b700db6db684be6a26ad961babefa52e52ba67
- pristine_git_object: cf721f58bd04c9e9525e4f9c2d57ee526dd7134d
+ last_write_checksum: sha1:7040bfdaefd86fd0cfbcd84253a03cb571c78500
+ pristine_git_object: 09ec1e9265382b113df9fa03a6fe8cef09691cb5
docs/sdks/byok/README.mdx:
id: 17792f3b180d
last_write_checksum: sha1:7d9373a5a1d5c41b8dc333641812c07a4837965b
pristine_git_object: 9326d7d136af4b87609a5794331bc4a78bd6bdde
docs/sdks/chat/README.mdx:
id: 1dd859c23fe1
- last_write_checksum: sha1:75cf7ecf14bdaae4b3e2c767dc3d2c9d9e41d31c
- pristine_git_object: 071595b609eebe84729cd8472ef8ce816422f7f9
+ last_write_checksum: sha1:6dd9a73ac931d57cc5e641488548daf51e98cb39
+ pristine_git_object: faa37892c8b3e939974fe23ea1f17f4c2033a2f8
docs/sdks/classifications/README.mdx:
id: 4786130ef02a
last_write_checksum: sha1:87ea3518b4147ae7833b01da7be92d41a4b49df7
@@ -9428,8 +9428,8 @@ trackedFiles:
pristine_git_object: e45eea9aea2f1a8c6ec83a496b844524f17cae00
docs/sdks/presets/README.mdx:
id: eeb505928d52
- last_write_checksum: sha1:65e7d837befd019c63cc694f18cf713b979dcc15
- pristine_git_object: ae5b31b25dda5ce9df2a247a13217f613c68b141
+ last_write_checksum: sha1:ebc87ab54dcbb875e226a24240f896ee3c579aa6
+ pristine_git_object: b6a18f2e5a54aa0328d777dfc0cff7764416f68d
docs/sdks/privateendpoints/README.mdx:
id: 54dee70b8cc7
last_write_checksum: sha1:3ee195b23b8124d351823bb7ef8fb956ef281c3c
@@ -9444,8 +9444,8 @@ trackedFiles:
pristine_git_object: 0836d633a90c128d0062b9fb532aa8be2591927a
docs/sdks/responses/README.mdx:
id: abab319e080e
- last_write_checksum: sha1:2a490c42fcc07f6dac5ef0204c9bb055f3e0afa5
- pristine_git_object: eec7232321e300a7e5c2232ec0a6754d6cac11d1
+ last_write_checksum: sha1:bb12fa0fb8db15cf708d3ce40df2d9f02f2aa062
+ pristine_git_object: 004b36f713afed67d1e6197eb65650e6e9495c7a
docs/sdks/scim/README.mdx:
id: fd6d8a91a053
last_write_checksum: sha1:0a28f030bf42146fde86887c2b79831af07f1d4f
@@ -9480,8 +9480,8 @@ trackedFiles:
pristine_git_object: 3e38f1a929f7d6b1d6de74604aa87e3d8f010544
pyproject.toml:
id: 5d07e7d72637
- last_write_checksum: sha1:41d2608f056d9be9baf7e5e037ea0343091c42cd
- pristine_git_object: f830583b6813b44c8701c21828f86c23d0769854
+ last_write_checksum: sha1:ffbd5d5d3a1a3aaa413ae93da4d950ab87dd6f20
+ pristine_git_object: 953dedaac785f704a55815a976a56ea2d0542c57
scripts/prepare_readme.py:
id: e0c5957a6035
last_write_checksum: sha1:77f44b60b98bc126557ec27391f91dfba764bb54
@@ -9508,8 +9508,8 @@ trackedFiles:
pristine_git_object: 86713cfea633e09d33b3d4e65281071fe20e6137
src/openrouter/_version.py:
id: d8d15ad6c586
- last_write_checksum: sha1:becbef69a8ff6acced5c232de9d26cc6749275f6
- pristine_git_object: 6e1c5673a8db2ca66ca5b2ed2bf695f03c1b1944
+ last_write_checksum: sha1:c4880ef34499c45e2dc27fde44c9b19cc342197f
+ pristine_git_object: 0f901af47cee90907a21b2f6b0f0c75fd3104d39
src/openrouter/alpha.py:
id: 306c4d93308d
last_write_checksum: sha1:30f55a360f41376ab194b9ea725fe4e001a5ae1a
@@ -9540,16 +9540,16 @@ trackedFiles:
pristine_git_object: aa95a65cf560fd7edb0c3c6f5fed707ca5015da3
src/openrouter/beta_responses.py:
id: 001eaf848bf1
- last_write_checksum: sha1:185af24a8f8b484bd56e2d0c9132699e7200a0fb
- pristine_git_object: 4e287344f0570c64f5faa3be7c06fd6100fe31cc
+ last_write_checksum: sha1:f931137f4afd89e90223c490ce08af10080f09ce
+ pristine_git_object: c5a9c739c250db79395f5b0f7db30f56b6fcdaff
src/openrouter/byok.py:
id: bec352462ae1
last_write_checksum: sha1:ed4d5ecf5b0bf3fd004cc8629474ec0e100c0df5
pristine_git_object: 2cc6ba136f9a53803849ba48ab3c85db9ed12436
src/openrouter/chat.py:
id: 723fdce15c1d
- last_write_checksum: sha1:1a4e5375fb95884be6db0f508e1c2c9cda27b20e
- pristine_git_object: 7ea55df0463528a5e251eb93eba8580d4cec8593
+ last_write_checksum: sha1:9d9c3915a58c41a74bfebda2b743911324d0bde3
+ pristine_git_object: 899a30f9d9dbb60f1a14dd989eea748453d0ca8d
src/openrouter/classifications.py:
id: 4f2efc24b95b
last_write_checksum: sha1:3ab6fcd9ef679cd2149ec8dfbb4a86109cd152ad
@@ -10316,8 +10316,8 @@ trackedFiles:
pristine_git_object: 8bbde153b53887e426989fd5b0486b6dc1122810
src/openrouter/components/chatrequest.py:
id: 5e39eaefa9cf
- last_write_checksum: sha1:a6cac9daba13ffa180d87cc64aa53de0d411b4aa
- pristine_git_object: 416e26207d90de151047ce101a9d6ad49c6d44e9
+ last_write_checksum: sha1:7bb09c81a30bfdf6caad28379c5eba19edcfad97
+ pristine_git_object: 3746e1749babb30a94faf241d9a3972d9d2b9ca3
src/openrouter/components/chatresult.py:
id: 9062fe2935fe
last_write_checksum: sha1:2ff1963c54e7e79500115c8549c8d395f4e336d1
@@ -12008,8 +12008,8 @@ trackedFiles:
pristine_git_object: 6cf5d428899c8722a49aa4ed5918ffe95b793549
src/openrouter/components/providerresponse.py:
id: ad3887be54c5
- last_write_checksum: sha1:66a7c5746d98e7fc89cf21c71f96eca6d59a17ce
- pristine_git_object: 9918b5604df2e083a4e52e631eee90e0c4da32a9
+ last_write_checksum: sha1:a35f4afc0c5172233847ed70e4811adfd8332117
+ pristine_git_object: 0f35a76c37e8f904e5fd8bb2bcbc4470f6a0b5d8
src/openrouter/components/providersort.py:
id: 348e382bf494
last_write_checksum: sha1:57551507f95cd2e16ef995e1c13f859fd0726152
@@ -12156,8 +12156,8 @@ trackedFiles:
pristine_git_object: 1364b9bd6384600f0f59afbe181eb0e940f686eb
src/openrouter/components/responsesrequest.py:
id: 8c850080ec5d
- last_write_checksum: sha1:28de0d1f80241721fe79a381e4d80cc421ba827d
- pristine_git_object: e5665b8c7ac98d857160688ea5abeccffb359c39
+ last_write_checksum: sha1:ff7f75de5b451b8a7e65e6152d8aa7b3931e6e00
+ pristine_git_object: 710a27a25adbe3a110592fc7ef778522999686bb
src/openrouter/components/responsesstreamingresponse.py:
id: 142379f3bd90
last_write_checksum: sha1:a48d6f3610f5d15248ef4d25a0d88a0a4a1a9db7
@@ -13508,8 +13508,8 @@ trackedFiles:
pristine_git_object: 6e0305fa49d62303a81e2dfbac78ae63449a94d6
src/openrouter/presets.py:
id: bd0c40379dcd
- last_write_checksum: sha1:307215f3fac400989dbdb85eb13dbf9fbbc4b038
- pristine_git_object: 38cc2e32a7d072feef1331b24b3ec22cfb789588
+ last_write_checksum: sha1:ebc4d89ee34dcc7d6fb2462948e69567f5e5b9ff
+ pristine_git_object: 779e5f8da93b67e9203c9a1e718c1ebb19bec381
src/openrouter/private_endpoints.py:
id: fdbce8846ed0
last_write_checksum: sha1:f24e5496b1b57429d578c6422aef9ecb3b204760
@@ -13528,8 +13528,8 @@ trackedFiles:
pristine_git_object: 9db3cc719bd343ee7be0183d0c87b50c29e5eb31
src/openrouter/responses.py:
id: f2108fb635e1
- last_write_checksum: sha1:914622b408bf676070df8314389005ac1bebd064
- pristine_git_object: a40be38d4cbf460ecf921723841b1a414d638dee
+ last_write_checksum: sha1:c1983c2bfd74feb72990dc26104bfdefcd7390a2
+ pristine_git_object: 9e2598ff83d20980410acdd68d4ace3b951dc28c
src/openrouter/scim.py:
id: 351afd6b616e
last_write_checksum: sha1:2f58c4cf29650549257e79c476de0a9500a33772
@@ -17406,4 +17406,4 @@ examples:
"504":
application/json: {"error": {"code": 504, "message": "Vault request timed out"}}
examplesVersion: 1.0.2
-releaseNotes: "## Python SDK Changes:\n* `open_router.stt.create_transcription()`: \n * `request` **Changed** (Breaking ⚠️)\n * `response` **Changed**\n* `open_router.embeddings.generate()`: `request.provider` **Changed**\n* `open_router.endpoints.list()`: `response.data.endpoints[].provider_name.enum(eleven_labs)` **Added**\n* `open_router.stt.create_transcription_multipart()`: \n * `request` **Changed**\n * `response` **Changed**\n* `open_router.batch.create_batches()`: \n * `request.provider.only[].union(ProviderName).enum(eleven_labs)` **Added**\n* `open_router.byok.list()`: \n * `request.provider` **Changed**\n * `response.data[].provider.enum(elevenlabs)` **Added**\n* `open_router.byok.create()`: \n * `request.provider.enum(elevenlabs)` **Added**\n * `response.data.provider.enum(elevenlabs)` **Added**\n* `open_router.byok.get()`: `response.data.provider.enum(elevenlabs)` **Added**\n* `open_router.byok.update()`: `response.data.provider.enum(elevenlabs)` **Added**\n* `open_router.chat.send()`: `request.provider` **Changed**\n* `open_router.generations.get_generation()`: `response.data.provider_responses[].provider_name.enum(eleven_labs)` **Added**\n* `open_router.tts.create_speech()`: \n * `request.provider.options.elevenlabs` **Added**\n* `open_router.endpoints.list_zdr_endpoints()`: `response.data[].provider_name.enum(eleven_labs)` **Added**\n* `open_router.alpha.decisions.create()`: `request.provider` **Changed**\n* `open_router.images.generate()`: `request.provider` **Changed**\n* `open_router.presets.create_presets_chat_completions()`: `request.provider` **Changed**\n* `open_router.presets.create_presets_messages()`: `request.provider` **Changed**\n* `open_router.presets.create_presets_responses()`: `request.provider` **Changed**\n* `open_router.rerank.rerank()`: `request.provider` **Changed**\n* `open_router.responses.send()`: `request.provider` **Changed**\n* `open_router.beta.responses.send()`: `request.provider` **Changed**\n* `open_router.system_one.create()`: `request.provider` **Changed**\n* `open_router.video_generation.generate()`: \n * `request.provider.options.elevenlabs` **Added**\n"
+releaseNotes: "## Python SDK Changes:\n* `open_router.chat.send()`: \n * `request.service_tier.enum(ultrafast)` **Added**\n* `open_router.generations.get_generation()`: `response.data.provider_responses[].routed_service_tier.enum(ultrafast)` **Added**\n* `open_router.presets.create_presets_chat_completions()`: \n * `request.service_tier.enum(ultrafast)` **Added**\n* `open_router.presets.create_presets_responses()`: \n * `request.service_tier.enum(ultrafast)` **Added**\n* `open_router.responses.send()`: \n * `request.service_tier.enum(ultrafast)` **Added**\n* `open_router.beta.responses.send()`: \n * `request.service_tier.enum(ultrafast)` **Added**\n"
diff --git a/.speakeasy/gen.yaml b/.speakeasy/gen.yaml
index 0ad5daf2..90199269 100644
--- a/.speakeasy/gen.yaml
+++ b/.speakeasy/gen.yaml
@@ -36,7 +36,7 @@ generation:
documentation: mintlify
preApplyUnionDiscriminators: true
python:
- version: 1.3.6
+ version: 1.3.7
additionalDependencies:
dev: {}
main: {}
diff --git a/.speakeasy/out.openapi.yaml b/.speakeasy/out.openapi.yaml
index 4239932f..7079a9aa 100644
--- a/.speakeasy/out.openapi.yaml
+++ b/.speakeasy/out.openapi.yaml
@@ -7210,7 +7210,7 @@ components:
- 'integer'
- 'null'
service_tier:
- description: 'The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.'
+ description: 'The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.'
enum:
- 'auto'
- 'default'
@@ -7218,6 +7218,7 @@ components:
- 'flex'
- 'priority'
- 'scale'
+ - 'ultrafast'
- null
example: 'auto'
type:
@@ -25841,6 +25842,7 @@ components:
enum:
- 'flex'
- 'priority'
+ - 'ultrafast'
example: 'priority'
type: 'string'
x-speakeasy-unknown-values: allow
@@ -27187,7 +27189,7 @@ components:
- 'null'
service_tier:
default: 'auto'
- description: 'The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.'
+ description: 'The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.'
enum:
- 'auto'
- 'default'
@@ -27195,6 +27197,7 @@ components:
- 'flex'
- 'priority'
- 'scale'
+ - 'ultrafast'
- null
type:
- 'string'
@@ -27566,6 +27569,7 @@ components:
- 'flex'
- 'priority'
- 'scale'
+ - 'ultrafast'
- null
example: 'default'
type:
diff --git a/.speakeasy/workflow.lock b/.speakeasy/workflow.lock
index a2766631..ad35fdc8 100644
--- a/.speakeasy/workflow.lock
+++ b/.speakeasy/workflow.lock
@@ -2,8 +2,8 @@ speakeasyVersion: 1.787.0
sources:
OpenRouter API:
sourceNamespace: open-router-chat-completions-api
- sourceRevisionDigest: sha256:fa682f646ac7fac9fd7baeae8f6356547792c4e8f0c14d7b471bb055db9e47e3
- sourceBlobDigest: sha256:5caef8a5438f249e60312655dc3aa58171d47772f0a3b24f23a745a84162bc82
+ sourceRevisionDigest: sha256:468a0f30895dcd4a5b86d7ea89839daa81f79dccb0c2af8f8ec3dea52b216ad1
+ sourceBlobDigest: sha256:de433ae6845c61966cf3cff7a0dde2c7e38fd61a7228067b768298d3463093c1
tags:
- latest
- 1.0.0
@@ -11,10 +11,10 @@ targets:
open-router:
source: OpenRouter API
sourceNamespace: open-router-chat-completions-api
- sourceRevisionDigest: sha256:fa682f646ac7fac9fd7baeae8f6356547792c4e8f0c14d7b471bb055db9e47e3
- sourceBlobDigest: sha256:5caef8a5438f249e60312655dc3aa58171d47772f0a3b24f23a745a84162bc82
+ sourceRevisionDigest: sha256:468a0f30895dcd4a5b86d7ea89839daa81f79dccb0c2af8f8ec3dea52b216ad1
+ sourceBlobDigest: sha256:de433ae6845c61966cf3cff7a0dde2c7e38fd61a7228067b768298d3463093c1
codeSamplesNamespace: open-router-python-code-samples
- codeSamplesRevisionDigest: sha256:fa882c2f6a037169bc675adc4611202adcf873d3290137352cf5de5be043495f
+ codeSamplesRevisionDigest: sha256:97272069449099b7e76284cc25b252790cecc78c736278104bd0d422cc02c93f
workflow:
workflowVersion: 1.0.0
speakeasyVersion: 1.787.0
diff --git a/RELEASES.md b/RELEASES.md
index 64a52a49..8212ed9d 100644
--- a/RELEASES.md
+++ b/RELEASES.md
@@ -2829,4 +2829,14 @@ Based on:
### Generated
- [python v1.3.6] .
### Releases
-- [PyPI v1.3.6] https://pypi.org/project/openrouter/1.3.6 - .
\ No newline at end of file
+- [PyPI v1.3.6] https://pypi.org/project/openrouter/1.3.6 - .
+
+## 2026-09-29 19:29:20
+### Changes
+Based on:
+- OpenAPI Doc
+- Speakeasy CLI 1.787.0 (2.914.0) https://github.com/speakeasy-api/speakeasy
+### Generated
+- [python v1.3.7] .
+### Releases
+- [PyPI v1.3.7] https://pypi.org/project/openrouter/1.3.7 - .
\ No newline at end of file
diff --git a/docs/components/chatrequest.mdx b/docs/components/chatrequest.mdx
index e9cddf12..84d57986 100644
--- a/docs/components/chatrequest.mdx
+++ b/docs/components/chatrequest.mdx
@@ -35,7 +35,7 @@ Chat completion request parameters
| `repetition_penalty` | *OptionalNullable[float]* | :heavy_minus_sign: | Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter. | 1 |
| `response_format` | [Optional[components.ResponseFormat]](../components/responseformat.mdx) | :heavy_minus_sign: | Response format configuration | \{
"type": "json_object"
} |
| `seed` | *OptionalNullable[int]* | :heavy_minus_sign: | Random seed for deterministic outputs | 42 |
-| `service_tier` | [OptionalNullable[components.ChatRequestServiceTier]](../components/chatrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. | auto |
+| `service_tier` | [OptionalNullable[components.ChatRequestServiceTier]](../components/chatrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints. | auto |
| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | |
| `stop` | [OptionalNullable[components.Stop]](../components/stop.mdx) | :heavy_minus_sign: | Stop sequences (up to 4) | [
"\n"
] |
| `stop_server_tools_when` | List[[components.StopServerToolsWhenCondition](../components/stopservertoolswhencondition.mdx)] | :heavy_minus_sign: | Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call. | [
\{
"step_count": 5,
"type": "step_count_is"
},
\{
"max_cost_in_dollars": 0.5,
"type": "max_cost"
}
] |
diff --git a/docs/components/chatrequestservicetier.mdx b/docs/components/chatrequestservicetier.mdx
index cb8cba52..d215d672 100644
--- a/docs/components/chatrequestservicetier.mdx
+++ b/docs/components/chatrequestservicetier.mdx
@@ -2,7 +2,7 @@
title: "ChatRequestServiceTier"
---
-The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
## Example Usage
@@ -24,3 +24,4 @@ This is an open enum. Unrecognized values will not fail type checks.
- `"flex"`
- `"priority"`
- `"scale"`
+- `"ultrafast"`
diff --git a/docs/components/responsesrequest.mdx b/docs/components/responsesrequest.mdx
index 61e284e3..97e9c52a 100644
--- a/docs/components/responsesrequest.mdx
+++ b/docs/components/responsesrequest.mdx
@@ -33,7 +33,7 @@ Request schema for Responses endpoint
| `provider` | [OptionalNullable[components.ProviderPreferences]](../components/providerpreferences.mdx) | :heavy_minus_sign: | When multiple model providers are available, optionally indicate your routing preference. | \{
"allow_fallbacks": true
} |
| `reasoning` | [OptionalNullable[components.ReasoningConfig]](../components/reasoningconfig.mdx) | :heavy_minus_sign: | Configuration for reasoning mode in the response | \{
"enabled": true,
"summary": "auto"
} |
| `safety_identifier` | *OptionalNullable[str]* | :heavy_minus_sign: | Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account. | user-123 |
-| `service_tier` | [OptionalNullable[components.ResponsesRequestServiceTier]](../components/responsesrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. | |
+| `service_tier` | [OptionalNullable[components.ResponsesRequestServiceTier]](../components/responsesrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints. | |
| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | |
| `stop_server_tools_when` | List[[components.StopServerToolsWhenCondition](../components/stopservertoolswhencondition.mdx)] | :heavy_minus_sign: | Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call. | [
\{
"step_count": 5,
"type": "step_count_is"
},
\{
"max_cost_in_dollars": 0.5,
"type": "max_cost"
}
] |
| `store` | *Optional[Literal[False]]* | :heavy_minus_sign: | N/A | |
diff --git a/docs/components/responsesrequestservicetier.mdx b/docs/components/responsesrequestservicetier.mdx
index 61ea7fe2..e4c42ad5 100644
--- a/docs/components/responsesrequestservicetier.mdx
+++ b/docs/components/responsesrequestservicetier.mdx
@@ -2,7 +2,7 @@
title: "ResponsesRequestServiceTier"
---
-The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
## Example Usage
@@ -24,3 +24,4 @@ This is an open enum. Unrecognized values will not fail type checks.
- `"flex"`
- `"priority"`
- `"scale"`
+- `"ultrafast"`
diff --git a/docs/components/routedservicetier.mdx b/docs/components/routedservicetier.mdx
index 0c826312..acaa2fd7 100644
--- a/docs/components/routedservicetier.mdx
+++ b/docs/components/routedservicetier.mdx
@@ -20,3 +20,4 @@ This is an open enum. Unrecognized values will not fail type checks.
- `"flex"`
- `"priority"`
+- `"ultrafast"`
diff --git a/docs/sdks/betaresponses/README.mdx b/docs/sdks/betaresponses/README.mdx
index cf721f58..09ec1e92 100644
--- a/docs/sdks/betaresponses/README.mdx
+++ b/docs/sdks/betaresponses/README.mdx
@@ -92,7 +92,7 @@ with OpenRouter(
| `provider` | [OptionalNullable[components.ProviderPreferences]](../../components/providerpreferences.mdx) | :heavy_minus_sign: | When multiple model providers are available, optionally indicate your routing preference. | \{
"allow_fallbacks": true
} |
| `reasoning` | [OptionalNullable[components.ReasoningConfig]](../../components/reasoningconfig.mdx) | :heavy_minus_sign: | Configuration for reasoning mode in the response | \{
"enabled": true,
"summary": "auto"
} |
| `safety_identifier` | *OptionalNullable[str]* | :heavy_minus_sign: | Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account. | user-123 |
-| `service_tier` | [OptionalNullable[components.ResponsesRequestServiceTier]](../../components/responsesrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. | |
+| `service_tier` | [OptionalNullable[components.ResponsesRequestServiceTier]](../../components/responsesrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints. | |
| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | |
| `stop_server_tools_when` | List[[components.StopServerToolsWhenCondition](../../components/stopservertoolswhencondition.mdx)] | :heavy_minus_sign: | Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call. | [
\{
"step_count": 5,
"type": "step_count_is"
},
\{
"max_cost_in_dollars": 0.5,
"type": "max_cost"
}
] |
| `stream` | *Optional[bool]* | :heavy_minus_sign: | N/A | |
diff --git a/docs/sdks/chat/README.mdx b/docs/sdks/chat/README.mdx
index 071595b6..faa37892 100644
--- a/docs/sdks/chat/README.mdx
+++ b/docs/sdks/chat/README.mdx
@@ -109,7 +109,7 @@ with OpenRouter(
| `repetition_penalty` | *OptionalNullable[float]* | :heavy_minus_sign: | Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter. | 1 |
| `response_format` | [Optional[components.ResponseFormat]](../../components/responseformat.mdx) | :heavy_minus_sign: | Response format configuration | \{
"type": "json_object"
} |
| `seed` | *OptionalNullable[int]* | :heavy_minus_sign: | Random seed for deterministic outputs | 42 |
-| `service_tier` | [OptionalNullable[components.ChatRequestServiceTier]](../../components/chatrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. | auto |
+| `service_tier` | [OptionalNullable[components.ChatRequestServiceTier]](../../components/chatrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints. | auto |
| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | |
| `stop` | [OptionalNullable[components.Stop]](../../components/stop.mdx) | :heavy_minus_sign: | Stop sequences (up to 4) | [
"\n"
] |
| `stop_server_tools_when` | List[[components.StopServerToolsWhenCondition](../../components/stopservertoolswhencondition.mdx)] | :heavy_minus_sign: | Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call. | [
\{
"step_count": 5,
"type": "step_count_is"
},
\{
"max_cost_in_dollars": 0.5,
"type": "max_cost"
}
] |
diff --git a/docs/sdks/presets/README.mdx b/docs/sdks/presets/README.mdx
index ae5b31b2..b6a18f2e 100644
--- a/docs/sdks/presets/README.mdx
+++ b/docs/sdks/presets/README.mdx
@@ -185,7 +185,7 @@ with OpenRouter(
| `repetition_penalty` | *OptionalNullable[float]* | :heavy_minus_sign: | Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter. | 1 |
| `response_format` | [Optional[components.ResponseFormat]](../../components/responseformat.mdx) | :heavy_minus_sign: | Response format configuration | \{
"type": "json_object"
} |
| `seed` | *OptionalNullable[int]* | :heavy_minus_sign: | Random seed for deterministic outputs | 42 |
-| `service_tier` | [OptionalNullable[components.ChatRequestServiceTier]](../../components/chatrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. | auto |
+| `service_tier` | [OptionalNullable[components.ChatRequestServiceTier]](../../components/chatrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints. | auto |
| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | |
| `stop` | [OptionalNullable[components.Stop]](../../components/stop.mdx) | :heavy_minus_sign: | Stop sequences (up to 4) | [
"\n"
] |
| `stop_server_tools_when` | List[[components.StopServerToolsWhenCondition](../../components/stopservertoolswhencondition.mdx)] | :heavy_minus_sign: | Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call. | [
\{
"step_count": 5,
"type": "step_count_is"
},
\{
"max_cost_in_dollars": 0.5,
"type": "max_cost"
}
] |
@@ -358,7 +358,7 @@ with OpenRouter(
| `provider` | [OptionalNullable[components.ProviderPreferences]](../../components/providerpreferences.mdx) | :heavy_minus_sign: | When multiple model providers are available, optionally indicate your routing preference. | \{
"allow_fallbacks": true
} |
| `reasoning` | [OptionalNullable[components.ReasoningConfig]](../../components/reasoningconfig.mdx) | :heavy_minus_sign: | Configuration for reasoning mode in the response | \{
"enabled": true,
"summary": "auto"
} |
| `safety_identifier` | *OptionalNullable[str]* | :heavy_minus_sign: | Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account. | user-123 |
-| `service_tier` | [OptionalNullable[components.ResponsesRequestServiceTier]](../../components/responsesrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. | |
+| `service_tier` | [OptionalNullable[components.ResponsesRequestServiceTier]](../../components/responsesrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints. | |
| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | |
| `stop_server_tools_when` | List[[components.StopServerToolsWhenCondition](../../components/stopservertoolswhencondition.mdx)] | :heavy_minus_sign: | Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call. | [
\{
"step_count": 5,
"type": "step_count_is"
},
\{
"max_cost_in_dollars": 0.5,
"type": "max_cost"
}
] |
| `stream` | *Optional[bool]* | :heavy_minus_sign: | N/A | |
diff --git a/docs/sdks/responses/README.mdx b/docs/sdks/responses/README.mdx
index eec72323..004b36f7 100644
--- a/docs/sdks/responses/README.mdx
+++ b/docs/sdks/responses/README.mdx
@@ -92,7 +92,7 @@ with OpenRouter(
| `provider` | [OptionalNullable[components.ProviderPreferences]](../../components/providerpreferences.mdx) | :heavy_minus_sign: | When multiple model providers are available, optionally indicate your routing preference. | \{
"allow_fallbacks": true
} |
| `reasoning` | [OptionalNullable[components.ReasoningConfig]](../../components/reasoningconfig.mdx) | :heavy_minus_sign: | Configuration for reasoning mode in the response | \{
"enabled": true,
"summary": "auto"
} |
| `safety_identifier` | *OptionalNullable[str]* | :heavy_minus_sign: | Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account. | user-123 |
-| `service_tier` | [OptionalNullable[components.ResponsesRequestServiceTier]](../../components/responsesrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. | |
+| `service_tier` | [OptionalNullable[components.ResponsesRequestServiceTier]](../../components/responsesrequestservicetier.mdx) | :heavy_minus_sign: | The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints. | |
| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | |
| `stop_server_tools_when` | List[[components.StopServerToolsWhenCondition](../../components/stopservertoolswhencondition.mdx)] | :heavy_minus_sign: | Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call. | [
\{
"step_count": 5,
"type": "step_count_is"
},
\{
"max_cost_in_dollars": 0.5,
"type": "max_cost"
}
] |
| `stream` | *Optional[bool]* | :heavy_minus_sign: | N/A | |
diff --git a/pyproject.toml b/pyproject.toml
index f830583b..953dedaa 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,6 +1,6 @@
[project]
name = "openrouter"
-version = "1.3.6"
+version = "1.3.7"
description = "Official Python Client SDK for OpenRouter."
authors = [{ name = "OpenRouter" },]
readme = "README-PYPI.md"
diff --git a/src/openrouter/_version.py b/src/openrouter/_version.py
index 6e1c5673..0f901af4 100644
--- a/src/openrouter/_version.py
+++ b/src/openrouter/_version.py
@@ -3,10 +3,10 @@
import importlib.metadata
__title__: str = "openrouter"
-__version__: str = "1.3.6"
+__version__: str = "1.3.7"
__openapi_doc_version__: str = "1.0.0"
__gen_version__: str = "2.914.0"
-__user_agent__: str = "speakeasy-sdk/python 1.3.6 2.914.0 1.0.0 openrouter"
+__user_agent__: str = "speakeasy-sdk/python 1.3.7 2.914.0 1.0.0 openrouter"
try:
if __package__ is not None:
diff --git a/src/openrouter/beta_responses.py b/src/openrouter/beta_responses.py
index 4e287344..c5a9c739 100644
--- a/src/openrouter/beta_responses.py
+++ b/src/openrouter/beta_responses.py
@@ -160,7 +160,7 @@ def send(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -318,7 +318,7 @@ def send(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -479,7 +479,7 @@ def send(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -639,7 +639,7 @@ def send(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -1079,7 +1079,7 @@ async def send_async(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -1237,7 +1237,7 @@ async def send_async(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -1398,7 +1398,7 @@ async def send_async(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -1558,7 +1558,7 @@ async def send_async(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
diff --git a/src/openrouter/chat.py b/src/openrouter/chat.py
index 7ea55df0..899a30f9 100644
--- a/src/openrouter/chat.py
+++ b/src/openrouter/chat.py
@@ -167,7 +167,7 @@ def send(
:param repetition_penalty: Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter.
:param response_format: Response format configuration
:param seed: Random seed for deterministic outputs
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop: Stop sequences (up to 4)
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
@@ -335,7 +335,7 @@ def send(
:param repetition_penalty: Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter.
:param response_format: Response format configuration
:param seed: Random seed for deterministic outputs
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop: Stop sequences (up to 4)
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
@@ -505,7 +505,7 @@ def send(
:param repetition_penalty: Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter.
:param response_format: Response format configuration
:param seed: Random seed for deterministic outputs
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop: Stop sequences (up to 4)
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
@@ -674,7 +674,7 @@ def send(
:param repetition_penalty: Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter.
:param response_format: Response format configuration
:param seed: Random seed for deterministic outputs
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop: Stop sequences (up to 4)
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
@@ -1127,7 +1127,7 @@ async def send_async(
:param repetition_penalty: Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter.
:param response_format: Response format configuration
:param seed: Random seed for deterministic outputs
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop: Stop sequences (up to 4)
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
@@ -1295,7 +1295,7 @@ async def send_async(
:param repetition_penalty: Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter.
:param response_format: Response format configuration
:param seed: Random seed for deterministic outputs
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop: Stop sequences (up to 4)
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
@@ -1466,7 +1466,7 @@ async def send_async(
:param repetition_penalty: Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter.
:param response_format: Response format configuration
:param seed: Random seed for deterministic outputs
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop: Stop sequences (up to 4)
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
@@ -1636,7 +1636,7 @@ async def send_async(
:param repetition_penalty: Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter.
:param response_format: Response format configuration
:param seed: Random seed for deterministic outputs
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop: Stop sequences (up to 4)
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
diff --git a/src/openrouter/components/chatrequest.py b/src/openrouter/components/chatrequest.py
index 416e2620..3746e174 100644
--- a/src/openrouter/components/chatrequest.py
+++ b/src/openrouter/components/chatrequest.py
@@ -223,10 +223,11 @@ def serialize_model(self, handler):
"flex",
"priority",
"scale",
+ "ultrafast",
],
UnrecognizedStr,
]
-r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`."""
+r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints."""
StopTypedDict = TypeAliasType("StopTypedDict", Union[str, List[str]])
@@ -292,7 +293,7 @@ class ChatRequestTypedDict(TypedDict):
seed: NotRequired[Nullable[int]]
r"""Random seed for deterministic outputs"""
service_tier: NotRequired[Nullable[ChatRequestServiceTier]]
- r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`."""
+ r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints."""
session_id: NotRequired[str]
r"""A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters."""
stop: NotRequired[Nullable[StopTypedDict]]
@@ -404,7 +405,7 @@ class ChatRequest(BaseModel):
r"""Random seed for deterministic outputs"""
service_tier: OptionalNullable[ChatRequestServiceTier] = UNSET
- r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`."""
+ r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints."""
session_id: Optional[str] = None
r"""A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters."""
diff --git a/src/openrouter/components/providerresponse.py b/src/openrouter/components/providerresponse.py
index 9918b560..0f35a76c 100644
--- a/src/openrouter/components/providerresponse.py
+++ b/src/openrouter/components/providerresponse.py
@@ -161,6 +161,7 @@
Literal[
"flex",
"priority",
+ "ultrafast",
],
UnrecognizedStr,
]
diff --git a/src/openrouter/components/responsesrequest.py b/src/openrouter/components/responsesrequest.py
index e5665b8c..710a27a2 100644
--- a/src/openrouter/components/responsesrequest.py
+++ b/src/openrouter/components/responsesrequest.py
@@ -230,10 +230,11 @@ def serialize_model(self, handler):
"flex",
"priority",
"scale",
+ "ultrafast",
],
UnrecognizedStr,
]
-r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`."""
+r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints."""
ResponsesRequestType = Literal["function",]
@@ -416,7 +417,7 @@ class ResponsesRequestTypedDict(TypedDict):
safety_identifier: NotRequired[Nullable[str]]
r"""Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account."""
service_tier: NotRequired[Nullable[ResponsesRequestServiceTier]]
- r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`."""
+ r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints."""
session_id: NotRequired[str]
r"""A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters."""
stop_server_tools_when: NotRequired[List[StopServerToolsWhenConditionTypedDict]]
@@ -503,7 +504,7 @@ class ResponsesRequest(BaseModel):
r"""Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account."""
service_tier: OptionalNullable[ResponsesRequestServiceTier] = "auto"
- r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`."""
+ r"""The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints."""
session_id: Optional[str] = None
r"""A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters."""
diff --git a/src/openrouter/presets.py b/src/openrouter/presets.py
index 38cc2e32..779e5f8d 100644
--- a/src/openrouter/presets.py
+++ b/src/openrouter/presets.py
@@ -768,7 +768,7 @@ def create_presets_chat_completions(
:param repetition_penalty: Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter.
:param response_format: Response format configuration
:param seed: Random seed for deterministic outputs
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop: Stop sequences (up to 4)
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
@@ -1133,7 +1133,7 @@ async def create_presets_chat_completions_async(
:param repetition_penalty: Penalizes tokens based on how much they have already appeared in the text. A value of 1.0 means no penalty. Values above 1.0 penalize repeated tokens more strongly. Not all providers support this parameter.
:param response_format: Response format configuration
:param seed: Random seed for deterministic outputs
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop: Stop sequences (up to 4)
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
@@ -2133,7 +2133,7 @@ def create_presets_responses(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -2485,7 +2485,7 @@ async def create_presets_responses_async(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
diff --git a/src/openrouter/responses.py b/src/openrouter/responses.py
index a40be38d..9e2598ff 100644
--- a/src/openrouter/responses.py
+++ b/src/openrouter/responses.py
@@ -160,7 +160,7 @@ def send(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -318,7 +318,7 @@ def send(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -479,7 +479,7 @@ def send(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -639,7 +639,7 @@ def send(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -1079,7 +1079,7 @@ async def send_async(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -1237,7 +1237,7 @@ async def send_async(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -1398,7 +1398,7 @@ async def send_async(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
@@ -1558,7 +1558,7 @@ async def send_async(
:param provider: When multiple model providers are available, optionally indicate your routing preference.
:param reasoning: Configuration for reasoning mode in the response
:param safety_identifier: Recommended per-end-user identifier for abuse isolation. Use a stable ID, hash, or pseudonym. When a provider requires a user identity, OpenRouter folds it into the hashed identity sent upstream and never forwards it raw. If omitted, requests use an account-level identity, so provider policy blocks can affect the whole account.
- :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`.
+ :param service_tier: The service tier to use for processing this request. `fast` is accepted as an alias for `priority`. `ultrafast` prefers ultrafast endpoints and falls back to `priority`, then default endpoints.
:param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). When provided, OpenRouter uses it as the sticky routing key, routing all requests in the session to the same provider to maximize prompt cache hits. Also used for observability grouping. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters.
:param stop_server_tools_when: Stop conditions for the server-tool agent loop. Any condition firing halts the loop (OR logic). When set, this overrides `max_tool_calls`. When a condition fires while the model is still emitting tool calls, the pending tool calls are executed and one final turn is made with tool calls disabled so the response ends with a natural-language answer instead of an unfinished tool call.
:param stream:
diff --git a/uv.lock b/uv.lock
index 21aa38a3..1bf62969 100644
--- a/uv.lock
+++ b/uv.lock
@@ -213,7 +213,7 @@ wheels = [
[[package]]
name = "openrouter"
-version = "1.3.6"
+version = "1.3.7"
source = { editable = "." }
dependencies = [
{ name = "httpcore" },