From 983730f5913751b0409ce1f81ebd5e34cc78e9cc Mon Sep 17 00:00:00 2001 From: speakeasybot Date: Tue, 29 Sep 2026 19:08:54 +0000 Subject: [PATCH 1/2] =?UTF-8?q?##=20Python=20SDK=20Changes:=20*=20`open=5F?= =?UTF-8?q?router.stt.create=5Ftranscription()`:=20=20=20*=20=20`request`?= =?UTF-8?q?=20**Changed**=20(Breaking=20=E2=9A=A0=EF=B8=8F)=20=20=20*=20?= =?UTF-8?q?=20`response`=20**Changed**=20*=20`open=5Frouter.embeddings.gen?= =?UTF-8?q?erate()`:=20=20`request.provider`=20**Changed**=20*=20`open=5Fr?= =?UTF-8?q?outer.endpoints.list()`:=20=20`response.data.endpoints[].provid?= =?UTF-8?q?er=5Fname.enum(eleven=5Flabs)`=20**Added**=20*=20`open=5Frouter?= =?UTF-8?q?.stt.create=5Ftranscription=5Fmultipart()`:=20=20=20*=20=20`req?= =?UTF-8?q?uest`=20**Changed**=20=20=20*=20=20`response`=20**Changed**=20*?= =?UTF-8?q?=20`open=5Frouter.batch.create=5Fbatches()`:=20=20=20*=20=20`re?= =?UTF-8?q?quest.provider.only[].union(ProviderName).enum(eleven=5Flabs)`?= =?UTF-8?q?=20**Added**=20*=20`open=5Frouter.byok.list()`:=20=20=20*=20=20?= =?UTF-8?q?`request.provider`=20**Changed**=20=20=20*=20=20`response.data[?= =?UTF-8?q?].provider.enum(elevenlabs)`=20**Added**=20*=20`open=5Frouter.b?= =?UTF-8?q?yok.create()`:=20=20=20*=20=20`request.provider.enum(elevenlabs?= =?UTF-8?q?)`=20**Added**=20=20=20*=20=20`response.data.provider.enum(elev?= =?UTF-8?q?enlabs)`=20**Added**=20*=20`open=5Frouter.byok.get()`:=20=20`re?= =?UTF-8?q?sponse.data.provider.enum(elevenlabs)`=20**Added**=20*=20`open?= =?UTF-8?q?=5Frouter.byok.update()`:=20=20`response.data.provider.enum(ele?= =?UTF-8?q?venlabs)`=20**Added**=20*=20`open=5Frouter.chat.send()`:=20=20`?= =?UTF-8?q?request.provider`=20**Changed**=20*=20`open=5Frouter.generation?= =?UTF-8?q?s.get=5Fgeneration()`:=20=20`response.data.provider=5Fresponses?= =?UTF-8?q?[].provider=5Fname.enum(eleven=5Flabs)`=20**Added**=20*=20`open?= =?UTF-8?q?=5Frouter.tts.create=5Fspeech()`:=20=20=20*=20=20`request.provi?= =?UTF-8?q?der.options.elevenlabs`=20**Added**=20*=20`open=5Frouter.endpoi?= =?UTF-8?q?nts.list=5Fzdr=5Fendpoints()`:=20=20`response.data[].provider?= =?UTF-8?q?=5Fname.enum(eleven=5Flabs)`=20**Added**=20*=20`open=5Frouter.a?= =?UTF-8?q?lpha.decisions.create()`:=20=20`request.provider`=20**Changed**?= =?UTF-8?q?=20*=20`open=5Frouter.images.generate()`:=20=20`request.provide?= =?UTF-8?q?r`=20**Changed**=20*=20`open=5Frouter.presets.create=5Fpresets?= =?UTF-8?q?=5Fchat=5Fcompletions()`:=20=20`request.provider`=20**Changed**?= =?UTF-8?q?=20*=20`open=5Frouter.presets.create=5Fpresets=5Fmessages()`:?= =?UTF-8?q?=20=20`request.provider`=20**Changed**=20*=20`open=5Frouter.pre?= =?UTF-8?q?sets.create=5Fpresets=5Fresponses()`:=20=20`request.provider`?= =?UTF-8?q?=20**Changed**=20*=20`open=5Frouter.rerank.rerank()`:=20=20`req?= =?UTF-8?q?uest.provider`=20**Changed**=20*=20`open=5Frouter.responses.sen?= =?UTF-8?q?d()`:=20=20`request.provider`=20**Changed**=20*=20`open=5Froute?= =?UTF-8?q?r.beta.responses.send()`:=20=20`request.provider`=20**Changed**?= =?UTF-8?q?=20*=20`open=5Frouter.system=5Fone.create()`:=20=20`request.pro?= =?UTF-8?q?vider`=20**Changed**=20*=20`open=5Frouter.video=5Fgeneration.ge?= =?UTF-8?q?nerate()`:=20=20=20*=20=20`request.provider.options.elevenlabs`?= =?UTF-8?q?=20**Added**?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .speakeasy/gen.lock | 168 ++++++++++-------- .speakeasy/gen.yaml | 2 +- .speakeasy/out.openapi.yaml | 141 ++++++++++++++- .speakeasy/workflow.lock | 10 +- README-PYPI.md | 4 +- README.md | 4 +- RELEASES.md | 12 +- docs/components/byokproviderslug.mdx | 1 + ...gegenerationproviderpreferencesoptions.mdx | 1 + docs/components/providername.mdx | 1 + docs/components/provideroptions.mdx | 1 + .../providerresponseprovidername.mdx | 1 + docs/components/sttentity.mdx | 15 ++ docs/components/sttinlineinputaudio.mdx | 13 ++ docs/components/sttinputaudio.mdx | 20 ++- docs/components/sttrequest.mdx | 28 +-- docs/components/sttresponse.mdx | 2 + docs/components/sttsegment.mdx | 28 +-- docs/components/stturlinputaudio.mdx | 13 ++ docs/components/sttword.mdx | 17 +- docs/components/sttwordtype.mdx | 22 +++ .../videogenerationrequestoptions.mdx | 1 + ...udiotranscriptionsmultipartrequestbody.mdx | 26 +-- docs/operations/provider.mdx | 1 + docs/sdks/stt/README.mdx | 76 ++++---- pyproject.toml | 2 +- src/openrouter/_version.py | 4 +- src/openrouter/components/__init__.py | 19 +- src/openrouter/components/byokproviderslug.py | 1 + .../imagegenerationproviderpreferences.py | 4 + src/openrouter/components/providername.py | 1 + src/openrouter/components/provideroptions.py | 4 + src/openrouter/components/providerresponse.py | 1 + src/openrouter/components/sttentity.py | 34 ++++ .../components/sttinlineinputaudio.py | 31 ++++ src/openrouter/components/sttinputaudio.py | 37 ++-- src/openrouter/components/sttrequest.py | 20 ++- src/openrouter/components/sttresponse.py | 23 ++- src/openrouter/components/sttsegment.py | 12 ++ src/openrouter/components/stturlinputaudio.py | 49 +++++ src/openrouter/components/sttword.py | 37 +++- .../components/videogenerationrequest.py | 4 + .../createaudiotranscriptions_multipart.py | 43 ++++- src/openrouter/operations/listbyokkeys.py | 1 + src/openrouter/stt.py | 76 ++++++-- uv.lock | 2 +- 46 files changed, 775 insertions(+), 238 deletions(-) create mode 100644 docs/components/sttentity.mdx create mode 100644 docs/components/sttinlineinputaudio.mdx create mode 100644 docs/components/stturlinputaudio.mdx create mode 100644 docs/components/sttwordtype.mdx create mode 100644 src/openrouter/components/sttentity.py create mode 100644 src/openrouter/components/sttinlineinputaudio.py create mode 100644 src/openrouter/components/stturlinputaudio.py diff --git a/.speakeasy/gen.lock b/.speakeasy/gen.lock index cddb72ef..d230f16a 100644 --- a/.speakeasy/gen.lock +++ b/.speakeasy/gen.lock @@ -1,19 +1,19 @@ lockVersion: 2.0.0 id: c48cf606-fb42-4a45-9c23-8f0555307828 management: - docChecksum: 4b995c3ae9d0d0264f71a9ed6e1dbe4e + docChecksum: a9795ca524d72817a1dfc7d052af6535 docVersion: 1.0.0 speakeasyVersion: 1.787.0 generationVersion: 2.914.0 - releaseVersion: 1.3.5 - configChecksum: b2ad70554941d940596d8806b84a3e06 + releaseVersion: 1.3.6 + configChecksum: 189bc993faf2640dc049feb3dd575398 repoURL: https://github.com/OpenRouterTeam/python-sdk.git installationURL: https://github.com/OpenRouterTeam/python-sdk.git published: true persistentEdits: - generation_id: 5ca86c1a-0e74-4c00-9ef7-8a296de5d7a6 - pristine_commit_hash: 9f828376ba0a862a7bb45adfddb62e8e6cb82154 - pristine_tree_hash: d044ff5478833490d5ee26ead314cea5763041ef + generation_id: a8d2066d-821b-4c4a-9b68-e6a90abbdf10 + pristine_commit_hash: 4d9c4591e1c2739cd52212023a3cf0479cfd85a9 + pristine_tree_hash: be6292cecd56d6ee419c2c5de058124aa857adde features: python: acceptHeaders: 3.0.0 @@ -1580,8 +1580,8 @@ trackedFiles: pristine_git_object: 21f69ce3212aa1fca5c7ce02b78d9b249f15f3ff docs/components/byokproviderslug.mdx: id: 6ea70819f6d8 - last_write_checksum: sha1:d882a4dbedca7f2989deaa0024ff106b56ca30b5 - pristine_git_object: 3ab00cd1d221eff61e0ffdbe1ef2b8b45d8d391f + last_write_checksum: sha1:77ea10099b6bea344e8afac51865e7b2d69983f6 + pristine_git_object: 5d4b87a59be181224b8b7fc99c49495c448f4a11 docs/components/capabilitydescriptor.mdx: id: 34ac0e581d79 last_write_checksum: sha1:62ee9ccab6cec4572383cef654997ab4106d4676 @@ -3368,8 +3368,8 @@ trackedFiles: pristine_git_object: 11f1471c8b6c04f52af89d74d6dec2e9d12c2d82 docs/components/imagegenerationproviderpreferencesoptions.mdx: id: 32d5c4b3e06b - last_write_checksum: sha1:6181ca965732fae3efb0b80beed7085dc5e9e647 - pristine_git_object: 5515aeac6c5d2cd4b82762c05f32a59c2f5d2630 + last_write_checksum: sha1:1a7b2cb1d31e33d58d4b943898062599b51898d0 + pristine_git_object: 815789c44ac2f5dbf945dae1e47ff3ee9d01670e docs/components/imagegenerationproviderpreferencesorder.mdx: id: 0547877073f3 last_write_checksum: sha1:3e890c691097373cc9a2d7e47c3594f56b57809c @@ -5856,12 +5856,12 @@ trackedFiles: pristine_git_object: 1d91d32cd0bf4b64699bbbf05a7e95ab835b10cf docs/components/providername.mdx: id: aa3a4904b228 - last_write_checksum: sha1:01bc2dd9972a438f9bd28db3d933dedcc26dc96c - pristine_git_object: f57f9b678686a1b48fbd15f32c6cb702827f9212 + last_write_checksum: sha1:91965c05eec0a8240e5b536dc23be65208ec473d + pristine_git_object: ae094515bf037034d3d300afe607788271de1806 docs/components/provideroptions.mdx: id: 2697807524fe - last_write_checksum: sha1:b57db6c571573894305c48fc75594bcb0fa682ba - pristine_git_object: 88e2e2f88cdffd60442fedaafe781dedffc547cb + last_write_checksum: sha1:597f40b9e8a502f443aed4887c9d40a22f48493a + pristine_git_object: cb5a52ce9811c4300354a8476574d90fbe74c4bb docs/components/provideroverloadedresponseerrordata.mdx: id: dc500cd14889 last_write_checksum: sha1:8326118941c8899a398931c8c7871af21b716b00 @@ -5892,8 +5892,8 @@ trackedFiles: pristine_git_object: dd376f0952fa0517dd83b06ed7afc2c856123dbc docs/components/providerresponseprovidername.mdx: id: 2fc17b61e546 - last_write_checksum: sha1:c27298bb1e2b615f39d54f5a8653a2095914ab3a - pristine_git_object: 33179f4bf559857be94633cb3b461bb7e075d399 + last_write_checksum: sha1:639c58bee3bde7821c3f8817d9879346c6e58a80 + pristine_git_object: 092624b3273b04c5f7e80a11797bb13a85b21693 docs/components/providersort.mdx: id: d73fa13eeb95 last_write_checksum: sha1:b7e9e6842741e3433d705bd576f2ac5383f00134 @@ -6574,14 +6574,22 @@ trackedFiles: id: 2f508094a4f9 last_write_checksum: sha1:9fc321c80bf682f4eae2def394b0cf01cfae2feb pristine_git_object: 5ebe2d199734dd2729a8247420ae7e106a90e65a + docs/components/sttentity.mdx: + id: 7631aafebd82 + last_write_checksum: sha1:da5a1266aba4ce1383c1d1d1711da48970fa9557 + pristine_git_object: 70d1ffaf4915576b3df129947f541429b8520403 + docs/components/sttinlineinputaudio.mdx: + id: e4e2a7df033f + last_write_checksum: sha1:6c593d6d89f3c60b2ccd07943781fb9e4bc12fa3 + pristine_git_object: dfcdc6b10b6d784931adb140ac6808eef83d075a docs/components/sttinputaudio.mdx: id: f4fc56cf1641 - last_write_checksum: sha1:816a73a74abbff305ca8405d04870595d525b7bd - pristine_git_object: 70a2660b7c0bc35aed96124e1faa3852b3117054 + last_write_checksum: sha1:8dda73e1c8b42654b56a61e458cb7aca4d0836c3 + pristine_git_object: 45d1eef7be13575aac15391f23c310480852abf8 docs/components/sttrequest.mdx: id: 6a5e33cf6dbe - last_write_checksum: sha1:7cb7650a4ac48f68a357c211ff45f660c9741ddb - pristine_git_object: 3f4f75894aa5eb09c7490818c38f28260a07a824 + last_write_checksum: sha1:d15f6c0a8e2fc792b37668f55807879cf446586f + pristine_git_object: 49ab89dfaa79e89ab50bfa71b815a8139303a004 docs/components/sttrequestprovider.mdx: id: d8b1bac8e745 last_write_checksum: sha1:e15551eddbe4c1f7c4515fa556d4873075512eea @@ -6592,24 +6600,32 @@ trackedFiles: pristine_git_object: d016021a79783dfa11b1110edc231f49f45df4d0 docs/components/sttresponse.mdx: id: a421f69065a0 - last_write_checksum: sha1:25440205f1688ffcc95c14ff3d3b850079d9c828 - pristine_git_object: cc34241b8d783d307b5f4bd7ea96580810f6debc + last_write_checksum: sha1:787247cc3993e22dc456b3487cc3c2950e66accd + pristine_git_object: 8cc1932650e93f44d5a9d75e1c2b54ef31802833 docs/components/sttsegment.mdx: id: 7264ea9cdd57 - last_write_checksum: sha1:a72a8567ac80b37f20db0e483de5b1c96d45e5fe - pristine_git_object: 78744c25e6714a49518190e39107c59720c9b58d + last_write_checksum: sha1:bc6dae2fcd2689f157f36c63181b2e7064eaa3f1 + pristine_git_object: dbe1e88ec3093a8cda896b9eab8d62479ff8dd73 docs/components/stttimestampgranularity.mdx: id: 2a365c57c3a1 last_write_checksum: sha1:5932b36c214d97201f0cce0d0787203fed2baf4e pristine_git_object: 3505a518e362a11e06a3c098eb31de2718ee1c04 + docs/components/stturlinputaudio.mdx: + id: f267f2b95d48 + last_write_checksum: sha1:56d42f162b37da75d6b925ae779e6e039294bf20 + pristine_git_object: bc1eb6aefad7e21a9cef84607ab42454d9059c3e docs/components/sttusage.mdx: id: 11bce3b257f6 last_write_checksum: sha1:a777bc390f180c1163612213633f2150a9566313 pristine_git_object: f0b56839c067ce29024888f689d379ed83f86ff4 docs/components/sttword.mdx: id: 665227ec6052 - last_write_checksum: sha1:e299fae37efb17ab322fdc3da5513a6d1798f361 - pristine_git_object: bccae8c6fefa5a1be61bf78a6bd6471afc56dde1 + last_write_checksum: sha1:9dc2238ce02f30caf816365e0bd24cf4ca8aea1a + pristine_git_object: 48e877265107a6500ab627da050d667758a4f254 + docs/components/sttwordtype.mdx: + id: f7dc430a99ee + last_write_checksum: sha1:8c400ba5d1640a195532dd47f20d10346f350b17 + pristine_git_object: 76b0577c9e3e61bda3c499ca130526d9bcd9b241 docs/components/subagentnestedtool.mdx: id: 12547a9efd5f last_write_checksum: sha1:60f49c10d3b5c552a65ef4462d24e02a615cbaa8 @@ -7292,8 +7308,8 @@ trackedFiles: pristine_git_object: 67f56887bb4758559189e3e9e6123dff9a38df81 docs/components/videogenerationrequestoptions.mdx: id: 1d67160e7e7a - last_write_checksum: sha1:a21e4218516ce8bf9c1123da4df0036dcbf6e2db - pristine_git_object: adceddc6c91134f4e9ad845cc49594a22b6ac44a + last_write_checksum: sha1:b2bb732b57c2ba59422496fa304be8fede59df60 + pristine_git_object: 66923241831cc3e827a20c4b55e6706ded5837e7 docs/components/videogenerationrequestprovider.mdx: id: d6424dd5de9d last_write_checksum: sha1:72aa46a3335220b579db56574ea179d150dbe2b6 @@ -7744,8 +7760,8 @@ trackedFiles: pristine_git_object: b5031668ec1156a4157609e4812b7a3937b802a2 docs/operations/createaudiotranscriptionsmultipartrequestbody.mdx: id: 0550cc56c73a - last_write_checksum: sha1:45ce0af51dcacc830616021ab7c8288373983a01 - pristine_git_object: 654d92baa9e84a3f25074f605c7b28654eb5adbc + last_write_checksum: sha1:5f488a3854feb0fb865ce68b50b39a735a3bd94c + pristine_git_object: 3c8affe78a49e03bde0b56b5f7865daa3e88a8b3 docs/operations/createaudiotranscriptionsrequest.mdx: id: 68a813383936 last_write_checksum: sha1:a3ad7e4691917927539d6c614788da7f4d6b4892 @@ -9032,8 +9048,8 @@ trackedFiles: pristine_git_object: 0f0b54e43577a8a835cd971f82e48c3921e3d88e docs/operations/provider.mdx: id: c75b3bbeeaa0 - last_write_checksum: sha1:f64dd5aad93145dbf88089d9a9a8edaf1f85d3ca - pristine_git_object: dae7c717a3ab34fd9593d479c0e9bb5ec787a698 + last_write_checksum: sha1:3c98c4d2b39bec0811d85d0b980d6a309ec5031c + pristine_git_object: 0129bf418ce306edf69d5de64cab51b1b0bd8b57 docs/operations/provisioninternglobals.mdx: id: 5e34eb9c579a last_write_checksum: sha1:e5ddc9680dd8f7c0bbec6bb900f0754c8d8da955 @@ -9436,8 +9452,8 @@ trackedFiles: pristine_git_object: 240f5da9d86fc9dbbaa7efe345da740768c773b3 docs/sdks/stt/README.mdx: id: 190b0dc9a5d1 - last_write_checksum: sha1:9a13f8f1e26550bccf3935a36e9c3426febc6fcb - pristine_git_object: 1a12d8375fe6ddf9ef6ee1911ce0ed6831a16942 + last_write_checksum: sha1:a01c41c73ba10dc2ae69a7f1b404551721312c7d + pristine_git_object: c4b66757c280205b008221b84356d92015974ec7 docs/sdks/systemone/README.mdx: id: 55c317c71f3d last_write_checksum: sha1:b1ad82361fe1dbef6bedeead3783c75ec36283b6 @@ -9464,8 +9480,8 @@ trackedFiles: pristine_git_object: 3e38f1a929f7d6b1d6de74604aa87e3d8f010544 pyproject.toml: id: 5d07e7d72637 - last_write_checksum: sha1:c605659c8eb3c334251d12b44a34c888d1b440ff - pristine_git_object: df47bedc608b76a01be2c40c8d4c1c00b4052ac8 + last_write_checksum: sha1:41d2608f056d9be9baf7e5e037ea0343091c42cd + pristine_git_object: f830583b6813b44c8701c21828f86c23d0769854 scripts/prepare_readme.py: id: e0c5957a6035 last_write_checksum: sha1:77f44b60b98bc126557ec27391f91dfba764bb54 @@ -9492,8 +9508,8 @@ trackedFiles: pristine_git_object: 86713cfea633e09d33b3d4e65281071fe20e6137 src/openrouter/_version.py: id: d8d15ad6c586 - last_write_checksum: sha1:bcd746841358a883683aaee5218212a4a643f4e0 - pristine_git_object: f99b6685971cc9b713494bc45236782fbef88d9c + last_write_checksum: sha1:becbef69a8ff6acced5c232de9d26cc6749275f6 + pristine_git_object: 6e1c5673a8db2ca66ca5b2ed2bf695f03c1b1944 src/openrouter/alpha.py: id: 306c4d93308d last_write_checksum: sha1:30f55a360f41376ab194b9ea725fe4e001a5ae1a @@ -9540,8 +9556,8 @@ trackedFiles: pristine_git_object: ad3d247954547814054c01989a2dff3d12b3e4e1 src/openrouter/components/__init__.py: id: 81754e97b3f4 - last_write_checksum: sha1:87da07af0941e246f7369f761ef79dd407c934a2 - pristine_git_object: 26f2d3ddf7feb78661aa36786af1c39a08c92098 + last_write_checksum: sha1:031f8a9dc89c46b53f83b4848cc479232b2fc07e + pristine_git_object: d95413f20103ac38ac4c5bd6c304f29507b28d83 src/openrouter/components/aabenchmarkentry.py: id: e2e0f0b48c82 last_write_checksum: sha1:fab4d9a24d2cea937bb749d46c5f83941e99d65c @@ -10192,8 +10208,8 @@ trackedFiles: pristine_git_object: 22f7dfead05fc7561df8b836371fd7a2c83188d6 src/openrouter/components/byokproviderslug.py: id: 6738a8516caf - last_write_checksum: sha1:ddb03780cfa10727927fac58a4fd1a98dcdb190c - pristine_git_object: 8fe1b36c8fbd04b2726d9bb2c2703eb6a859a603 + last_write_checksum: sha1:019fc4c14dcb066768ed1b5f8d7e023425f72e6d + pristine_git_object: 47154c651c40406f0342cb5fc764d65d6b569fe7 src/openrouter/components/capabilitydescriptor.py: id: 368790a86f10 last_write_checksum: sha1:a167551924ee82b4b5876a0ab698a45fe8c87a95 @@ -10980,8 +10996,8 @@ trackedFiles: pristine_git_object: 6b9a8990973466285f9b49b366bbc5fc1ae42e41 src/openrouter/components/imagegenerationproviderpreferences.py: id: 100a82a54e2d - last_write_checksum: sha1:7ea34f90d099afe1382b7d6f811704b589b45dc5 - pristine_git_object: c9f6cc2de39d5ba28d67ca280ebef4748d549c1b + last_write_checksum: sha1:32017d2e8858fd0874f6825184c00e1f41b68a7b + pristine_git_object: 07eb905283ae4a5c716755ae8a636b1c6896579f src/openrouter/components/imagegenerationrequest.py: id: f8291a7be9a0 last_write_checksum: sha1:43396d2e7f0b98b79ed93e34cfe8d23bbb23640f @@ -11976,12 +11992,12 @@ trackedFiles: pristine_git_object: 170ae25c582bb8decd74e1c69bfa8353bd44e86d src/openrouter/components/providername.py: id: fcc722fa2fce - last_write_checksum: sha1:1a41b261891c7758e63141e61aed88b4a2da35ef - pristine_git_object: 480e20a7e2e2f8e5c18fe2a192fee982451015e1 + last_write_checksum: sha1:8342875d36174090e8aa114dac860d860759c838 + pristine_git_object: 1ea21048c5819904e606a065e3096cd10d81cec5 src/openrouter/components/provideroptions.py: id: 73dde6c8f359 - last_write_checksum: sha1:e1e411d68347dbcf6502a10f1115017199534576 - pristine_git_object: e8208511dabf44bfc9cfda47d7776decf91a172f + last_write_checksum: sha1:3c57d46f6291fa37a9a9376adbc1e6071c266fba + pristine_git_object: 57232d4d4008a373ad07d26563c540538c5fbe3a src/openrouter/components/provideroverloadedresponseerrordata.py: id: 5b693682570e last_write_checksum: sha1:b54682560ff95dad18d26f9aaf887f8f6e0e1f71 @@ -11992,8 +12008,8 @@ trackedFiles: pristine_git_object: 6cf5d428899c8722a49aa4ed5918ffe95b793549 src/openrouter/components/providerresponse.py: id: ad3887be54c5 - last_write_checksum: sha1:dda929c67b6f4125711efe155bbd94d259af1556 - pristine_git_object: 2f04468a84eefe6e42a1bf1f92ae739e9c377aca + last_write_checksum: sha1:66a7c5746d98e7fc89cf21c71f96eca6d59a17ce + pristine_git_object: 9918b5604df2e083a4e52e631eee90e0c4da32a9 src/openrouter/components/providersort.py: id: 348e382bf494 last_write_checksum: sha1:57551507f95cd2e16ef995e1c13f859fd0726152 @@ -12334,34 +12350,46 @@ trackedFiles: id: e39bb8998edb last_write_checksum: sha1:aae5b69a7dc7ced3af224d28b6ab485ad33ff2ef pristine_git_object: 79b6d3487431628f261a66e552bd97668ded59a2 + src/openrouter/components/sttentity.py: + id: a0df8a8395ea + last_write_checksum: sha1:1acd9736217b336bd5e87f21401cd733b74585a9 + pristine_git_object: 84ac78c76fae8a637c7b0d8d3868784302e1ca65 + src/openrouter/components/sttinlineinputaudio.py: + id: 231110dbface + last_write_checksum: sha1:3b614994056762fa40cc45ac5bd6d2d6a3c1c5a2 + pristine_git_object: 97e46c7ab81e308c55d4a8a844dc4b9074e6b55c src/openrouter/components/sttinputaudio.py: id: d59edf301037 - last_write_checksum: sha1:758dcb0aa6ccf3f2dd73b8d8d2771d339a2fecc2 - pristine_git_object: 76c4b466e22e57b34098b587e95cb358089dd87c + last_write_checksum: sha1:64acb2020058419f44c38accfed689fdf370f9ed + pristine_git_object: 75281f477b0ce85cd1032ed4b0c949a30a51b7f9 src/openrouter/components/sttrequest.py: id: 5fb1d469e16e - last_write_checksum: sha1:47753d76a574b1c421e7a28da152079e127a606f - pristine_git_object: 44c22985654d81c37b3ab0ac764510dafbc8bd80 + last_write_checksum: sha1:84b196eb8bfb9eac3c35879bcd28602ce666ff70 + pristine_git_object: 452ad0871db45705ff25bdf582e4dd245c2de4b8 src/openrouter/components/sttresponse.py: id: 2dc8eb8daaca - last_write_checksum: sha1:e5ccf0c70fbf8e6cf38aea6b90eee0599c533aa4 - pristine_git_object: 6ea5f83a1f12cce9cf96e67d0b0894ed5fda8a11 + last_write_checksum: sha1:420c65736ba0db3da84592d5639b41b81b6f87c4 + pristine_git_object: 2132dfe06d0a356e6215570a13099678ec0b94f3 src/openrouter/components/sttsegment.py: id: f3439539736c - last_write_checksum: sha1:9b19b8c0becf911a2939c28314d999b4e3efcf14 - pristine_git_object: 024e8a332cd8604dcc81c035121140716104e26e + last_write_checksum: sha1:1fc94db16341c2af2fd25a38f8953f504d4554a5 + pristine_git_object: f17dccc69716222c5d6edfc9d0bfcf48340b264c src/openrouter/components/stttimestampgranularity.py: id: d6f8b068026d last_write_checksum: sha1:9d1aff48024b9ab1446a05784f9aeb33aebdcee4 pristine_git_object: 30edd2ffd22cad78e4950c8751b8de3314909250 + src/openrouter/components/stturlinputaudio.py: + id: 0f7d3b4abba6 + last_write_checksum: sha1:499473a82d8f2ea0c65957a3aa43e189263750c9 + pristine_git_object: 5d7fcefdb8229786fa12bf6174fcc3ef78eb7d6f src/openrouter/components/sttusage.py: id: abe8329af787 last_write_checksum: sha1:4b7b79be71c2d1e392aef738a37d2d75f0bb29c4 pristine_git_object: 1feb5db4da1422d55744c5a4bb10eabda6c50c4f src/openrouter/components/sttword.py: id: b46ff7c5fa20 - last_write_checksum: sha1:8d01ff3a3b758053a9b25a08c76954558afa675c - pristine_git_object: d7e59a3e40d74540194515546d14a8ae0de8898e + last_write_checksum: sha1:bdf2782fe9d71628baa5bfb1b47e99800bd0bc8d + pristine_git_object: 2f6c9e97b357af4d2844b543867156ad9b7f1193 src/openrouter/components/subagentnestedtool.py: id: 911a465973a0 last_write_checksum: sha1:6aff7a7050393855944bbf85d437800336d12215 @@ -12608,8 +12636,8 @@ trackedFiles: pristine_git_object: b4a3db2fc4c7edf92a475be2952f01b5e2070205 src/openrouter/components/videogenerationrequest.py: id: 70e3c9ff288c - last_write_checksum: sha1:1e4c4016eea9f0dee474ce3b819ba2daf6aa5a04 - pristine_git_object: 026c12a1a17b9c39b70fd71ffa0133de5666fefd + last_write_checksum: sha1:eaa69f6486f1b33f006eea429aec726446396c29 + pristine_git_object: f2e96fc6eeca4714ec87fab130170cd110662cda src/openrouter/components/videogenerationresponse.py: id: 541f1321b072 last_write_checksum: sha1:945a98a9738528de20d5edde41c9dd9800d2b547 @@ -12964,8 +12992,8 @@ trackedFiles: pristine_git_object: 4e8414f223370bff1e4f060919f36f70da6ec40c src/openrouter/operations/createaudiotranscriptions_multipart.py: id: da07d463225a - last_write_checksum: sha1:a0d6838eba0635f3057db76527784340e1bf9743 - pristine_git_object: 30c86b20152166e13b91cd3d4bfd1082675702c8 + last_write_checksum: sha1:d7909d045c4116267c398ac0c735037b382c78bc + pristine_git_object: 91eaa7501c49f081ae44ba9d0c307f46b2630e43 src/openrouter/operations/createauthkeyscode.py: id: 4253a437de22 last_write_checksum: sha1:8f496fd0e965aba76a0882a3af0382649d0f8c4d @@ -13260,8 +13288,8 @@ trackedFiles: pristine_git_object: 6f60496e1a42385ab2253a36dfcd7f3698d5ae7a src/openrouter/operations/listbyokkeys.py: id: b6dd42b3e05f - last_write_checksum: sha1:52ce5b52f1f8f35dd39efb7502bf7b617d2cec27 - pristine_git_object: e15fe3efbc942ac48f958c92ad7a566231f5da98 + last_write_checksum: sha1:87e0d498821c5aacb790c2dcb0689ed6bebb4c74 + pristine_git_object: 3204af92c6d4ed1d987070c80eb6739ac9d5b5de src/openrouter/operations/listcontainerfiles.py: id: 7a30b9291c76 last_write_checksum: sha1:c51c428703923d69ea0fd6178d6313314e7b0ee9 @@ -13516,8 +13544,8 @@ trackedFiles: pristine_git_object: 26433165a0b1c25d7bd3115b9b129b18f2dc9387 src/openrouter/stt.py: id: fc0c2f669423 - last_write_checksum: sha1:294961a1d783e04ac54c3733e4563494e241fcd7 - pristine_git_object: 7502cb976473e1324ca78330ed0f22505d01c925 + last_write_checksum: sha1:2bd6bfa152883e3b95fbafb097fe77156d89379f + pristine_git_object: d368541ca6c3dc15e7adf5925db8d698afd93ad6 src/openrouter/systemone.py: id: 7e0a0e237bb1 last_write_checksum: sha1:fffd7413e70f4840b793888763631b51929eb766 @@ -14450,7 +14478,7 @@ examples: application/json: {"error": {"code": 504, "message": "The operation was aborted due to timeout"}} speakeasy-default-create-audio-transcriptions-multipart: requestBody: - multipart/form-data: {"file": "x-file: example.file", "model": "Escalade"} + multipart/form-data: {"model": "Escalade"} responses: "200": application/json: {"text": "Hello, this is a test of OpenAI speech-to-text transcription."} @@ -17378,4 +17406,4 @@ examples: "504": application/json: {"error": {"code": 504, "message": "Vault request timed out"}} examplesVersion: 1.0.2 -releaseNotes: "## Python SDK Changes:\n* `open_router.chat.send()`: \n * `request.plugins[].union(jev-router).allowed_models` **Added**\n* `open_router.presets.create_presets_chat_completions()`: \n * `request.plugins[].union(jev-router).allowed_models` **Added**\n* `open_router.presets.create_presets_messages()`: \n * `request.plugins[].union(jev-router).allowed_models` **Added**\n* `open_router.presets.create_presets_responses()`: \n * `request.plugins[].union(jev-router).allowed_models` **Added**\n* `open_router.responses.send()`: \n * `request.plugins[].union(jev-router).allowed_models` **Added**\n* `open_router.beta.responses.send()`: \n * `request.plugins[].union(jev-router).allowed_models` **Added**\n" +releaseNotes: "## Python SDK Changes:\n* `open_router.stt.create_transcription()`: \n * `request` **Changed** (Breaking ⚠️)\n * `response` **Changed**\n* `open_router.embeddings.generate()`: `request.provider` **Changed**\n* `open_router.endpoints.list()`: `response.data.endpoints[].provider_name.enum(eleven_labs)` **Added**\n* `open_router.stt.create_transcription_multipart()`: \n * `request` **Changed**\n * `response` **Changed**\n* `open_router.batch.create_batches()`: \n * `request.provider.only[].union(ProviderName).enum(eleven_labs)` **Added**\n* `open_router.byok.list()`: \n * `request.provider` **Changed**\n * `response.data[].provider.enum(elevenlabs)` **Added**\n* `open_router.byok.create()`: \n * `request.provider.enum(elevenlabs)` **Added**\n * `response.data.provider.enum(elevenlabs)` **Added**\n* `open_router.byok.get()`: `response.data.provider.enum(elevenlabs)` **Added**\n* `open_router.byok.update()`: `response.data.provider.enum(elevenlabs)` **Added**\n* `open_router.chat.send()`: `request.provider` **Changed**\n* `open_router.generations.get_generation()`: `response.data.provider_responses[].provider_name.enum(eleven_labs)` **Added**\n* `open_router.tts.create_speech()`: \n * `request.provider.options.elevenlabs` **Added**\n* `open_router.endpoints.list_zdr_endpoints()`: `response.data[].provider_name.enum(eleven_labs)` **Added**\n* `open_router.alpha.decisions.create()`: `request.provider` **Changed**\n* `open_router.images.generate()`: `request.provider` **Changed**\n* `open_router.presets.create_presets_chat_completions()`: `request.provider` **Changed**\n* `open_router.presets.create_presets_messages()`: `request.provider` **Changed**\n* `open_router.presets.create_presets_responses()`: `request.provider` **Changed**\n* `open_router.rerank.rerank()`: `request.provider` **Changed**\n* `open_router.responses.send()`: `request.provider` **Changed**\n* `open_router.beta.responses.send()`: `request.provider` **Changed**\n* `open_router.system_one.create()`: `request.provider` **Changed**\n* `open_router.video_generation.generate()`: \n * `request.provider.options.elevenlabs` **Added**\n" diff --git a/.speakeasy/gen.yaml b/.speakeasy/gen.yaml index 5e0f4887..0ad5daf2 100644 --- a/.speakeasy/gen.yaml +++ b/.speakeasy/gen.yaml @@ -36,7 +36,7 @@ generation: documentation: mintlify preApplyUnionDiscriminators: true python: - version: 1.3.5 + version: 1.3.6 additionalDependencies: dev: {} main: {} diff --git a/.speakeasy/out.openapi.yaml b/.speakeasy/out.openapi.yaml index 9ab41f00..4239932f 100644 --- a/.speakeasy/out.openapi.yaml +++ b/.speakeasy/out.openapi.yaml @@ -6252,6 +6252,7 @@ components: - 'deepseek' - 'dekallm' - 'digitalocean' + - 'elevenlabs' - 'featherless' - 'fireworks' - 'fish-audio' @@ -17441,6 +17442,7 @@ components: - 'DeepSeek' - 'DekaLLM' - 'DigitalOcean' + - 'ElevenLabs' - 'Featherless' - 'Fireworks' - 'Fish Audio' @@ -24972,6 +24974,7 @@ components: - 'DeepSeek' - 'DekaLLM' - 'DigitalOcean' + - 'ElevenLabs' - 'Featherless' - 'Fireworks' - 'Fish Audio' @@ -25186,6 +25189,9 @@ components: digitalocean: additionalProperties: {} type: 'object' + elevenlabs: + additionalProperties: {} + type: 'object' enfer: additionalProperties: {} type: 'object' @@ -25750,6 +25756,7 @@ components: - 'DeepSeek' - 'DekaLLM' - 'DigitalOcean' + - 'ElevenLabs' - 'Featherless' - 'Fireworks' - 'Fish Audio' @@ -28518,8 +28525,38 @@ components: - 111 logprob: -0.5 token: 'Hello' - STTInputAudio: - description: 'Base64-encoded audio to transcribe' + STTEntity: + description: 'A detected entity, returned when the provider runs entity detection' + example: + end_char: 25 + start_char: 15 + text: 'John Smith' + type: 'name' + properties: + end_char: + description: 'Zero-based exclusive character offset of the entity end within the response-level text (not seconds)' + example: 25 + type: 'integer' + start_char: + description: 'Zero-based character offset of the entity start within the response-level text (not seconds)' + example: 15 + type: 'integer' + text: + description: 'Entity text as it appears in the transcript' + type: 'string' + type: + description: 'Provider entity type label' + example: 'name' + type: 'string' + required: + - 'text' + - 'type' + - 'start_char' + - 'end_char' + type: 'object' + STTInlineInputAudio: + additionalProperties: false + description: 'Inline base64 audio input for speech-to-text' example: data: 'UklGRiQA...' format: 'wav' @@ -28528,15 +28565,20 @@ components: description: 'Base64-encoded audio data (raw bytes, not a data URI)' type: 'string' format: - description: 'Audio format (e.g., wav, mp3, flac, m4a, ogg, webm, aac). Supported formats vary by provider.' + description: 'Audio format (e.g., wav, mp3, flac, m4a, ogg, webm, aac). Supported formats vary by provider. "pcm" means headerless signed 16-bit little-endian mono audio at 16 kHz.' pattern: '^[a-zA-Z0-9][a-zA-Z0-9+._-]{0,15}$' type: 'string' required: - 'data' - 'format' type: 'object' + STTInputAudio: + anyOf: + - $ref: '#/components/schemas/STTInlineInputAudio' + - $ref: '#/components/schemas/STTUrlInputAudio' + description: 'Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly.' STTRequest: - description: 'Speech-to-text request input. Accepts a JSON body with input_audio containing base64-encoded audio.' + description: 'Speech-to-text request input. Accepts a JSON body with input_audio containing base64-encoded audio or a URL the provider downloads.' example: input_audio: data: 'UklGRiQA...' @@ -28544,8 +28586,23 @@ components: language: 'en' model: 'openai/whisper-large-v3' properties: + diarize: + description: 'Label each word with the speaker who said it. Speaker labels are returned on the words array (speaker, speaker_label), so response_format must be "verbose_json" (a "json" request is rejected with a 400) and word timestamps are included even when timestamp_granularities omits "word". Only supported by some providers; the request is rejected with a 400 when the selected model cannot diarize. Providers may charge extra.' + example: true + type: 'boolean' input_audio: $ref: '#/components/schemas/STTInputAudio' + keyterms: + description: 'Domain terms, names, or phrases to bias recognition toward. Only supported by some providers; the request is rejected with a 400 when the selected model cannot use keyterms. Providers may cap the number of terms or characters per term and may charge extra.' + example: + - 'OpenRouter' + - 'Scribe' + items: + maxLength: 100 + minLength: 1 + type: 'string' + maxItems: 1000 + type: 'array' language: description: 'ISO-639-1 language code (e.g., "en", "ja"). Auto-detected if omitted.' example: 'en' @@ -28617,10 +28674,20 @@ components: example: 9.2 format: 'double' type: 'number' + entities: + description: 'Detected entities with character offsets into text, present when the provider runs entity detection' + items: + $ref: '#/components/schemas/STTEntity' + type: 'array' language: description: 'Detected or forced language, present when response_format is verbose_json' example: 'english' type: 'string' + language_confidence: + description: 'Provider confidence in the detected language from 0 to 1, present when response_format is verbose_json and the provider scores language detection' + example: 0.98 + format: 'double' + type: 'number' segments: description: 'Timestamped transcript segments, present when response_format is verbose_json' items: @@ -28666,6 +28733,10 @@ components: description: 'Average log probability of the segment' format: 'double' type: 'number' + channel: + description: 'Zero-based audio channel index for the segment, present when the provider transcribes channels separately' + example: 0 + type: 'integer' compression_ratio: description: 'Compression ratio of the segment' format: 'double' @@ -28691,6 +28762,10 @@ components: description: 'Speaker index for the segment, present when the provider returns diarization data' example: 0 type: 'integer' + speaker_label: + description: 'Provider speaker label for the segment, present when the provider labels speakers with a string' + example: 'speaker_0' + type: 'string' start: description: 'Segment start time in seconds' example: 0 @@ -28723,6 +28798,25 @@ components: example: 'word' type: 'string' x-speakeasy-unknown-values: allow + STTUrlInputAudio: + additionalProperties: false + description: 'Audio input fetched by the provider from a URL' + example: + format: 'mp3' + url: 'https://example.com/meeting.mp3' + properties: + format: + description: 'Audio format of the file at the URL. Defaults to the extension of the URL path; required when the path has no extension.' + pattern: '^[a-zA-Z0-9][a-zA-Z0-9+._-]{0,15}$' + type: 'string' + url: + description: 'Publicly reachable http(s) URL of the audio file. The provider downloads it directly, so the inline upload size limit does not apply. Only supported by some providers.' + format: 'uri' + maxLength: 8000 + type: 'string' + required: + - 'url' + type: 'object' STTUsage: description: 'Aggregated usage statistics for the request' example: @@ -28764,6 +28858,10 @@ components: start: 0 word: 'Hello' properties: + channel: + description: 'Zero-based audio channel index for the word, present when the provider transcribes channels separately' + example: 0 + type: 'integer' confidence: description: 'Provider confidence for the word from 0 to 1, present when the provider returns per-word confidence' example: 0.98 @@ -28778,13 +28876,25 @@ components: description: 'Speaker index for the word, present when the provider returns diarization data' example: 0 type: 'integer' + speaker_label: + description: 'Provider speaker label for the word, present when the provider labels speakers with a string' + example: 'speaker_0' + type: 'string' start: description: 'Word start time in seconds' example: 0 format: 'double' type: 'number' + type: + description: 'Kind of entry; omitted or "word" for spoken words, "audio_event" for non-speech sounds the provider tags with timestamps' + enum: + - 'word' + - 'audio_event' + example: 'word' + type: 'string' + x-speakeasy-unknown-values: allow word: - description: 'The transcribed word' + description: 'The transcribed word, or the event tag such as "(laughter)" when type is audio_event' example: 'Hello' type: 'string' required: @@ -33201,7 +33311,7 @@ paths: - $ref: "#/components/parameters/AppCategories" /audio/transcriptions: post: - description: 'Transcribes audio into text. Accepts base64-encoded audio input as JSON or an OpenAI-style multipart/form-data file upload, and returns the transcribed text.' + description: 'Transcribes audio into text. Accepts base64-encoded audio input as JSON, an OpenAI-style multipart/form-data file upload, or a URL the provider downloads directly, and returns the transcribed text.' operationId: 'createAudioTranscriptions' requestBody: content: @@ -33221,16 +33331,27 @@ paths: model: 'openai/whisper-large-v3' schema: properties: + diarize: + description: 'Label each word with the speaker who said it (words[].speaker, words[].speaker_label). Requires response_format "verbose_json" (400 otherwise); word timestamps are included even when timestamp_granularities[] omits "word". Only supported by some providers; 400 when the selected model cannot diarize.' + type: 'boolean' file: - description: 'The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio.' + description: 'The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio, or by URL via source_url. Exactly one of file or source_url is required.' format: 'binary' type: 'string' + keyterms[]: + description: 'Domain terms, names, or phrases to bias recognition toward; repeat the part once per term (keyterms=... is also accepted). Only supported by some providers; 400 when the selected model cannot use keyterms.' + items: + type: 'string' + type: 'array' language: description: 'The language of the input audio (ISO-639-1).' type: 'string' model: description: 'The model to use for transcription.' type: 'string' + provider: + description: 'JSON-encoded provider preferences object, the same shape as the JSON body field: { "options": { "": { ... } } }. Only options for the matched provider are forwarded. Must decode to a JSON object.' + type: 'string' response_format: description: 'The response format. "json" (default) returns { text, usage }; "verbose_json" additionally returns task, language, duration, and segment-level timestamps (OpenAI-compatible providers only).' enum: @@ -33242,6 +33363,10 @@ paths: description: 'A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence.' maxLength: 256 type: 'string' + source_url: + description: 'Publicly reachable http(s) URL of the audio file, downloaded by the provider directly (no size limit on our side). The format is derived from the URL path extension. Only supported by some providers; exactly one of file or source_url is required.' + format: 'uri' + type: 'string' temperature: description: 'The sampling temperature.' type: 'number' @@ -33262,7 +33387,6 @@ paths: maxLength: 256 type: 'string' required: - - 'file' - 'model' type: 'object' required: true @@ -34461,6 +34585,7 @@ paths: - 'deepseek' - 'dekallm' - 'digitalocean' + - 'elevenlabs' - 'featherless' - 'fireworks' - 'fish-audio' diff --git a/.speakeasy/workflow.lock b/.speakeasy/workflow.lock index 659d618d..a2766631 100644 --- a/.speakeasy/workflow.lock +++ b/.speakeasy/workflow.lock @@ -2,8 +2,8 @@ speakeasyVersion: 1.787.0 sources: OpenRouter API: sourceNamespace: open-router-chat-completions-api - sourceRevisionDigest: sha256:a3666c04ea7f6d0bd1851a3961f51c8fef2cee3f8c6b1ded741777d1f2ebc088 - sourceBlobDigest: sha256:007c9b444c48e255ebb0cf0711d3c99659dee78aef07ec978d938635c05bd1cf + sourceRevisionDigest: sha256:fa682f646ac7fac9fd7baeae8f6356547792c4e8f0c14d7b471bb055db9e47e3 + sourceBlobDigest: sha256:5caef8a5438f249e60312655dc3aa58171d47772f0a3b24f23a745a84162bc82 tags: - latest - 1.0.0 @@ -11,10 +11,10 @@ targets: open-router: source: OpenRouter API sourceNamespace: open-router-chat-completions-api - sourceRevisionDigest: sha256:a3666c04ea7f6d0bd1851a3961f51c8fef2cee3f8c6b1ded741777d1f2ebc088 - sourceBlobDigest: sha256:007c9b444c48e255ebb0cf0711d3c99659dee78aef07ec978d938635c05bd1cf + sourceRevisionDigest: sha256:fa682f646ac7fac9fd7baeae8f6356547792c4e8f0c14d7b471bb055db9e47e3 + sourceBlobDigest: sha256:5caef8a5438f249e60312655dc3aa58171d47772f0a3b24f23a745a84162bc82 codeSamplesNamespace: open-router-python-code-samples - codeSamplesRevisionDigest: sha256:2f4bce7e2782ef7bc1a43785675f95b624c865c6c6cbc4989ea78802b2371e83 + codeSamplesRevisionDigest: sha256:fa882c2f6a037169bc675adc4611202adcf873d3290137352cf5de5be043495f workflow: workflowVersion: 1.0.0 speakeasyVersion: 1.787.0 diff --git a/README-PYPI.md b/README-PYPI.md index 01bc2a8f..baf0dde0 100644 --- a/README-PYPI.md +++ b/README-PYPI.md @@ -276,10 +276,10 @@ with OpenRouter( api_key=os.getenv("OPENROUTER_API_KEY", ""), ) as open_router: - res = open_router.stt.create_transcription_multipart(file={ + res = open_router.stt.create_transcription_multipart(model="openai/whisper-large-v3", file={ "file_name": "example.file", "content": open("example.file", "rb"), - }, model="openai/whisper-large-v3", language="en") + }, language="en") # Handle response print(res) diff --git a/README.md b/README.md index f696c00d..2384a839 100644 --- a/README.md +++ b/README.md @@ -276,10 +276,10 @@ with OpenRouter( api_key=os.getenv("OPENROUTER_API_KEY", ""), ) as open_router: - res = open_router.stt.create_transcription_multipart(file={ + res = open_router.stt.create_transcription_multipart(model="openai/whisper-large-v3", file={ "file_name": "example.file", "content": open("example.file", "rb"), - }, model="openai/whisper-large-v3", language="en") + }, language="en") # Handle response print(res) diff --git a/RELEASES.md b/RELEASES.md index bd7411e4..64a52a49 100644 --- a/RELEASES.md +++ b/RELEASES.md @@ -2819,4 +2819,14 @@ Based on: ### Generated - [python v1.3.5] . ### Releases -- [PyPI v1.3.5] https://pypi.org/project/openrouter/1.3.5 - . \ No newline at end of file +- [PyPI v1.3.5] https://pypi.org/project/openrouter/1.3.5 - . + +## 2026-09-29 19:06:13 +### Changes +Based on: +- OpenAPI Doc +- Speakeasy CLI 1.787.0 (2.914.0) https://github.com/speakeasy-api/speakeasy +### Generated +- [python v1.3.6] . +### Releases +- [PyPI v1.3.6] https://pypi.org/project/openrouter/1.3.6 - . \ No newline at end of file diff --git a/docs/components/byokproviderslug.mdx b/docs/components/byokproviderslug.mdx index 3ab00cd1..5d4b87a5 100644 --- a/docs/components/byokproviderslug.mdx +++ b/docs/components/byokproviderslug.mdx @@ -55,6 +55,7 @@ This is an open enum. Unrecognized values will not fail type checks. - `"deepseek"` - `"dekallm"` - `"digitalocean"` +- `"elevenlabs"` - `"featherless"` - `"fireworks"` - `"fish-audio"` diff --git a/docs/components/imagegenerationproviderpreferencesoptions.mdx b/docs/components/imagegenerationproviderpreferencesoptions.mdx index 5515aeac..815789c4 100644 --- a/docs/components/imagegenerationproviderpreferencesoptions.mdx +++ b/docs/components/imagegenerationproviderpreferencesoptions.mdx @@ -52,6 +52,7 @@ Provider-specific options keyed by provider slug. Only options for the matched p | `deepseek` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `dekallm` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `digitalocean` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | +| `elevenlabs` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `enfer` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `fake_provider` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `featherless` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | diff --git a/docs/components/providername.mdx b/docs/components/providername.mdx index f57f9b67..ae094515 100644 --- a/docs/components/providername.mdx +++ b/docs/components/providername.mdx @@ -53,6 +53,7 @@ This is an open enum. Unrecognized values will not fail type checks. - `"DeepSeek"` - `"DekaLLM"` - `"DigitalOcean"` +- `"ElevenLabs"` - `"Featherless"` - `"Fireworks"` - `"Fish Audio"` diff --git a/docs/components/provideroptions.mdx b/docs/components/provideroptions.mdx index 88e2e2f8..cb5a52ce 100644 --- a/docs/components/provideroptions.mdx +++ b/docs/components/provideroptions.mdx @@ -52,6 +52,7 @@ Provider-specific options keyed by provider slug. Only options for the matched p | `deepseek` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `dekallm` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `digitalocean` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | +| `elevenlabs` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `enfer` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `fake_provider` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `featherless` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | diff --git a/docs/components/providerresponseprovidername.mdx b/docs/components/providerresponseprovidername.mdx index 33179f4b..092624b3 100644 --- a/docs/components/providerresponseprovidername.mdx +++ b/docs/components/providerresponseprovidername.mdx @@ -83,6 +83,7 @@ This is an open enum. Unrecognized values will not fail type checks. - `"DeepSeek"` - `"DekaLLM"` - `"DigitalOcean"` +- `"ElevenLabs"` - `"Featherless"` - `"Fireworks"` - `"Fish Audio"` diff --git a/docs/components/sttentity.mdx b/docs/components/sttentity.mdx new file mode 100644 index 00000000..70d1ffaf --- /dev/null +++ b/docs/components/sttentity.mdx @@ -0,0 +1,15 @@ +--- +title: "STTEntity" +--- + +A detected entity, returned when the provider runs entity detection + + +## Fields + +| Field | Type | Required | Description | Example | +| ---------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------- | +| `end_char` | *int* | :heavy_check_mark: | Zero-based exclusive character offset of the entity end within the response-level text (not seconds) | 25 | +| `start_char` | *int* | :heavy_check_mark: | Zero-based character offset of the entity start within the response-level text (not seconds) | 15 | +| `text` | *str* | :heavy_check_mark: | Entity text as it appears in the transcript | | +| `type` | *str* | :heavy_check_mark: | Provider entity type label | name | \ No newline at end of file diff --git a/docs/components/sttinlineinputaudio.mdx b/docs/components/sttinlineinputaudio.mdx new file mode 100644 index 00000000..dfcdc6b1 --- /dev/null +++ b/docs/components/sttinlineinputaudio.mdx @@ -0,0 +1,13 @@ +--- +title: "STTInlineInputAudio" +--- + +Inline base64 audio input for speech-to-text + + +## Fields + +| Field | Type | Required | Description | +| ---------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `data` | *str* | :heavy_check_mark: | Base64-encoded audio data (raw bytes, not a data URI) | +| `format_` | *str* | :heavy_check_mark: | Audio format (e.g., wav, mp3, flac, m4a, ogg, webm, aac). Supported formats vary by provider. "pcm" means headerless signed 16-bit little-endian mono audio at 16 kHz. | \ No newline at end of file diff --git a/docs/components/sttinputaudio.mdx b/docs/components/sttinputaudio.mdx index 70a2660b..45d1eef7 100644 --- a/docs/components/sttinputaudio.mdx +++ b/docs/components/sttinputaudio.mdx @@ -2,12 +2,20 @@ title: "STTInputAudio" --- -Base64-encoded audio to transcribe +Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly. -## Fields +## Supported Types + +### `components.STTInlineInputAudio` + +```python +value: components.STTInlineInputAudio = /* values here */ +``` + +### `components.STTURLInputAudio` + +```python +value: components.STTURLInputAudio = /* values here */ +``` -| Field | Type | Required | Description | -| --------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------- | -| `data` | *str* | :heavy_check_mark: | Base64-encoded audio data (raw bytes, not a data URI) | -| `format_` | *str* | :heavy_check_mark: | Audio format (e.g., wav, mp3, flac, m4a, ogg, webm, aac). Supported formats vary by provider. | \ No newline at end of file diff --git a/docs/components/sttrequest.mdx b/docs/components/sttrequest.mdx index 3f4f7589..49ab89df 100644 --- a/docs/components/sttrequest.mdx +++ b/docs/components/sttrequest.mdx @@ -2,20 +2,22 @@ title: "STTRequest" --- -Speech-to-text request input. Accepts a JSON body with input_audio containing base64-encoded audio. +Speech-to-text request input. Accepts a JSON body with input_audio containing base64-encoded audio or a URL the provider downloads. ## Fields -| Field | Type | Required | Description | Example | -| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `input_audio` | [components.STTInputAudio](../components/sttinputaudio.mdx) | :heavy_check_mark: | Base64-encoded audio to transcribe | \{
"data": "UklGRiQA...",
"format": "wav"
} | -| `language` | *Optional[str]* | :heavy_minus_sign: | ISO-639-1 language code (e.g., "en", "ja"). Auto-detected if omitted. | en | -| `model` | *str* | :heavy_check_mark: | STT model identifier | openai/whisper-large-v3 | -| `provider` | [Optional[components.STTRequestProvider]](../components/sttrequestprovider.mdx) | :heavy_minus_sign: | Provider-specific passthrough configuration | | -| `response_format` | [Optional[components.STTRequestResponseFormat]](../components/sttrequestresponseformat.mdx) | :heavy_minus_sign: | Output format. "json" (default) returns \{ text, usage }. "verbose_json" additionally returns task, language, duration, and segment-level timestamps; only supported by OpenAI-compatible providers. | json | -| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | session-1234 | -| `temperature` | *Optional[float]* | :heavy_minus_sign: | Sampling temperature for transcription | 0 | -| `timestamp_granularities` | List[[components.STTTimestampGranularity](../components/stttimestampgranularity.mdx)] | :heavy_minus_sign: | Timestamp detail levels to include when response_format is "verbose_json". "segment" returns segment-level timestamps; "word" additionally returns word-level timestamps in the words array. Ignored unless response_format is "verbose_json". | [
"segment"
] | -| `trace` | [Optional[components.TraceConfig]](../components/traceconfig.mdx) | :heavy_minus_sign: | Metadata for observability and tracing. Known keys (trace_id, trace_name, span_name, generation_name, parent_span_id) have special handling. Additional keys are passed through as custom metadata to configured broadcast destinations. | \{
"trace_id": "trace-abc123",
"trace_name": "my-app-trace"
} | -| `user` | *Optional[str]* | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | user-1234 | \ No newline at end of file +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `diarize` | *Optional[bool]* | :heavy_minus_sign: | Label each word with the speaker who said it. Speaker labels are returned on the words array (speaker, speaker_label), so response_format must be "verbose_json" (a "json" request is rejected with a 400) and word timestamps are included even when timestamp_granularities omits "word". Only supported by some providers; the request is rejected with a 400 when the selected model cannot diarize. Providers may charge extra. | true | +| `input_audio` | [components.STTInputAudio](../components/sttinputaudio.mdx) | :heavy_check_mark: | Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly. | | +| `keyterms` | List[*str*] | :heavy_minus_sign: | Domain terms, names, or phrases to bias recognition toward. Only supported by some providers; the request is rejected with a 400 when the selected model cannot use keyterms. Providers may cap the number of terms or characters per term and may charge extra. | [
"OpenRouter",
"Scribe"
] | +| `language` | *Optional[str]* | :heavy_minus_sign: | ISO-639-1 language code (e.g., "en", "ja"). Auto-detected if omitted. | en | +| `model` | *str* | :heavy_check_mark: | STT model identifier | openai/whisper-large-v3 | +| `provider` | [Optional[components.STTRequestProvider]](../components/sttrequestprovider.mdx) | :heavy_minus_sign: | Provider-specific passthrough configuration | | +| `response_format` | [Optional[components.STTRequestResponseFormat]](../components/sttrequestresponseformat.mdx) | :heavy_minus_sign: | Output format. "json" (default) returns \{ text, usage }. "verbose_json" additionally returns task, language, duration, and segment-level timestamps; only supported by OpenAI-compatible providers. | json | +| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | session-1234 | +| `temperature` | *Optional[float]* | :heavy_minus_sign: | Sampling temperature for transcription | 0 | +| `timestamp_granularities` | List[[components.STTTimestampGranularity](../components/stttimestampgranularity.mdx)] | :heavy_minus_sign: | Timestamp detail levels to include when response_format is "verbose_json". "segment" returns segment-level timestamps; "word" additionally returns word-level timestamps in the words array. Ignored unless response_format is "verbose_json". | [
"segment"
] | +| `trace` | [Optional[components.TraceConfig]](../components/traceconfig.mdx) | :heavy_minus_sign: | Metadata for observability and tracing. Known keys (trace_id, trace_name, span_name, generation_name, parent_span_id) have special handling. Additional keys are passed through as custom metadata to configured broadcast destinations. | \{
"trace_id": "trace-abc123",
"trace_name": "my-app-trace"
} | +| `user` | *Optional[str]* | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | user-1234 | \ No newline at end of file diff --git a/docs/components/sttresponse.mdx b/docs/components/sttresponse.mdx index cc34241b..8cc19326 100644 --- a/docs/components/sttresponse.mdx +++ b/docs/components/sttresponse.mdx @@ -11,7 +11,9 @@ STT response containing transcribed text and optional usage statistics | -------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------- | | `confidence` | *Optional[float]* | :heavy_minus_sign: | Provider confidence for the whole transcript from 0 to 1, present when response_format is verbose_json and the provider scores the full transcript | 0.94 | | `duration` | *Optional[float]* | :heavy_minus_sign: | Duration of the input audio in seconds, present when response_format is verbose_json | 9.2 | +| `entities` | List[[components.STTEntity](../components/sttentity.mdx)] | :heavy_minus_sign: | Detected entities with character offsets into text, present when the provider runs entity detection | | | `language` | *Optional[str]* | :heavy_minus_sign: | Detected or forced language, present when response_format is verbose_json | english | +| `language_confidence` | *Optional[float]* | :heavy_minus_sign: | Provider confidence in the detected language from 0 to 1, present when response_format is verbose_json and the provider scores language detection | 0.98 | | `segments` | List[[components.STTSegment](../components/sttsegment.mdx)] | :heavy_minus_sign: | Timestamped transcript segments, present when response_format is verbose_json | | | `task` | *Optional[str]* | :heavy_minus_sign: | The task performed, present when response_format is verbose_json | transcribe | | `text` | *str* | :heavy_check_mark: | The transcribed text | Hello, this is a test of OpenAI speech-to-text transcription. The weather is sunny today and the temperature is around 72 degrees. | diff --git a/docs/components/sttsegment.mdx b/docs/components/sttsegment.mdx index 78744c25..dbe1e88e 100644 --- a/docs/components/sttsegment.mdx +++ b/docs/components/sttsegment.mdx @@ -7,16 +7,18 @@ A timestamped transcript segment, returned when response_format is verbose_json ## Fields -| Field | Type | Required | Description | Example | -| --------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- | -| `avg_logprob` | *Optional[float]* | :heavy_minus_sign: | Average log probability of the segment | | -| `compression_ratio` | *Optional[float]* | :heavy_minus_sign: | Compression ratio of the segment | | -| `end` | *float* | :heavy_check_mark: | Segment end time in seconds | 3.2 | -| `id` | *int* | :heavy_check_mark: | Segment index within the transcript | 0 | -| `no_speech_prob` | *Optional[float]* | :heavy_minus_sign: | Probability the segment contains no speech | | -| `seek` | *Optional[int]* | :heavy_minus_sign: | Seek offset of the segment | 0 | -| `speaker` | *Optional[int]* | :heavy_minus_sign: | Speaker index for the segment, present when the provider returns diarization data | 0 | -| `start` | *float* | :heavy_check_mark: | Segment start time in seconds | 0 | -| `temperature` | *Optional[float]* | :heavy_minus_sign: | Temperature used for the segment | | -| `text` | *str* | :heavy_check_mark: | Transcribed text of the segment | Hello there. | -| `tokens` | List[*int*] | :heavy_minus_sign: | Token IDs of the segment | | \ No newline at end of file +| Field | Type | Required | Description | Example | +| --------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------- | +| `avg_logprob` | *Optional[float]* | :heavy_minus_sign: | Average log probability of the segment | | +| `channel` | *Optional[int]* | :heavy_minus_sign: | Zero-based audio channel index for the segment, present when the provider transcribes channels separately | 0 | +| `compression_ratio` | *Optional[float]* | :heavy_minus_sign: | Compression ratio of the segment | | +| `end` | *float* | :heavy_check_mark: | Segment end time in seconds | 3.2 | +| `id` | *int* | :heavy_check_mark: | Segment index within the transcript | 0 | +| `no_speech_prob` | *Optional[float]* | :heavy_minus_sign: | Probability the segment contains no speech | | +| `seek` | *Optional[int]* | :heavy_minus_sign: | Seek offset of the segment | 0 | +| `speaker` | *Optional[int]* | :heavy_minus_sign: | Speaker index for the segment, present when the provider returns diarization data | 0 | +| `speaker_label` | *Optional[str]* | :heavy_minus_sign: | Provider speaker label for the segment, present when the provider labels speakers with a string | speaker_0 | +| `start` | *float* | :heavy_check_mark: | Segment start time in seconds | 0 | +| `temperature` | *Optional[float]* | :heavy_minus_sign: | Temperature used for the segment | | +| `text` | *str* | :heavy_check_mark: | Transcribed text of the segment | Hello there. | +| `tokens` | List[*int*] | :heavy_minus_sign: | Token IDs of the segment | | \ No newline at end of file diff --git a/docs/components/stturlinputaudio.mdx b/docs/components/stturlinputaudio.mdx new file mode 100644 index 00000000..bc1eb6ae --- /dev/null +++ b/docs/components/stturlinputaudio.mdx @@ -0,0 +1,13 @@ +--- +title: "STTURLInputAudio" +--- + +Audio input fetched by the provider from a URL + + +## Fields + +| Field | Type | Required | Description | +| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `format_` | *Optional[str]* | :heavy_minus_sign: | Audio format of the file at the URL. Defaults to the extension of the URL path; required when the path has no extension. | +| `url` | *str* | :heavy_check_mark: | Publicly reachable http(s) URL of the audio file. The provider downloads it directly, so the inline upload size limit does not apply. Only supported by some providers. | \ No newline at end of file diff --git a/docs/components/sttword.mdx b/docs/components/sttword.mdx index bccae8c6..48e87726 100644 --- a/docs/components/sttword.mdx +++ b/docs/components/sttword.mdx @@ -7,10 +7,13 @@ A timestamped word, returned when the provider includes word-level timestamps ## Fields -| Field | Type | Required | Description | Example | -| --------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------- | -| `confidence` | *Optional[float]* | :heavy_minus_sign: | Provider confidence for the word from 0 to 1, present when the provider returns per-word confidence | 0.98 | -| `end` | *float* | :heavy_check_mark: | Word end time in seconds | 0.4 | -| `speaker` | *Optional[int]* | :heavy_minus_sign: | Speaker index for the word, present when the provider returns diarization data | 0 | -| `start` | *float* | :heavy_check_mark: | Word start time in seconds | 0 | -| `word` | *str* | :heavy_check_mark: | The transcribed word | Hello | \ No newline at end of file +| Field | Type | Required | Description | Example | +| ------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------ | +| `channel` | *Optional[int]* | :heavy_minus_sign: | Zero-based audio channel index for the word, present when the provider transcribes channels separately | 0 | +| `confidence` | *Optional[float]* | :heavy_minus_sign: | Provider confidence for the word from 0 to 1, present when the provider returns per-word confidence | 0.98 | +| `end` | *float* | :heavy_check_mark: | Word end time in seconds | 0.4 | +| `speaker` | *Optional[int]* | :heavy_minus_sign: | Speaker index for the word, present when the provider returns diarization data | 0 | +| `speaker_label` | *Optional[str]* | :heavy_minus_sign: | Provider speaker label for the word, present when the provider labels speakers with a string | speaker_0 | +| `start` | *float* | :heavy_check_mark: | Word start time in seconds | 0 | +| `type` | [Optional[components.STTWordType]](../components/sttwordtype.mdx) | :heavy_minus_sign: | Kind of entry; omitted or "word" for spoken words, "audio_event" for non-speech sounds the provider tags with timestamps | word | +| `word` | *str* | :heavy_check_mark: | The transcribed word, or the event tag such as "(laughter)" when type is audio_event | Hello | \ No newline at end of file diff --git a/docs/components/sttwordtype.mdx b/docs/components/sttwordtype.mdx new file mode 100644 index 00000000..76b0577c --- /dev/null +++ b/docs/components/sttwordtype.mdx @@ -0,0 +1,22 @@ +--- +title: "STTWordType" +--- + +Kind of entry; omitted or "word" for spoken words, "audio_event" for non-speech sounds the provider tags with timestamps + +## Example Usage + +```python +from openrouter.components import STTWordType + +# Open enum: unrecognized values are captured as UnrecognizedStr +value: STTWordType = "word" +``` + + +## Values + +This is an open enum. Unrecognized values will not fail type checks. + +- `"word"` +- `"audio_event"` diff --git a/docs/components/videogenerationrequestoptions.mdx b/docs/components/videogenerationrequestoptions.mdx index adceddc6..66923241 100644 --- a/docs/components/videogenerationrequestoptions.mdx +++ b/docs/components/videogenerationrequestoptions.mdx @@ -52,6 +52,7 @@ Provider-specific options keyed by provider slug. Only options for the matched p | `deepseek` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `dekallm` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `digitalocean` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | +| `elevenlabs` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `enfer` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `fake_provider` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | | `featherless` | Dict[str, *Any*] | :heavy_minus_sign: | N/A | diff --git a/docs/operations/createaudiotranscriptionsmultipartrequestbody.mdx b/docs/operations/createaudiotranscriptionsmultipartrequestbody.mdx index 654d92ba..3c8affe7 100644 --- a/docs/operations/createaudiotranscriptionsmultipartrequestbody.mdx +++ b/docs/operations/createaudiotranscriptionsmultipartrequestbody.mdx @@ -4,14 +4,18 @@ title: "CreateAudioTranscriptionsMultipartRequestBody" ## Fields -| Field | Type | Required | Description | -| -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `file` | [operations.CreateAudioTranscriptionsMultipartFile](../operations/createaudiotranscriptionsmultipartfile.mdx) | :heavy_check_mark: | The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio. | -| `language` | *Optional[str]* | :heavy_minus_sign: | The language of the input audio (ISO-639-1). | -| `model` | *str* | :heavy_check_mark: | The model to use for transcription. | -| `response_format` | [Optional[operations.ResponseFormat]](../operations/responseformat.mdx) | :heavy_minus_sign: | The response format. "json" (default) returns \{ text, usage }; "verbose_json" additionally returns task, language, duration, and segment-level timestamps (OpenAI-compatible providers only). | -| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. | -| `temperature` | *Optional[float]* | :heavy_minus_sign: | The sampling temperature. | -| `timestamp_granularities` | List[[operations.TimestampGranularities](../operations/timestampgranularities.mdx)] | :heavy_minus_sign: | Timestamp detail levels to include when response_format is "verbose_json". "word" additionally returns word-level timestamps in the words array. | -| `trace` | *Optional[str]* | :heavy_minus_sign: | JSON-encoded trace metadata object (trace_id, trace_name, span_name, generation_name, parent_span_id and custom keys) attached to the Broadcast trace. Must decode to a JSON object. | -| `user` | *Optional[str]* | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | \ No newline at end of file +| Field | Type | Required | Description | +| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `diarize` | *Optional[bool]* | :heavy_minus_sign: | Label each word with the speaker who said it (words[].speaker, words[].speaker_label). Requires response_format "verbose_json" (400 otherwise); word timestamps are included even when timestamp_granularities[] omits "word". Only supported by some providers; 400 when the selected model cannot diarize. | +| `file` | [Optional[operations.CreateAudioTranscriptionsMultipartFile]](../operations/createaudiotranscriptionsmultipartfile.mdx) | :heavy_minus_sign: | The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio, or by URL via source_url. Exactly one of file or source_url is required. | +| `keyterms` | List[*str*] | :heavy_minus_sign: | Domain terms, names, or phrases to bias recognition toward; repeat the part once per term (keyterms=... is also accepted). Only supported by some providers; 400 when the selected model cannot use keyterms. | +| `language` | *Optional[str]* | :heavy_minus_sign: | The language of the input audio (ISO-639-1). | +| `model` | *str* | :heavy_check_mark: | The model to use for transcription. | +| `provider` | *Optional[str]* | :heavy_minus_sign: | JSON-encoded provider preferences object, the same shape as the JSON body field: \{ "options": \{ "\": \{ ... } } }. Only options for the matched provider are forwarded. Must decode to a JSON object. | +| `response_format` | [Optional[operations.ResponseFormat]](../operations/responseformat.mdx) | :heavy_minus_sign: | The response format. "json" (default) returns \{ text, usage }; "verbose_json" additionally returns task, language, duration, and segment-level timestamps (OpenAI-compatible providers only). | +| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. | +| `source_url` | *Optional[str]* | :heavy_minus_sign: | Publicly reachable http(s) URL of the audio file, downloaded by the provider directly (no size limit on our side). The format is derived from the URL path extension. Only supported by some providers; exactly one of file or source_url is required. | +| `temperature` | *Optional[float]* | :heavy_minus_sign: | The sampling temperature. | +| `timestamp_granularities` | List[[operations.TimestampGranularities](../operations/timestampgranularities.mdx)] | :heavy_minus_sign: | Timestamp detail levels to include when response_format is "verbose_json". "word" additionally returns word-level timestamps in the words array. | +| `trace` | *Optional[str]* | :heavy_minus_sign: | JSON-encoded trace metadata object (trace_id, trace_name, span_name, generation_name, parent_span_id and custom keys) attached to the Broadcast trace. Must decode to a JSON object. | +| `user` | *Optional[str]* | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | \ No newline at end of file diff --git a/docs/operations/provider.mdx b/docs/operations/provider.mdx index dae7c717..0129bf41 100644 --- a/docs/operations/provider.mdx +++ b/docs/operations/provider.mdx @@ -55,6 +55,7 @@ This is an open enum. Unrecognized values will not fail type checks. - `"deepseek"` - `"dekallm"` - `"digitalocean"` +- `"elevenlabs"` - `"featherless"` - `"fireworks"` - `"fish-audio"` diff --git a/docs/sdks/stt/README.mdx b/docs/sdks/stt/README.mdx index 1a12d837..c4b66757 100644 --- a/docs/sdks/stt/README.mdx +++ b/docs/sdks/stt/README.mdx @@ -14,7 +14,7 @@ Speech-to-text endpoints ## create_transcription -Transcribes audio into text. Accepts base64-encoded audio input as JSON or an OpenAI-style multipart/form-data file upload, and returns the transcribed text. +Transcribes audio into text. Accepts base64-encoded audio input as JSON, an OpenAI-style multipart/form-data file upload, or a URL the provider downloads directly, and returns the transcribed text. ### Example Usage @@ -42,22 +42,24 @@ with OpenRouter( ### Parameters -| Parameter | Type | Required | Description | Example | -| ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `input_audio` | [components.STTInputAudio](../../components/sttinputaudio.mdx) | :heavy_check_mark: | Base64-encoded audio to transcribe | \{
"data": "UklGRiQA...",
"format": "wav"
} | -| `model` | *str* | :heavy_check_mark: | STT model identifier | openai/whisper-large-v3 | -| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| | -| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| | -| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| | -| `language` | *Optional[str]* | :heavy_minus_sign: | ISO-639-1 language code (e.g., "en", "ja"). Auto-detected if omitted. | en | -| `provider` | [Optional[components.STTRequestProvider]](../../components/sttrequestprovider.mdx) | :heavy_minus_sign: | Provider-specific passthrough configuration | | -| `response_format` | [Optional[components.STTRequestResponseFormat]](../../components/sttrequestresponseformat.mdx) | :heavy_minus_sign: | Output format. "json" (default) returns \{ text, usage }. "verbose_json" additionally returns task, language, duration, and segment-level timestamps; only supported by OpenAI-compatible providers. | json | -| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | session-1234 | -| `temperature` | *Optional[float]* | :heavy_minus_sign: | Sampling temperature for transcription | 0 | -| `timestamp_granularities` | List[[components.STTTimestampGranularity](../../components/stttimestampgranularity.mdx)] | :heavy_minus_sign: | Timestamp detail levels to include when response_format is "verbose_json". "segment" returns segment-level timestamps; "word" additionally returns word-level timestamps in the words array. Ignored unless response_format is "verbose_json". | [
"segment"
] | -| `trace` | [Optional[components.TraceConfig]](../../components/traceconfig.mdx) | :heavy_minus_sign: | Metadata for observability and tracing. Known keys (trace_id, trace_name, span_name, generation_name, parent_span_id) have special handling. Additional keys are passed through as custom metadata to configured broadcast destinations. | \{
"trace_id": "trace-abc123",
"trace_name": "my-app-trace"
} | -| `user` | *Optional[str]* | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | user-1234 | -| `retries` | [Optional[utils.RetryConfig]](../../models/utils/retryconfig.mdx) | :heavy_minus_sign: | Configuration to override the default retry behavior of the client. | | +| Parameter | Type | Required | Description | Example | +| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `input_audio` | [components.STTInputAudio](../../components/sttinputaudio.mdx) | :heavy_check_mark: | Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly. | | +| `model` | *str* | :heavy_check_mark: | STT model identifier | openai/whisper-large-v3 | +| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| | +| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| | +| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| | +| `diarize` | *Optional[bool]* | :heavy_minus_sign: | Label each word with the speaker who said it. Speaker labels are returned on the words array (speaker, speaker_label), so response_format must be "verbose_json" (a "json" request is rejected with a 400) and word timestamps are included even when timestamp_granularities omits "word". Only supported by some providers; the request is rejected with a 400 when the selected model cannot diarize. Providers may charge extra. | true | +| `keyterms` | List[*str*] | :heavy_minus_sign: | Domain terms, names, or phrases to bias recognition toward. Only supported by some providers; the request is rejected with a 400 when the selected model cannot use keyterms. Providers may cap the number of terms or characters per term and may charge extra. | [
"OpenRouter",
"Scribe"
] | +| `language` | *Optional[str]* | :heavy_minus_sign: | ISO-639-1 language code (e.g., "en", "ja"). Auto-detected if omitted. | en | +| `provider` | [Optional[components.STTRequestProvider]](../../components/sttrequestprovider.mdx) | :heavy_minus_sign: | Provider-specific passthrough configuration | | +| `response_format` | [Optional[components.STTRequestResponseFormat]](../../components/sttrequestresponseformat.mdx) | :heavy_minus_sign: | Output format. "json" (default) returns \{ text, usage }. "verbose_json" additionally returns task, language, duration, and segment-level timestamps; only supported by OpenAI-compatible providers. | json | +| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. Maximum of 256 characters. | session-1234 | +| `temperature` | *Optional[float]* | :heavy_minus_sign: | Sampling temperature for transcription | 0 | +| `timestamp_granularities` | List[[components.STTTimestampGranularity](../../components/stttimestampgranularity.mdx)] | :heavy_minus_sign: | Timestamp detail levels to include when response_format is "verbose_json". "segment" returns segment-level timestamps; "word" additionally returns word-level timestamps in the words array. Ignored unless response_format is "verbose_json". | [
"segment"
] | +| `trace` | [Optional[components.TraceConfig]](../../components/traceconfig.mdx) | :heavy_minus_sign: | Metadata for observability and tracing. Known keys (trace_id, trace_name, span_name, generation_name, parent_span_id) have special handling. Additional keys are passed through as custom metadata to configured broadcast destinations. | \{
"trace_id": "trace-abc123",
"trace_name": "my-app-trace"
} | +| `user` | *Optional[str]* | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | user-1234 | +| `retries` | [Optional[utils.RetryConfig]](../../models/utils/retryconfig.mdx) | :heavy_minus_sign: | Configuration to override the default retry behavior of the client. | | ### Response @@ -84,7 +86,7 @@ with OpenRouter( ## create_transcription_multipart -Transcribes audio into text. Accepts base64-encoded audio input as JSON or an OpenAI-style multipart/form-data file upload, and returns the transcribed text. +Transcribes audio into text. Accepts base64-encoded audio input as JSON, an OpenAI-style multipart/form-data file upload, or a URL the provider downloads directly, and returns the transcribed text. ### Example Usage @@ -100,10 +102,10 @@ with OpenRouter( api_key=os.getenv("OPENROUTER_API_KEY", ""), ) as open_router: - res = open_router.stt.create_transcription_multipart(file={ + res = open_router.stt.create_transcription_multipart(model="openai/whisper-large-v3", file={ "file_name": "example.file", "content": open("example.file", "rb"), - }, model="openai/whisper-large-v3", language="en") + }, language="en") # Handle response print(res) @@ -112,21 +114,25 @@ with OpenRouter( ### Parameters -| Parameter | Type | Required | Description | -| -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `file` | [operations.CreateAudioTranscriptionsMultipartFile](../../operations/createaudiotranscriptionsmultipartfile.mdx) | :heavy_check_mark: | The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio. | -| `model` | *str* | :heavy_check_mark: | The model to use for transcription. | -| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| -| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| -| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| -| `language` | *Optional[str]* | :heavy_minus_sign: | The language of the input audio (ISO-639-1). | -| `response_format` | [Optional[operations.ResponseFormat]](../../operations/responseformat.mdx) | :heavy_minus_sign: | The response format. "json" (default) returns \{ text, usage }; "verbose_json" additionally returns task, language, duration, and segment-level timestamps (OpenAI-compatible providers only). | -| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. | -| `temperature` | *Optional[float]* | :heavy_minus_sign: | The sampling temperature. | -| `timestamp_granularities` | List[[operations.TimestampGranularities](../../operations/timestampgranularities.mdx)] | :heavy_minus_sign: | Timestamp detail levels to include when response_format is "verbose_json". "word" additionally returns word-level timestamps in the words array. | -| `trace` | *Optional[str]* | :heavy_minus_sign: | JSON-encoded trace metadata object (trace_id, trace_name, span_name, generation_name, parent_span_id and custom keys) attached to the Broadcast trace. Must decode to a JSON object. | -| `user` | *Optional[str]* | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | -| `retries` | [Optional[utils.RetryConfig]](../../models/utils/retryconfig.mdx) | :heavy_minus_sign: | Configuration to override the default retry behavior of the client. | +| Parameter | Type | Required | Description | +| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `model` | *str* | :heavy_check_mark: | The model to use for transcription. | +| `http_referer` | *Optional[str]* | :heavy_minus_sign: | The app identifier should be your app's URL and is used as the primary identifier for rankings.
This is used to track API usage per application.
| +| `x_open_router_title` | *Optional[str]* | :heavy_minus_sign: | The app display name allows you to customize how your app appears in OpenRouter's dashboard.
| +| `x_open_router_categories` | *Optional[str]* | :heavy_minus_sign: | Comma-separated list of app categories (e.g. "cli-agent,cloud-agent"). Used for marketplace rankings.
| +| `diarize` | *Optional[bool]* | :heavy_minus_sign: | Label each word with the speaker who said it (words[].speaker, words[].speaker_label). Requires response_format "verbose_json" (400 otherwise); word timestamps are included even when timestamp_granularities[] omits "word". Only supported by some providers; 400 when the selected model cannot diarize. | +| `file` | [Optional[operations.CreateAudioTranscriptionsMultipartFile]](../../operations/createaudiotranscriptionsmultipartfile.mdx) | :heavy_minus_sign: | The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio, or by URL via source_url. Exactly one of file or source_url is required. | +| `keyterms` | List[*str*] | :heavy_minus_sign: | Domain terms, names, or phrases to bias recognition toward; repeat the part once per term (keyterms=... is also accepted). Only supported by some providers; 400 when the selected model cannot use keyterms. | +| `language` | *Optional[str]* | :heavy_minus_sign: | The language of the input audio (ISO-639-1). | +| `provider` | *Optional[str]* | :heavy_minus_sign: | JSON-encoded provider preferences object, the same shape as the JSON body field: \{ "options": \{ "\": \{ ... } } }. Only options for the matched provider are forwarded. Must decode to a JSON object. | +| `response_format` | [Optional[operations.ResponseFormat]](../../operations/responseformat.mdx) | :heavy_minus_sign: | The response format. "json" (default) returns \{ text, usage }; "verbose_json" additionally returns task, language, duration, and segment-level timestamps (OpenAI-compatible providers only). | +| `session_id` | *Optional[str]* | :heavy_minus_sign: | A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. | +| `source_url` | *Optional[str]* | :heavy_minus_sign: | Publicly reachable http(s) URL of the audio file, downloaded by the provider directly (no size limit on our side). The format is derived from the URL path extension. Only supported by some providers; exactly one of file or source_url is required. | +| `temperature` | *Optional[float]* | :heavy_minus_sign: | The sampling temperature. | +| `timestamp_granularities` | List[[operations.TimestampGranularities](../../operations/timestampgranularities.mdx)] | :heavy_minus_sign: | Timestamp detail levels to include when response_format is "verbose_json". "word" additionally returns word-level timestamps in the words array. | +| `trace` | *Optional[str]* | :heavy_minus_sign: | JSON-encoded trace metadata object (trace_id, trace_name, span_name, generation_name, parent_span_id and custom keys) attached to the Broadcast trace. Must decode to a JSON object. | +| `user` | *Optional[str]* | :heavy_minus_sign: | A unique identifier representing your end-user. Forwarded to Broadcast and private logging as the end-user id; never sent to the provider. | +| `retries` | [Optional[utils.RetryConfig]](../../models/utils/retryconfig.mdx) | :heavy_minus_sign: | Configuration to override the default retry behavior of the client. | ### Response diff --git a/pyproject.toml b/pyproject.toml index df47bedc..f830583b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "openrouter" -version = "1.3.5" +version = "1.3.6" description = "Official Python Client SDK for OpenRouter." authors = [{ name = "OpenRouter" },] readme = "README-PYPI.md" diff --git a/src/openrouter/_version.py b/src/openrouter/_version.py index f99b6685..6e1c5673 100644 --- a/src/openrouter/_version.py +++ b/src/openrouter/_version.py @@ -3,10 +3,10 @@ import importlib.metadata __title__: str = "openrouter" -__version__: str = "1.3.5" +__version__: str = "1.3.6" __openapi_doc_version__: str = "1.0.0" __gen_version__: str = "2.914.0" -__user_agent__: str = "speakeasy-sdk/python 1.3.5 2.914.0 1.0.0 openrouter" +__user_agent__: str = "speakeasy-sdk/python 1.3.6 2.914.0 1.0.0 openrouter" try: if __package__ is not None: diff --git a/src/openrouter/components/__init__.py b/src/openrouter/components/__init__.py index 26f2d3dd..d95413f2 100644 --- a/src/openrouter/components/__init__.py +++ b/src/openrouter/components/__init__.py @@ -3733,6 +3733,8 @@ StreamLogprobTopLogprob, StreamLogprobTopLogprobTypedDict, ) + from .sttentity import STTEntity, STTEntityTypedDict + from .sttinlineinputaudio import STTInlineInputAudio, STTInlineInputAudioTypedDict from .sttinputaudio import STTInputAudio, STTInputAudioTypedDict from .sttrequest import ( STTRequest, @@ -3744,8 +3746,9 @@ from .sttresponse import STTResponse, STTResponseTypedDict from .sttsegment import STTSegment, STTSegmentTypedDict from .stttimestampgranularity import STTTimestampGranularity + from .stturlinputaudio import STTURLInputAudio, STTURLInputAudioTypedDict from .sttusage import STTUsage, STTUsageTypedDict - from .sttword import STTWord, STTWordTypedDict + from .sttword import STTWord, STTWordType, STTWordTypedDict from .subagentnestedtool import SubagentNestedTool, SubagentNestedToolTypedDict from .subagentreasoning import ( SubagentReasoning, @@ -6586,6 +6589,10 @@ "Rule", "RuleTypedDict", "STT", + "STTEntity", + "STTEntityTypedDict", + "STTInlineInputAudio", + "STTInlineInputAudioTypedDict", "STTInputAudio", "STTInputAudioTypedDict", "STTRequest", @@ -6599,9 +6606,12 @@ "STTSegmentTypedDict", "STTTimestampGranularity", "STTTypedDict", + "STTURLInputAudio", + "STTURLInputAudioTypedDict", "STTUsage", "STTUsageTypedDict", "STTWord", + "STTWordType", "STTWordTypedDict", "ScimGroup", "ScimGroupMapping", @@ -9841,6 +9851,10 @@ "StreamLogprobTypedDict": ".streamlogprob", "StreamLogprobTopLogprob": ".streamlogprobtoplogprob", "StreamLogprobTopLogprobTypedDict": ".streamlogprobtoplogprob", + "STTEntity": ".sttentity", + "STTEntityTypedDict": ".sttentity", + "STTInlineInputAudio": ".sttinlineinputaudio", + "STTInlineInputAudioTypedDict": ".sttinlineinputaudio", "STTInputAudio": ".sttinputaudio", "STTInputAudioTypedDict": ".sttinputaudio", "STTRequest": ".sttrequest", @@ -9853,9 +9867,12 @@ "STTSegment": ".sttsegment", "STTSegmentTypedDict": ".sttsegment", "STTTimestampGranularity": ".stttimestampgranularity", + "STTURLInputAudio": ".stturlinputaudio", + "STTURLInputAudioTypedDict": ".stturlinputaudio", "STTUsage": ".sttusage", "STTUsageTypedDict": ".sttusage", "STTWord": ".sttword", + "STTWordType": ".sttword", "STTWordTypedDict": ".sttword", "SubagentNestedTool": ".subagentnestedtool", "SubagentNestedToolTypedDict": ".subagentnestedtool", diff --git a/src/openrouter/components/byokproviderslug.py b/src/openrouter/components/byokproviderslug.py index 8fe1b36c..47154c65 100644 --- a/src/openrouter/components/byokproviderslug.py +++ b/src/openrouter/components/byokproviderslug.py @@ -44,6 +44,7 @@ "deepseek", "dekallm", "digitalocean", + "elevenlabs", "featherless", "fireworks", "fish-audio", diff --git a/src/openrouter/components/imagegenerationproviderpreferences.py b/src/openrouter/components/imagegenerationproviderpreferences.py index c9f6cc2d..07eb9052 100644 --- a/src/openrouter/components/imagegenerationproviderpreferences.py +++ b/src/openrouter/components/imagegenerationproviderpreferences.py @@ -83,6 +83,7 @@ class ImageGenerationProviderPreferencesOptionsTypedDict(TypedDict): deepseek: NotRequired[Dict[str, Any]] dekallm: NotRequired[Dict[str, Any]] digitalocean: NotRequired[Dict[str, Any]] + elevenlabs: NotRequired[Dict[str, Any]] enfer: NotRequired[Dict[str, Any]] fake_provider: NotRequired[Dict[str, Any]] featherless: NotRequired[Dict[str, Any]] @@ -294,6 +295,8 @@ class ImageGenerationProviderPreferencesOptions(BaseModel): digitalocean: Optional[Dict[str, Any]] = None + elevenlabs: Optional[Dict[str, Any]] = None + enfer: Optional[Dict[str, Any]] = None fake_provider: Annotated[ @@ -577,6 +580,7 @@ def serialize_model(self, handler): "deepseek", "dekallm", "digitalocean", + "elevenlabs", "enfer", "fake-provider", "featherless", diff --git a/src/openrouter/components/providername.py b/src/openrouter/components/providername.py index 480e20a7..1ea21048 100644 --- a/src/openrouter/components/providername.py +++ b/src/openrouter/components/providername.py @@ -44,6 +44,7 @@ "DeepSeek", "DekaLLM", "DigitalOcean", + "ElevenLabs", "Featherless", "Fireworks", "Fish Audio", diff --git a/src/openrouter/components/provideroptions.py b/src/openrouter/components/provideroptions.py index e8208511..57232d4d 100644 --- a/src/openrouter/components/provideroptions.py +++ b/src/openrouter/components/provideroptions.py @@ -54,6 +54,7 @@ class ProviderOptionsTypedDict(TypedDict): deepseek: NotRequired[Dict[str, Any]] dekallm: NotRequired[Dict[str, Any]] digitalocean: NotRequired[Dict[str, Any]] + elevenlabs: NotRequired[Dict[str, Any]] enfer: NotRequired[Dict[str, Any]] fake_provider: NotRequired[Dict[str, Any]] featherless: NotRequired[Dict[str, Any]] @@ -265,6 +266,8 @@ class ProviderOptions(BaseModel): digitalocean: Optional[Dict[str, Any]] = None + elevenlabs: Optional[Dict[str, Any]] = None + enfer: Optional[Dict[str, Any]] = None fake_provider: Annotated[ @@ -548,6 +551,7 @@ def serialize_model(self, handler): "deepseek", "dekallm", "digitalocean", + "elevenlabs", "enfer", "fake-provider", "featherless", diff --git a/src/openrouter/components/providerresponse.py b/src/openrouter/components/providerresponse.py index 2f04468a..9918b560 100644 --- a/src/openrouter/components/providerresponse.py +++ b/src/openrouter/components/providerresponse.py @@ -74,6 +74,7 @@ "DeepSeek", "DekaLLM", "DigitalOcean", + "ElevenLabs", "Featherless", "Fireworks", "Fish Audio", diff --git a/src/openrouter/components/sttentity.py b/src/openrouter/components/sttentity.py new file mode 100644 index 00000000..84ac78c7 --- /dev/null +++ b/src/openrouter/components/sttentity.py @@ -0,0 +1,34 @@ +"""Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT.""" + +from __future__ import annotations +from openrouter.types import BaseModel +from typing_extensions import TypedDict + + +class STTEntityTypedDict(TypedDict): + r"""A detected entity, returned when the provider runs entity detection""" + + end_char: int + r"""Zero-based exclusive character offset of the entity end within the response-level text (not seconds)""" + start_char: int + r"""Zero-based character offset of the entity start within the response-level text (not seconds)""" + text: str + r"""Entity text as it appears in the transcript""" + type: str + r"""Provider entity type label""" + + +class STTEntity(BaseModel): + r"""A detected entity, returned when the provider runs entity detection""" + + end_char: int + r"""Zero-based exclusive character offset of the entity end within the response-level text (not seconds)""" + + start_char: int + r"""Zero-based character offset of the entity start within the response-level text (not seconds)""" + + text: str + r"""Entity text as it appears in the transcript""" + + type: str + r"""Provider entity type label""" diff --git a/src/openrouter/components/sttinlineinputaudio.py b/src/openrouter/components/sttinlineinputaudio.py new file mode 100644 index 00000000..97e46c7a --- /dev/null +++ b/src/openrouter/components/sttinlineinputaudio.py @@ -0,0 +1,31 @@ +"""Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT.""" + +from __future__ import annotations +from openrouter.types import BaseModel +import pydantic +from typing_extensions import Annotated, TypedDict + + +class STTInlineInputAudioTypedDict(TypedDict): + r"""Inline base64 audio input for speech-to-text""" + + data: str + r"""Base64-encoded audio data (raw bytes, not a data URI)""" + format_: str + r"""Audio format (e.g., wav, mp3, flac, m4a, ogg, webm, aac). Supported formats vary by provider. \"pcm\" means headerless signed 16-bit little-endian mono audio at 16 kHz.""" + + +class STTInlineInputAudio(BaseModel): + r"""Inline base64 audio input for speech-to-text""" + + data: str + r"""Base64-encoded audio data (raw bytes, not a data URI)""" + + format_: Annotated[str, pydantic.Field(alias="format")] + r"""Audio format (e.g., wav, mp3, flac, m4a, ogg, webm, aac). Supported formats vary by provider. \"pcm\" means headerless signed 16-bit little-endian mono audio at 16 kHz.""" + + +try: + STTInlineInputAudio.model_rebuild() +except NameError: + pass diff --git a/src/openrouter/components/sttinputaudio.py b/src/openrouter/components/sttinputaudio.py index 76c4b466..75281f47 100644 --- a/src/openrouter/components/sttinputaudio.py +++ b/src/openrouter/components/sttinputaudio.py @@ -1,31 +1,20 @@ """Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT.""" from __future__ import annotations -from openrouter.types import BaseModel -import pydantic -from typing_extensions import Annotated, TypedDict +from .sttinlineinputaudio import STTInlineInputAudio, STTInlineInputAudioTypedDict +from .stturlinputaudio import STTURLInputAudio, STTURLInputAudioTypedDict +from typing import Union +from typing_extensions import TypeAliasType -class STTInputAudioTypedDict(TypedDict): - r"""Base64-encoded audio to transcribe""" +STTInputAudioTypedDict = TypeAliasType( + "STTInputAudioTypedDict", + Union[STTInlineInputAudioTypedDict, STTURLInputAudioTypedDict], +) +r"""Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly.""" - data: str - r"""Base64-encoded audio data (raw bytes, not a data URI)""" - format_: str - r"""Audio format (e.g., wav, mp3, flac, m4a, ogg, webm, aac). Supported formats vary by provider.""" - -class STTInputAudio(BaseModel): - r"""Base64-encoded audio to transcribe""" - - data: str - r"""Base64-encoded audio data (raw bytes, not a data URI)""" - - format_: Annotated[str, pydantic.Field(alias="format")] - r"""Audio format (e.g., wav, mp3, flac, m4a, ogg, webm, aac). Supported formats vary by provider.""" - - -try: - STTInputAudio.model_rebuild() -except NameError: - pass +STTInputAudio = TypeAliasType( + "STTInputAudio", Union[STTInlineInputAudio, STTURLInputAudio] +) +r"""Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly.""" diff --git a/src/openrouter/components/sttrequest.py b/src/openrouter/components/sttrequest.py index 44c22985..452ad087 100644 --- a/src/openrouter/components/sttrequest.py +++ b/src/openrouter/components/sttrequest.py @@ -52,12 +52,16 @@ def serialize_model(self, handler): class STTRequestTypedDict(TypedDict): - r"""Speech-to-text request input. Accepts a JSON body with input_audio containing base64-encoded audio.""" + r"""Speech-to-text request input. Accepts a JSON body with input_audio containing base64-encoded audio or a URL the provider downloads.""" input_audio: STTInputAudioTypedDict - r"""Base64-encoded audio to transcribe""" + r"""Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly.""" model: str r"""STT model identifier""" + diarize: NotRequired[bool] + r"""Label each word with the speaker who said it. Speaker labels are returned on the words array (speaker, speaker_label), so response_format must be \"verbose_json\" (a \"json\" request is rejected with a 400) and word timestamps are included even when timestamp_granularities omits \"word\". Only supported by some providers; the request is rejected with a 400 when the selected model cannot diarize. Providers may charge extra.""" + keyterms: NotRequired[List[str]] + r"""Domain terms, names, or phrases to bias recognition toward. Only supported by some providers; the request is rejected with a 400 when the selected model cannot use keyterms. Providers may cap the number of terms or characters per term and may charge extra.""" language: NotRequired[str] r"""ISO-639-1 language code (e.g., \"en\", \"ja\"). Auto-detected if omitted.""" provider: NotRequired[STTRequestProviderTypedDict] @@ -77,14 +81,20 @@ class STTRequestTypedDict(TypedDict): class STTRequest(BaseModel): - r"""Speech-to-text request input. Accepts a JSON body with input_audio containing base64-encoded audio.""" + r"""Speech-to-text request input. Accepts a JSON body with input_audio containing base64-encoded audio or a URL the provider downloads.""" input_audio: STTInputAudio - r"""Base64-encoded audio to transcribe""" + r"""Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly.""" model: str r"""STT model identifier""" + diarize: Optional[bool] = None + r"""Label each word with the speaker who said it. Speaker labels are returned on the words array (speaker, speaker_label), so response_format must be \"verbose_json\" (a \"json\" request is rejected with a 400) and word timestamps are included even when timestamp_granularities omits \"word\". Only supported by some providers; the request is rejected with a 400 when the selected model cannot diarize. Providers may charge extra.""" + + keyterms: Optional[List[str]] = None + r"""Domain terms, names, or phrases to bias recognition toward. Only supported by some providers; the request is rejected with a 400 when the selected model cannot use keyterms. Providers may cap the number of terms or characters per term and may charge extra.""" + language: Optional[str] = None r"""ISO-639-1 language code (e.g., \"en\", \"ja\"). Auto-detected if omitted.""" @@ -113,6 +123,8 @@ class STTRequest(BaseModel): def serialize_model(self, handler): optional_fields = set( [ + "diarize", + "keyterms", "language", "provider", "response_format", diff --git a/src/openrouter/components/sttresponse.py b/src/openrouter/components/sttresponse.py index 6ea5f83a..2132dfe0 100644 --- a/src/openrouter/components/sttresponse.py +++ b/src/openrouter/components/sttresponse.py @@ -1,6 +1,7 @@ """Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT.""" from __future__ import annotations +from .sttentity import STTEntity, STTEntityTypedDict from .sttsegment import STTSegment, STTSegmentTypedDict from .sttusage import STTUsage, STTUsageTypedDict from .sttword import STTWord, STTWordTypedDict @@ -19,8 +20,12 @@ class STTResponseTypedDict(TypedDict): r"""Provider confidence for the whole transcript from 0 to 1, present when response_format is verbose_json and the provider scores the full transcript""" duration: NotRequired[float] r"""Duration of the input audio in seconds, present when response_format is verbose_json""" + entities: NotRequired[List[STTEntityTypedDict]] + r"""Detected entities with character offsets into text, present when the provider runs entity detection""" language: NotRequired[str] r"""Detected or forced language, present when response_format is verbose_json""" + language_confidence: NotRequired[float] + r"""Provider confidence in the detected language from 0 to 1, present when response_format is verbose_json and the provider scores language detection""" segments: NotRequired[List[STTSegmentTypedDict]] r"""Timestamped transcript segments, present when response_format is verbose_json""" task: NotRequired[str] @@ -43,9 +48,15 @@ class STTResponse(BaseModel): duration: Optional[float] = None r"""Duration of the input audio in seconds, present when response_format is verbose_json""" + entities: Optional[List[STTEntity]] = None + r"""Detected entities with character offsets into text, present when the provider runs entity detection""" + language: Optional[str] = None r"""Detected or forced language, present when response_format is verbose_json""" + language_confidence: Optional[float] = None + r"""Provider confidence in the detected language from 0 to 1, present when response_format is verbose_json and the provider scores language detection""" + segments: Optional[List[STTSegment]] = None r"""Timestamped transcript segments, present when response_format is verbose_json""" @@ -61,7 +72,17 @@ class STTResponse(BaseModel): @model_serializer(mode="wrap") def serialize_model(self, handler): optional_fields = set( - ["confidence", "duration", "language", "segments", "task", "usage", "words"] + [ + "confidence", + "duration", + "entities", + "language", + "language_confidence", + "segments", + "task", + "usage", + "words", + ] ) serialized = handler(self) m = {} diff --git a/src/openrouter/components/sttsegment.py b/src/openrouter/components/sttsegment.py index 024e8a33..f17dccc6 100644 --- a/src/openrouter/components/sttsegment.py +++ b/src/openrouter/components/sttsegment.py @@ -20,6 +20,8 @@ class STTSegmentTypedDict(TypedDict): r"""Transcribed text of the segment""" avg_logprob: NotRequired[float] r"""Average log probability of the segment""" + channel: NotRequired[int] + r"""Zero-based audio channel index for the segment, present when the provider transcribes channels separately""" compression_ratio: NotRequired[float] r"""Compression ratio of the segment""" no_speech_prob: NotRequired[float] @@ -28,6 +30,8 @@ class STTSegmentTypedDict(TypedDict): r"""Seek offset of the segment""" speaker: NotRequired[int] r"""Speaker index for the segment, present when the provider returns diarization data""" + speaker_label: NotRequired[str] + r"""Provider speaker label for the segment, present when the provider labels speakers with a string""" temperature: NotRequired[float] r"""Temperature used for the segment""" tokens: NotRequired[List[int]] @@ -52,6 +56,9 @@ class STTSegment(BaseModel): avg_logprob: Optional[float] = None r"""Average log probability of the segment""" + channel: Optional[int] = None + r"""Zero-based audio channel index for the segment, present when the provider transcribes channels separately""" + compression_ratio: Optional[float] = None r"""Compression ratio of the segment""" @@ -64,6 +71,9 @@ class STTSegment(BaseModel): speaker: Optional[int] = None r"""Speaker index for the segment, present when the provider returns diarization data""" + speaker_label: Optional[str] = None + r"""Provider speaker label for the segment, present when the provider labels speakers with a string""" + temperature: Optional[float] = None r"""Temperature used for the segment""" @@ -75,10 +85,12 @@ def serialize_model(self, handler): optional_fields = set( [ "avg_logprob", + "channel", "compression_ratio", "no_speech_prob", "seek", "speaker", + "speaker_label", "temperature", "tokens", ] diff --git a/src/openrouter/components/stturlinputaudio.py b/src/openrouter/components/stturlinputaudio.py new file mode 100644 index 00000000..5d7fcefd --- /dev/null +++ b/src/openrouter/components/stturlinputaudio.py @@ -0,0 +1,49 @@ +"""Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT.""" + +from __future__ import annotations +from openrouter.types import BaseModel, UNSET_SENTINEL +import pydantic +from pydantic import model_serializer +from typing import Optional +from typing_extensions import Annotated, NotRequired, TypedDict + + +class STTURLInputAudioTypedDict(TypedDict): + r"""Audio input fetched by the provider from a URL""" + + url: str + r"""Publicly reachable http(s) URL of the audio file. The provider downloads it directly, so the inline upload size limit does not apply. Only supported by some providers.""" + format_: NotRequired[str] + r"""Audio format of the file at the URL. Defaults to the extension of the URL path; required when the path has no extension.""" + + +class STTURLInputAudio(BaseModel): + r"""Audio input fetched by the provider from a URL""" + + url: str + r"""Publicly reachable http(s) URL of the audio file. The provider downloads it directly, so the inline upload size limit does not apply. Only supported by some providers.""" + + format_: Annotated[Optional[str], pydantic.Field(alias="format")] = None + r"""Audio format of the file at the URL. Defaults to the extension of the URL path; required when the path has no extension.""" + + @model_serializer(mode="wrap") + def serialize_model(self, handler): + optional_fields = set(["format"]) + serialized = handler(self) + m = {} + + for n, f in type(self).model_fields.items(): + k = f.alias or n + val = serialized.get(k, serialized.get(n)) + + if val != UNSET_SENTINEL: + if val is not None or k not in optional_fields: + m[k] = val + + return m + + +try: + STTURLInputAudio.model_rebuild() +except NameError: + pass diff --git a/src/openrouter/components/sttword.py b/src/openrouter/components/sttword.py index d7e59a3e..2f6c9e97 100644 --- a/src/openrouter/components/sttword.py +++ b/src/openrouter/components/sttword.py @@ -1,12 +1,22 @@ """Code generated by Speakeasy (https://speakeasy.com). DO NOT EDIT.""" from __future__ import annotations -from openrouter.types import BaseModel, UNSET_SENTINEL +from openrouter.types import BaseModel, UNSET_SENTINEL, UnrecognizedStr from pydantic import model_serializer -from typing import Optional +from typing import Literal, Optional, Union from typing_extensions import NotRequired, TypedDict +STTWordType = Union[ + Literal[ + "word", + "audio_event", + ], + UnrecognizedStr, +] +r"""Kind of entry; omitted or \"word\" for spoken words, \"audio_event\" for non-speech sounds the provider tags with timestamps""" + + class STTWordTypedDict(TypedDict): r"""A timestamped word, returned when the provider includes word-level timestamps""" @@ -15,11 +25,17 @@ class STTWordTypedDict(TypedDict): start: float r"""Word start time in seconds""" word: str - r"""The transcribed word""" + r"""The transcribed word, or the event tag such as \"(laughter)\" when type is audio_event""" + channel: NotRequired[int] + r"""Zero-based audio channel index for the word, present when the provider transcribes channels separately""" confidence: NotRequired[float] r"""Provider confidence for the word from 0 to 1, present when the provider returns per-word confidence""" speaker: NotRequired[int] r"""Speaker index for the word, present when the provider returns diarization data""" + speaker_label: NotRequired[str] + r"""Provider speaker label for the word, present when the provider labels speakers with a string""" + type: NotRequired[STTWordType] + r"""Kind of entry; omitted or \"word\" for spoken words, \"audio_event\" for non-speech sounds the provider tags with timestamps""" class STTWord(BaseModel): @@ -32,7 +48,10 @@ class STTWord(BaseModel): r"""Word start time in seconds""" word: str - r"""The transcribed word""" + r"""The transcribed word, or the event tag such as \"(laughter)\" when type is audio_event""" + + channel: Optional[int] = None + r"""Zero-based audio channel index for the word, present when the provider transcribes channels separately""" confidence: Optional[float] = None r"""Provider confidence for the word from 0 to 1, present when the provider returns per-word confidence""" @@ -40,9 +59,17 @@ class STTWord(BaseModel): speaker: Optional[int] = None r"""Speaker index for the word, present when the provider returns diarization data""" + speaker_label: Optional[str] = None + r"""Provider speaker label for the word, present when the provider labels speakers with a string""" + + type: Optional[STTWordType] = None + r"""Kind of entry; omitted or \"word\" for spoken words, \"audio_event\" for non-speech sounds the provider tags with timestamps""" + @model_serializer(mode="wrap") def serialize_model(self, handler): - optional_fields = set(["confidence", "speaker"]) + optional_fields = set( + ["channel", "confidence", "speaker", "speaker_label", "type"] + ) serialized = handler(self) m = {} diff --git a/src/openrouter/components/videogenerationrequest.py b/src/openrouter/components/videogenerationrequest.py index 026c12a1..f2e96fc6 100644 --- a/src/openrouter/components/videogenerationrequest.py +++ b/src/openrouter/components/videogenerationrequest.py @@ -74,6 +74,7 @@ class VideoGenerationRequestOptionsTypedDict(TypedDict): deepseek: NotRequired[Dict[str, Any]] dekallm: NotRequired[Dict[str, Any]] digitalocean: NotRequired[Dict[str, Any]] + elevenlabs: NotRequired[Dict[str, Any]] enfer: NotRequired[Dict[str, Any]] fake_provider: NotRequired[Dict[str, Any]] featherless: NotRequired[Dict[str, Any]] @@ -285,6 +286,8 @@ class VideoGenerationRequestOptions(BaseModel): digitalocean: Optional[Dict[str, Any]] = None + elevenlabs: Optional[Dict[str, Any]] = None + enfer: Optional[Dict[str, Any]] = None fake_provider: Annotated[ @@ -568,6 +571,7 @@ def serialize_model(self, handler): "deepseek", "dekallm", "digitalocean", + "elevenlabs", "enfer", "fake-provider", "featherless", diff --git a/src/openrouter/operations/createaudiotranscriptions_multipart.py b/src/openrouter/operations/createaudiotranscriptions_multipart.py index 30c86b20..91eaa750 100644 --- a/src/openrouter/operations/createaudiotranscriptions_multipart.py +++ b/src/openrouter/operations/createaudiotranscriptions_multipart.py @@ -139,16 +139,24 @@ def serialize_model(self, handler): class CreateAudioTranscriptionsMultipartRequestBodyTypedDict(TypedDict): - file: CreateAudioTranscriptionsMultipartFileTypedDict - r"""The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio.""" model: str r"""The model to use for transcription.""" + diarize: NotRequired[bool] + r"""Label each word with the speaker who said it (words[].speaker, words[].speaker_label). Requires response_format \"verbose_json\" (400 otherwise); word timestamps are included even when timestamp_granularities[] omits \"word\". Only supported by some providers; 400 when the selected model cannot diarize.""" + file: NotRequired[CreateAudioTranscriptionsMultipartFileTypedDict] + r"""The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio, or by URL via source_url. Exactly one of file or source_url is required.""" + keyterms: NotRequired[List[str]] + r"""Domain terms, names, or phrases to bias recognition toward; repeat the part once per term (keyterms=... is also accepted). Only supported by some providers; 400 when the selected model cannot use keyterms.""" language: NotRequired[str] r"""The language of the input audio (ISO-639-1).""" + provider: NotRequired[str] + r"""JSON-encoded provider preferences object, the same shape as the JSON body field: { \"options\": { \"\": { ... } } }. Only options for the matched provider are forwarded. Must decode to a JSON object.""" response_format: NotRequired[ResponseFormat] r"""The response format. \"json\" (default) returns { text, usage }; \"verbose_json\" additionally returns task, language, duration, and segment-level timestamps (OpenAI-compatible providers only).""" session_id: NotRequired[str] r"""A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence.""" + source_url: NotRequired[str] + r"""Publicly reachable http(s) URL of the audio file, downloaded by the provider directly (no size limit on our side). The format is derived from the URL path extension. Only supported by some providers; exactly one of file or source_url is required.""" temperature: NotRequired[float] r"""The sampling temperature.""" timestamp_granularities: NotRequired[List[TimestampGranularities]] @@ -160,18 +168,31 @@ class CreateAudioTranscriptionsMultipartRequestBodyTypedDict(TypedDict): class CreateAudioTranscriptionsMultipartRequestBody(BaseModel): + model: Annotated[str, FieldMetadata(multipart=True)] + r"""The model to use for transcription.""" + + diarize: Annotated[Optional[bool], FieldMetadata(multipart=True)] = None + r"""Label each word with the speaker who said it (words[].speaker, words[].speaker_label). Requires response_format \"verbose_json\" (400 otherwise); word timestamps are included even when timestamp_granularities[] omits \"word\". Only supported by some providers; 400 when the selected model cannot diarize.""" + file: Annotated[ - CreateAudioTranscriptionsMultipartFile, + Optional[CreateAudioTranscriptionsMultipartFile], FieldMetadata(multipart=MultipartFormMetadata(file=True)), - ] - r"""The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio.""" + ] = None + r"""The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio, or by URL via source_url. Exactly one of file or source_url is required.""" - model: Annotated[str, FieldMetadata(multipart=True)] - r"""The model to use for transcription.""" + keyterms: Annotated[ + Optional[List[str]], + pydantic.Field(alias="keyterms[]"), + FieldMetadata(multipart=True), + ] = None + r"""Domain terms, names, or phrases to bias recognition toward; repeat the part once per term (keyterms=... is also accepted). Only supported by some providers; 400 when the selected model cannot use keyterms.""" language: Annotated[Optional[str], FieldMetadata(multipart=True)] = None r"""The language of the input audio (ISO-639-1).""" + provider: Annotated[Optional[str], FieldMetadata(multipart=True)] = None + r"""JSON-encoded provider preferences object, the same shape as the JSON body field: { \"options\": { \"\": { ... } } }. Only options for the matched provider are forwarded. Must decode to a JSON object.""" + response_format: Annotated[ Optional[ResponseFormat], FieldMetadata(multipart=True) ] = None @@ -180,6 +201,9 @@ class CreateAudioTranscriptionsMultipartRequestBody(BaseModel): session_id: Annotated[Optional[str], FieldMetadata(multipart=True)] = None r"""A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence.""" + source_url: Annotated[Optional[str], FieldMetadata(multipart=True)] = None + r"""Publicly reachable http(s) URL of the audio file, downloaded by the provider directly (no size limit on our side). The format is derived from the URL path extension. Only supported by some providers; exactly one of file or source_url is required.""" + temperature: Annotated[Optional[float], FieldMetadata(multipart=True)] = None r"""The sampling temperature.""" @@ -200,9 +224,14 @@ class CreateAudioTranscriptionsMultipartRequestBody(BaseModel): def serialize_model(self, handler): optional_fields = set( [ + "diarize", + "file", + "keyterms[]", "language", + "provider", "response_format", "session_id", + "source_url", "temperature", "timestamp_granularities[]", "trace", diff --git a/src/openrouter/operations/listbyokkeys.py b/src/openrouter/operations/listbyokkeys.py index e15fe3ef..3204af92 100644 --- a/src/openrouter/operations/listbyokkeys.py +++ b/src/openrouter/operations/listbyokkeys.py @@ -121,6 +121,7 @@ def serialize_model(self, handler): "deepseek", "dekallm", "digitalocean", + "elevenlabs", "featherless", "fireworks", "fish-audio", diff --git a/src/openrouter/stt.py b/src/openrouter/stt.py index 7502cb97..d368541c 100644 --- a/src/openrouter/stt.py +++ b/src/openrouter/stt.py @@ -20,6 +20,8 @@ def create_transcription( http_referer: Optional[str] = None, x_open_router_title: Optional[str] = None, x_open_router_categories: Optional[str] = None, + diarize: Optional[bool] = None, + keyterms: Optional[Iterable[str]] = None, language: Optional[str] = None, provider: Optional[ Union[components.STTRequestProvider, components.STTRequestProviderTypedDict] @@ -41,9 +43,9 @@ def create_transcription( ) -> components.STTResponse: r"""Create transcription - Transcribes audio into text. Accepts base64-encoded audio input as JSON or an OpenAI-style multipart/form-data file upload, and returns the transcribed text. + Transcribes audio into text. Accepts base64-encoded audio input as JSON, an OpenAI-style multipart/form-data file upload, or a URL the provider downloads directly, and returns the transcribed text. - :param input_audio: Base64-encoded audio to transcribe + :param input_audio: Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly. :param model: STT model identifier :param http_referer: The app identifier should be your app's URL and is used as the primary identifier for rankings. This is used to track API usage per application. @@ -52,6 +54,8 @@ def create_transcription( :param x_open_router_categories: Comma-separated list of app categories (e.g. \"cli-agent,cloud-agent\"). Used for marketplace rankings. + :param diarize: Label each word with the speaker who said it. Speaker labels are returned on the words array (speaker, speaker_label), so response_format must be \"verbose_json\" (a \"json\" request is rejected with a 400) and word timestamps are included even when timestamp_granularities omits \"word\". Only supported by some providers; the request is rejected with a 400 when the selected model cannot diarize. Providers may charge extra. + :param keyterms: Domain terms, names, or phrases to bias recognition toward. Only supported by some providers; the request is rejected with a 400 when the selected model cannot use keyterms. Providers may cap the number of terms or characters per term and may charge extra. :param language: ISO-639-1 language code (e.g., \"en\", \"ja\"). Auto-detected if omitted. :param provider: Provider-specific passthrough configuration :param response_format: Output format. \"json\" (default) returns { text, usage }. \"verbose_json\" additionally returns task, language, duration, and segment-level timestamps; only supported by OpenAI-compatible providers. @@ -80,9 +84,11 @@ def create_transcription( x_open_router_title=x_open_router_title, x_open_router_categories=x_open_router_categories, stt_request=components.STTRequest( + diarize=diarize, input_audio=utils.get_pydantic_model( input_audio, components.STTInputAudio ), + keyterms=utils.unmarshal(keyterms, Optional[List[str]]), language=language, model=model, provider=utils.get_pydantic_model( @@ -243,6 +249,8 @@ async def create_transcription_async( http_referer: Optional[str] = None, x_open_router_title: Optional[str] = None, x_open_router_categories: Optional[str] = None, + diarize: Optional[bool] = None, + keyterms: Optional[Iterable[str]] = None, language: Optional[str] = None, provider: Optional[ Union[components.STTRequestProvider, components.STTRequestProviderTypedDict] @@ -264,9 +272,9 @@ async def create_transcription_async( ) -> components.STTResponse: r"""Create transcription - Transcribes audio into text. Accepts base64-encoded audio input as JSON or an OpenAI-style multipart/form-data file upload, and returns the transcribed text. + Transcribes audio into text. Accepts base64-encoded audio input as JSON, an OpenAI-style multipart/form-data file upload, or a URL the provider downloads directly, and returns the transcribed text. - :param input_audio: Base64-encoded audio to transcribe + :param input_audio: Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly. :param model: STT model identifier :param http_referer: The app identifier should be your app's URL and is used as the primary identifier for rankings. This is used to track API usage per application. @@ -275,6 +283,8 @@ async def create_transcription_async( :param x_open_router_categories: Comma-separated list of app categories (e.g. \"cli-agent,cloud-agent\"). Used for marketplace rankings. + :param diarize: Label each word with the speaker who said it. Speaker labels are returned on the words array (speaker, speaker_label), so response_format must be \"verbose_json\" (a \"json\" request is rejected with a 400) and word timestamps are included even when timestamp_granularities omits \"word\". Only supported by some providers; the request is rejected with a 400 when the selected model cannot diarize. Providers may charge extra. + :param keyterms: Domain terms, names, or phrases to bias recognition toward. Only supported by some providers; the request is rejected with a 400 when the selected model cannot use keyterms. Providers may cap the number of terms or characters per term and may charge extra. :param language: ISO-639-1 language code (e.g., \"en\", \"ja\"). Auto-detected if omitted. :param provider: Provider-specific passthrough configuration :param response_format: Output format. \"json\" (default) returns { text, usage }. \"verbose_json\" additionally returns task, language, duration, and segment-level timestamps; only supported by OpenAI-compatible providers. @@ -303,9 +313,11 @@ async def create_transcription_async( x_open_router_title=x_open_router_title, x_open_router_categories=x_open_router_categories, stt_request=components.STTRequest( + diarize=diarize, input_audio=utils.get_pydantic_model( input_audio, components.STTInputAudio ), + keyterms=utils.unmarshal(keyterms, Optional[List[str]]), language=language, model=model, provider=utils.get_pydantic_model( @@ -461,17 +473,23 @@ async def create_transcription_async( def create_transcription_multipart( self, *, - file: Union[ - operations.CreateAudioTranscriptionsMultipartFile, - operations.CreateAudioTranscriptionsMultipartFileTypedDict, - ], model: str, http_referer: Optional[str] = None, x_open_router_title: Optional[str] = None, x_open_router_categories: Optional[str] = None, + diarize: Optional[bool] = None, + file: Optional[ + Union[ + operations.CreateAudioTranscriptionsMultipartFile, + operations.CreateAudioTranscriptionsMultipartFileTypedDict, + ] + ] = None, + keyterms: Optional[Iterable[str]] = None, language: Optional[str] = None, + provider: Optional[str] = None, response_format: Optional[operations.ResponseFormat] = None, session_id: Optional[str] = None, + source_url: Optional[str] = None, temperature: Optional[float] = None, timestamp_granularities: Optional[ Iterable[operations.TimestampGranularities] @@ -485,9 +503,8 @@ def create_transcription_multipart( ) -> components.STTResponse: r"""Create transcription - Transcribes audio into text. Accepts base64-encoded audio input as JSON or an OpenAI-style multipart/form-data file upload, and returns the transcribed text. + Transcribes audio into text. Accepts base64-encoded audio input as JSON, an OpenAI-style multipart/form-data file upload, or a URL the provider downloads directly, and returns the transcribed text. - :param file: The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio. :param model: The model to use for transcription. :param http_referer: The app identifier should be your app's URL and is used as the primary identifier for rankings. This is used to track API usage per application. @@ -496,9 +513,14 @@ def create_transcription_multipart( :param x_open_router_categories: Comma-separated list of app categories (e.g. \"cli-agent,cloud-agent\"). Used for marketplace rankings. + :param diarize: Label each word with the speaker who said it (words[].speaker, words[].speaker_label). Requires response_format \"verbose_json\" (400 otherwise); word timestamps are included even when timestamp_granularities[] omits \"word\". Only supported by some providers; 400 when the selected model cannot diarize. + :param file: The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio, or by URL via source_url. Exactly one of file or source_url is required. + :param keyterms: Domain terms, names, or phrases to bias recognition toward; repeat the part once per term (keyterms=... is also accepted). Only supported by some providers; 400 when the selected model cannot use keyterms. :param language: The language of the input audio (ISO-639-1). + :param provider: JSON-encoded provider preferences object, the same shape as the JSON body field: { \"options\": { \"\": { ... } } }. Only options for the matched provider are forwarded. Must decode to a JSON object. :param response_format: The response format. \"json\" (default) returns { text, usage }; \"verbose_json\" additionally returns task, language, duration, and segment-level timestamps (OpenAI-compatible providers only). :param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. + :param source_url: Publicly reachable http(s) URL of the audio file, downloaded by the provider directly (no size limit on our side). The format is derived from the URL path extension. Only supported by some providers; exactly one of file or source_url is required. :param temperature: The sampling temperature. :param timestamp_granularities: Timestamp detail levels to include when response_format is \"verbose_json\". \"word\" additionally returns word-level timestamps in the words array. :param trace: JSON-encoded trace metadata object (trace_id, trace_name, span_name, generation_name, parent_span_id and custom keys) attached to the Broadcast trace. Must decode to a JSON object. @@ -523,13 +545,17 @@ def create_transcription_multipart( x_open_router_title=x_open_router_title, x_open_router_categories=x_open_router_categories, request_body=operations.CreateAudioTranscriptionsMultipartRequestBody( + diarize=diarize, file=utils.get_pydantic_model( - file, operations.CreateAudioTranscriptionsMultipartFile + file, Optional[operations.CreateAudioTranscriptionsMultipartFile] ), + keyterms=utils.unmarshal(keyterms, Optional[List[str]]), language=language, model=model, + provider=provider, response_format=response_format, session_id=session_id, + source_url=source_url, temperature=temperature, timestamp_granularities=utils.unmarshal( timestamp_granularities, @@ -682,17 +708,23 @@ def create_transcription_multipart( async def create_transcription_multipart_async( self, *, - file: Union[ - operations.CreateAudioTranscriptionsMultipartFile, - operations.CreateAudioTranscriptionsMultipartFileTypedDict, - ], model: str, http_referer: Optional[str] = None, x_open_router_title: Optional[str] = None, x_open_router_categories: Optional[str] = None, + diarize: Optional[bool] = None, + file: Optional[ + Union[ + operations.CreateAudioTranscriptionsMultipartFile, + operations.CreateAudioTranscriptionsMultipartFileTypedDict, + ] + ] = None, + keyterms: Optional[Iterable[str]] = None, language: Optional[str] = None, + provider: Optional[str] = None, response_format: Optional[operations.ResponseFormat] = None, session_id: Optional[str] = None, + source_url: Optional[str] = None, temperature: Optional[float] = None, timestamp_granularities: Optional[ Iterable[operations.TimestampGranularities] @@ -706,9 +738,8 @@ async def create_transcription_multipart_async( ) -> components.STTResponse: r"""Create transcription - Transcribes audio into text. Accepts base64-encoded audio input as JSON or an OpenAI-style multipart/form-data file upload, and returns the transcribed text. + Transcribes audio into text. Accepts base64-encoded audio input as JSON, an OpenAI-style multipart/form-data file upload, or a URL the provider downloads directly, and returns the transcribed text. - :param file: The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio. :param model: The model to use for transcription. :param http_referer: The app identifier should be your app's URL and is used as the primary identifier for rankings. This is used to track API usage per application. @@ -717,9 +748,14 @@ async def create_transcription_multipart_async( :param x_open_router_categories: Comma-separated list of app categories (e.g. \"cli-agent,cloud-agent\"). Used for marketplace rankings. + :param diarize: Label each word with the speaker who said it (words[].speaker, words[].speaker_label). Requires response_format \"verbose_json\" (400 otherwise); word timestamps are included even when timestamp_granularities[] omits \"word\". Only supported by some providers; 400 when the selected model cannot diarize. + :param file: The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio, or by URL via source_url. Exactly one of file or source_url is required. + :param keyterms: Domain terms, names, or phrases to bias recognition toward; repeat the part once per term (keyterms=... is also accepted). Only supported by some providers; 400 when the selected model cannot use keyterms. :param language: The language of the input audio (ISO-639-1). + :param provider: JSON-encoded provider preferences object, the same shape as the JSON body field: { \"options\": { \"\": { ... } } }. Only options for the matched provider are forwarded. Must decode to a JSON object. :param response_format: The response format. \"json\" (default) returns { text, usage }; \"verbose_json\" additionally returns task, language, duration, and segment-level timestamps (OpenAI-compatible providers only). :param session_id: A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence. + :param source_url: Publicly reachable http(s) URL of the audio file, downloaded by the provider directly (no size limit on our side). The format is derived from the URL path extension. Only supported by some providers; exactly one of file or source_url is required. :param temperature: The sampling temperature. :param timestamp_granularities: Timestamp detail levels to include when response_format is \"verbose_json\". \"word\" additionally returns word-level timestamps in the words array. :param trace: JSON-encoded trace metadata object (trace_id, trace_name, span_name, generation_name, parent_span_id and custom keys) attached to the Broadcast trace. Must decode to a JSON object. @@ -744,13 +780,17 @@ async def create_transcription_multipart_async( x_open_router_title=x_open_router_title, x_open_router_categories=x_open_router_categories, request_body=operations.CreateAudioTranscriptionsMultipartRequestBody( + diarize=diarize, file=utils.get_pydantic_model( - file, operations.CreateAudioTranscriptionsMultipartFile + file, Optional[operations.CreateAudioTranscriptionsMultipartFile] ), + keyterms=utils.unmarshal(keyterms, Optional[List[str]]), language=language, model=model, + provider=provider, response_format=response_format, session_id=session_id, + source_url=source_url, temperature=temperature, timestamp_granularities=utils.unmarshal( timestamp_granularities, diff --git a/uv.lock b/uv.lock index 439e359a..21aa38a3 100644 --- a/uv.lock +++ b/uv.lock @@ -213,7 +213,7 @@ wheels = [ [[package]] name = "openrouter" -version = "1.3.5" +version = "1.3.6" source = { editable = "." } dependencies = [ { name = "httpcore" }, From fda467ab0f4db4f073f29e6dc192e9df54a7552c Mon Sep 17 00:00:00 2001 From: "speakeasy-github[bot]" <128539517+speakeasy-github[bot]@users.noreply.github.com> Date: Tue, 29 Sep 2026 19:09:02 +0000 Subject: [PATCH 2/2] empty commit to trigger [run-tests] workflow