Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
140 changes: 132 additions & 8 deletions .speakeasy/in.openapi.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -6191,6 +6191,7 @@ components:
- 'deepseek'
- 'dekallm'
- 'digitalocean'
- 'elevenlabs'
- 'featherless'
- 'fireworks'
- 'fish-audio'
Expand Down Expand Up @@ -17308,6 +17309,7 @@ components:
- 'DeepSeek'
- 'DekaLLM'
- 'DigitalOcean'
- 'ElevenLabs'
- 'Featherless'
- 'Fireworks'
- 'Fish Audio'
Expand Down Expand Up @@ -24804,6 +24806,7 @@ components:
- 'DeepSeek'
- 'DekaLLM'
- 'DigitalOcean'
- 'ElevenLabs'
- 'Featherless'
- 'Fireworks'
- 'Fish Audio'
Expand Down Expand Up @@ -25017,6 +25020,9 @@ components:
digitalocean:
additionalProperties: {}
type: 'object'
elevenlabs:
additionalProperties: {}
type: 'object'
enfer:
additionalProperties: {}
type: 'object'
Expand Down Expand Up @@ -25580,6 +25586,7 @@ components:
- 'DeepSeek'
- 'DekaLLM'
- 'DigitalOcean'
- 'ElevenLabs'
- 'Featherless'
- 'Fireworks'
- 'Fish Audio'
Expand Down Expand Up @@ -28323,8 +28330,38 @@ components:
- 111
logprob: -0.5
token: 'Hello'
STTInputAudio:
description: 'Base64-encoded audio to transcribe'
STTEntity:
description: 'A detected entity, returned when the provider runs entity detection'
example:
end_char: 25
start_char: 15
text: 'John Smith'
type: 'name'
properties:
end_char:
description: 'Zero-based exclusive character offset of the entity end within the response-level text (not seconds)'
example: 25
type: 'integer'
start_char:
description: 'Zero-based character offset of the entity start within the response-level text (not seconds)'
example: 15
type: 'integer'
text:
description: 'Entity text as it appears in the transcript'
type: 'string'
type:
description: 'Provider entity type label'
example: 'name'
type: 'string'
required:
- 'text'
- 'type'
- 'start_char'
- 'end_char'
type: 'object'
STTInlineInputAudio:
additionalProperties: false
description: 'Inline base64 audio input for speech-to-text'
example:
data: 'UklGRiQA...'
format: 'wav'
Expand All @@ -28333,24 +28370,44 @@ components:
description: 'Base64-encoded audio data (raw bytes, not a data URI)'
type: 'string'
format:
description: 'Audio format (e.g., wav, mp3, flac, m4a, ogg, webm, aac). Supported formats vary by provider.'
description: 'Audio format (e.g., wav, mp3, flac, m4a, ogg, webm, aac). Supported formats vary by provider. "pcm" means headerless signed 16-bit little-endian mono audio at 16 kHz.'
pattern: '^[a-zA-Z0-9][a-zA-Z0-9+._-]{0,15}$'
type: 'string'
required:
- 'data'
- 'format'
type: 'object'
STTInputAudio:
anyOf:
- $ref: '#/components/schemas/STTInlineInputAudio'
- $ref: '#/components/schemas/STTUrlInputAudio'
description: 'Audio to transcribe: inline base64 bytes, or a URL the provider downloads directly.'
STTRequest:
description: 'Speech-to-text request input. Accepts a JSON body with input_audio containing base64-encoded audio.'
description: 'Speech-to-text request input. Accepts a JSON body with input_audio containing base64-encoded audio or a URL the provider downloads.'
example:
input_audio:
data: 'UklGRiQA...'
format: 'wav'
language: 'en'
model: 'openai/whisper-large-v3'
properties:
diarize:
description: 'Label each word with the speaker who said it. Speaker labels are returned on the words array (speaker, speaker_label), so response_format must be "verbose_json" (a "json" request is rejected with a 400) and word timestamps are included even when timestamp_granularities omits "word". Only supported by some providers; the request is rejected with a 400 when the selected model cannot diarize. Providers may charge extra.'
example: true
type: 'boolean'
input_audio:
$ref: '#/components/schemas/STTInputAudio'
keyterms:
description: 'Domain terms, names, or phrases to bias recognition toward. Only supported by some providers; the request is rejected with a 400 when the selected model cannot use keyterms. Providers may cap the number of terms or characters per term and may charge extra.'
example:
- 'OpenRouter'
- 'Scribe'
items:
maxLength: 100
minLength: 1
type: 'string'
maxItems: 1000
type: 'array'
language:
description: 'ISO-639-1 language code (e.g., "en", "ja"). Auto-detected if omitted.'
example: 'en'
Expand Down Expand Up @@ -28421,10 +28478,20 @@ components:
example: 9.2
format: 'double'
type: 'number'
entities:
description: 'Detected entities with character offsets into text, present when the provider runs entity detection'
items:
$ref: '#/components/schemas/STTEntity'
type: 'array'
language:
description: 'Detected or forced language, present when response_format is verbose_json'
example: 'english'
type: 'string'
language_confidence:
description: 'Provider confidence in the detected language from 0 to 1, present when response_format is verbose_json and the provider scores language detection'
example: 0.98
format: 'double'
type: 'number'
segments:
description: 'Timestamped transcript segments, present when response_format is verbose_json'
items:
Expand Down Expand Up @@ -28470,6 +28537,10 @@ components:
description: 'Average log probability of the segment'
format: 'double'
type: 'number'
channel:
description: 'Zero-based audio channel index for the segment, present when the provider transcribes channels separately'
example: 0
type: 'integer'
compression_ratio:
description: 'Compression ratio of the segment'
format: 'double'
Expand All @@ -28495,6 +28566,10 @@ components:
description: 'Speaker index for the segment, present when the provider returns diarization data'
example: 0
type: 'integer'
speaker_label:
description: 'Provider speaker label for the segment, present when the provider labels speakers with a string'
example: 'speaker_0'
type: 'string'
start:
description: 'Segment start time in seconds'
example: 0
Expand Down Expand Up @@ -28526,6 +28601,25 @@ components:
- 'segment'
example: 'word'
type: 'string'
STTUrlInputAudio:
additionalProperties: false
description: 'Audio input fetched by the provider from a URL'
example:
format: 'mp3'
url: 'https://example.com/meeting.mp3'
properties:
format:
description: 'Audio format of the file at the URL. Defaults to the extension of the URL path; required when the path has no extension.'
pattern: '^[a-zA-Z0-9][a-zA-Z0-9+._-]{0,15}$'
type: 'string'
url:
description: 'Publicly reachable http(s) URL of the audio file. The provider downloads it directly, so the inline upload size limit does not apply. Only supported by some providers.'
format: 'uri'
maxLength: 8000
type: 'string'
required:
- 'url'
type: 'object'
STTUsage:
description: 'Aggregated usage statistics for the request'
example:
Expand Down Expand Up @@ -28567,6 +28661,10 @@ components:
start: 0
word: 'Hello'
properties:
channel:
description: 'Zero-based audio channel index for the word, present when the provider transcribes channels separately'
example: 0
type: 'integer'
confidence:
description: 'Provider confidence for the word from 0 to 1, present when the provider returns per-word confidence'
example: 0.98
Expand All @@ -28581,13 +28679,24 @@ components:
description: 'Speaker index for the word, present when the provider returns diarization data'
example: 0
type: 'integer'
speaker_label:
description: 'Provider speaker label for the word, present when the provider labels speakers with a string'
example: 'speaker_0'
type: 'string'
start:
description: 'Word start time in seconds'
example: 0
format: 'double'
type: 'number'
type:
description: 'Kind of entry; omitted or "word" for spoken words, "audio_event" for non-speech sounds the provider tags with timestamps'
enum:
- 'word'
- 'audio_event'
example: 'word'
type: 'string'
word:
description: 'The transcribed word'
description: 'The transcribed word, or the event tag such as "(laughter)" when type is audio_event'
example: 'Hello'
type: 'string'
required:
Expand Down Expand Up @@ -32948,7 +33057,7 @@ paths:
x-speakeasy-name-override: 'createSpeech'
/audio/transcriptions:
post:
description: 'Transcribes audio into text. Accepts base64-encoded audio input as JSON or an OpenAI-style multipart/form-data file upload, and returns the transcribed text.'
description: 'Transcribes audio into text. Accepts base64-encoded audio input as JSON, an OpenAI-style multipart/form-data file upload, or a URL the provider downloads directly, and returns the transcribed text.'
operationId: 'createAudioTranscriptions'
requestBody:
content:
Expand All @@ -32968,16 +33077,27 @@ paths:
model: 'openai/whisper-large-v3'
schema:
properties:
diarize:
description: 'Label each word with the speaker who said it (words[].speaker, words[].speaker_label). Requires response_format "verbose_json" (400 otherwise); word timestamps are included even when timestamp_granularities[] omits "word". Only supported by some providers; 400 when the selected model cannot diarize.'
type: 'boolean'
file:
description: 'The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio.'
description: 'The audio file to transcribe. The format is derived from the filename extension or the file part content type. Max 25 MB; send larger files as base64 JSON via input_audio, or by URL via source_url. Exactly one of file or source_url is required.'
format: 'binary'
type: 'string'
keyterms[]:
description: 'Domain terms, names, or phrases to bias recognition toward; repeat the part once per term (keyterms=... is also accepted). Only supported by some providers; 400 when the selected model cannot use keyterms.'
items:
type: 'string'
type: 'array'
language:
description: 'The language of the input audio (ISO-639-1).'
type: 'string'
model:
description: 'The model to use for transcription.'
type: 'string'
provider:
description: 'JSON-encoded provider preferences object, the same shape as the JSON body field: { "options": { "<provider-slug>": { ... } } }. Only options for the matched provider are forwarded. Must decode to a JSON object.'
type: 'string'
response_format:
description: 'The response format. "json" (default) returns { text, usage }; "verbose_json" additionally returns task, language, duration, and segment-level timestamps (OpenAI-compatible providers only).'
enum:
Expand All @@ -32988,6 +33108,10 @@ paths:
description: 'A unique identifier for grouping related requests (e.g., a conversation or agent workflow). Used for observability grouping in Broadcast and private logging; never sent to the provider. If provided in both the request body and the x-session-id header, the body value takes precedence.'
maxLength: 256
type: 'string'
source_url:
description: 'Publicly reachable http(s) URL of the audio file, downloaded by the provider directly (no size limit on our side). The format is derived from the URL path extension. Only supported by some providers; exactly one of file or source_url is required.'
format: 'uri'
type: 'string'
temperature:
description: 'The sampling temperature.'
type: 'number'
Expand All @@ -33007,7 +33131,6 @@ paths:
maxLength: 256
type: 'string'
required:
- 'file'
- 'model'
type: 'object'
required: true
Expand Down Expand Up @@ -34173,6 +34296,7 @@ paths:
- 'deepseek'
- 'dekallm'
- 'digitalocean'
- 'elevenlabs'
- 'featherless'
- 'fireworks'
- 'fish-audio'
Expand Down
Loading