curl --request POST \
--url https://api.fish.audio/v1/asr \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: multipart/form-data' \
--form audio=@example-file \
--form 'language=<string>' \
--form ignore_timestamps=trueimport requests
url = "https://api.fish.audio/v1/asr"
files = { "audio": ("example-file", open("example-file", "rb")) }
payload = {
"language": "<string>",
"ignore_timestamps": "true"
}
headers = {"Authorization": "Bearer <token>"}
response = requests.post(url, data=payload, files=files, headers=headers)
print(response.text)const form = new FormData();
form.append('audio', '<string>');
form.append('language', '<string>');
form.append('ignore_timestamps', 'true');
const options = {method: 'POST', headers: {Authorization: 'Bearer <token>'}};
options.body = form;
fetch('https://api.fish.audio/v1/asr', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.fish.audio/v1/asr",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => "-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"audio\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"ignore_timestamps\"\r\n\r\ntrue\r\n-----011000010111000001101001--",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: multipart/form-data"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.fish.audio/v1/asr"
payload := strings.NewReader("-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"audio\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"ignore_timestamps\"\r\n\r\ntrue\r\n-----011000010111000001101001--")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.fish.audio/v1/asr")
.header("Authorization", "Bearer <token>")
.body("-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"audio\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"ignore_timestamps\"\r\n\r\ntrue\r\n-----011000010111000001101001--")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.fish.audio/v1/asr")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request.body = "-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"audio\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"ignore_timestamps\"\r\n\r\ntrue\r\n-----011000010111000001101001--"
response = http.request(request)
puts response.read_body{
"text": "<string>",
"duration": 123,
"segments": [
{
"text": "<string>",
"start": 123,
"end": 123
}
],
"language_code": "<string>",
"language": "<string>"
}{
"status": 123,
"message": "<string>",
"reason": "<string>"
}{
"status": 123,
"message": "<string>",
"reason": "<string>"
}{
"status": 123,
"message": "<string>",
"reason": "<string>"
}Speech to Text
Transcribe audio with transcribe-1-pro, the recommended speech-to-text model: request and response fields, speaker turns, limits, formats, and errors
curl --request POST \
--url https://api.fish.audio/v1/asr \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: multipart/form-data' \
--form audio=@example-file \
--form 'language=<string>' \
--form ignore_timestamps=trueimport requests
url = "https://api.fish.audio/v1/asr"
files = { "audio": ("example-file", open("example-file", "rb")) }
payload = {
"language": "<string>",
"ignore_timestamps": "true"
}
headers = {"Authorization": "Bearer <token>"}
response = requests.post(url, data=payload, files=files, headers=headers)
print(response.text)const form = new FormData();
form.append('audio', '<string>');
form.append('language', '<string>');
form.append('ignore_timestamps', 'true');
const options = {method: 'POST', headers: {Authorization: 'Bearer <token>'}};
options.body = form;
fetch('https://api.fish.audio/v1/asr', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.fish.audio/v1/asr",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => "-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"audio\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"ignore_timestamps\"\r\n\r\ntrue\r\n-----011000010111000001101001--",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: multipart/form-data"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.fish.audio/v1/asr"
payload := strings.NewReader("-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"audio\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"ignore_timestamps\"\r\n\r\ntrue\r\n-----011000010111000001101001--")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.fish.audio/v1/asr")
.header("Authorization", "Bearer <token>")
.body("-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"audio\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"ignore_timestamps\"\r\n\r\ntrue\r\n-----011000010111000001101001--")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.fish.audio/v1/asr")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request.body = "-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"audio\"; filename=\"example-file\"\r\nContent-Type: application/octet-stream\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"language\"\r\n\r\n<string>\r\n-----011000010111000001101001\r\nContent-Disposition: form-data; name=\"ignore_timestamps\"\r\n\r\ntrue\r\n-----011000010111000001101001--"
response = http.request(request)
puts response.read_body{
"text": "<string>",
"duration": 123,
"segments": [
{
"text": "<string>",
"start": 123,
"end": 123
}
],
"language_code": "<string>",
"language": "<string>"
}{
"status": 123,
"message": "<string>",
"reason": "<string>"
}{
"status": 123,
"message": "<string>",
"reason": "<string>"
}{
"status": 123,
"message": "<string>",
"reason": "<string>"
}multipart/form-data (a file upload) or application/msgpack. Base64-encoded
audio in a JSON body is not supported.Select a model
SendPOST https://api.fish.audio/v1/asr with Authorization: Bearer <API_KEY> and a model HTTP header that selects the model. The request is synchronous: the response arrives when the whole file has been transcribed.
| Header value | Use it for |
|---|---|
transcribe-1-pro | Recommended. Recordings up to 60 minutes, including multi-speaker conversations: speaker markers and speaker turns, with emotion and vocal-event cues preserved. |
transcribe-1 | General transcription of short recordings. Used when the model header is missing or not an exact match. |
model: transcribe-1-pro with every request, written exactly, in lowercase. If the header is missing, or its value is not an exact match (for example Transcribe-1-Pro or transcribe-1pro), the request is served and billed as transcribe-1, and no error is returned. If you expected transcribe-1-pro but the transcript has no speaker markers, check the header. This applies to the API playground on this page too: set its model header to transcribe-1-pro.
The model belongs in the header for both multipart and MessagePack requests; a model field in the request body is ignored. See Limits for recording length and request size.
To link the request to your own distributed trace, also send a W3C traceparent header. See Tracing & Performance Analysis.
Request fields
Multipart and MessagePack bodies use the same field names. Multipart values are text; MessagePack values are typed.| Field | Required | Default | Description |
|---|---|---|---|
audio | Yes | — | Exactly one audio file per request: a file upload for multipart, or binary bytes for MessagePack. The format is detected from the file contents, not the file name. See Supported audio formats. |
language | No | Unset | Optional language hint, as a lowercase ISO 639-1 code such as en, zh, or ja. The language is detected automatically either way. An empty multipart value counts as unset. See Language. |
ignore_timestamps | No | true | Default true (no timestamps). Set to false to get word-level segments and speaker_turns. Timestamps add processing time. |
tag_audio_events | No | true | true: bracketed emotion and vocal-event cues such as [laughter] or [高兴] stay in text and in speaker_turns. false: the cues are removed. Timestamps, duration, and billing do not change. |
diarize | No | auto | auto or true: return speaker_turns when timestamps are requested. false: omit speaker_turns. false does not change the transcript, and the speaker markers stay in text. |
num_speakers | No | Unset | The expected number of speakers, an integer of 1 or more. Cannot be combined with min_speakers or max_speakers. |
min_speakers, max_speakers | No | Unset | Lower and upper bounds on the number of speakers, integers of 1 or more. min_speakers must not exceed max_speakers. |
true or false for ignore_timestamps. Any value other than true (case-insensitive), including 1 or an empty value, is read as false and turns timestamps on. In MessagePack, send a boolean; a string or nil returns 400.
Speaker-count hints are applied on a best-effort basis: they guide speaker identification on longer recordings and may have no effect on short ones. They cannot be sent with diarize=false.
tag_audio_events, diarize, and the speaker-count hints are validated strictly:
- In multipart forms,
tag_audio_eventstakestrueorfalse, anddiarizetakesauto,true, orfalse, in any letter case. Speaker counts take digits only. Send each field at most once. - In MessagePack, send
tag_audio_eventsas a boolean,diarizeas a boolean or one of the lowercase stringsauto,true, orfalse, and speaker counts as integers, not floats or strings.nilmeans the default. - Any other value, a repeated multipart field, or a conflicting combination returns 400 with code
invalid_parameter.
transcribe-1: send audio, language, and ignore_timestamps. The other fields apply to transcribe-1-pro; send them only with model: transcribe-1-pro.
Examples
curl --request POST https://api.fish.audio/v1/asr \
--header "Authorization: Bearer $FISH_API_KEY" \
--header "model: transcribe-1-pro" \
--form audio=@conversation.wav \
--form ignore_timestamps=false
segments and speaker_turns. For a two-person interview, you can add a speaker-count hint, and remove emotion and vocal-event cues from the transcript:
curl --request POST https://api.fish.audio/v1/asr \
--header "Authorization: Bearer $FISH_API_KEY" \
--header "model: transcribe-1-pro" \
--form audio=@interview.mp3 \
--form ignore_timestamps=false \
--form num_speakers=2 \
--form tag_audio_events=false
Content-Type and boundary. For MessagePack, set Content-Type: application/msgpack and encode audio as binary bytes. See the Speech to Text guide for an example.
Read the response
A successful response (HTTP 200) is a JSON object:| Field | Description |
|---|---|
text | The full transcript, with inline speaker markers such as <|speaker:0|> and, unless tag_audio_events=false, bracketed emotion and vocal-event cues such as [高兴] (happy) or [laughter]. See Multi-speaker response format. |
duration | Length of the audio in seconds, including silence. |
segments | Word-level timestamps: an array of { "text", "start", "end" }, with times in seconds. Empty ([]) when ignore_timestamps=true (the default), when no speech was found, or when timing is temporarily unavailable. See Segments. |
speaker_turns | The transcript as timed speaker turns. Returned when ignore_timestamps=false and diarize is not false; otherwise absent. See Speaker turns. |
language | The detected language’s English name, such as English or Chinese. Omitted when it cannot be determined. Intended for display. |
language_code | Two-letter ISO 639-1 code for the language, such as en. Omitted when unknown. If the language cannot be determined and you sent a language hint, this reports your hint. See Language. |
request_id | A unique ID for this request, also present in transcribe-1-pro error bodies and in the x-request-id response header. Platform and network-edge errors do not carry it. Include it when you contact support. |
null. Do not depend on the order of keys in the JSON.
If you use transcribe-1: rely on text, duration, segments, language, and language_code. Speaker markers, speaker_turns, and request_id are transcribe-1-pro features.
transcribe-1-pro through the model header. In Python,
pass
request_options=RequestOptions(additional_headers={"model": "transcribe-1-pro"})
to asr.transcribe(). In JavaScript, pass
{ headers: { model: "transcribe-1-pro" } } as the second argument to
speechToText.convert(). The Python SDK currently returns only text,
duration, and segments, and neither SDK can send tag_audio_events,
diarize, or the speaker-count hints. To use them, call the API directly, as
in the examples above. The JavaScript SDK returns the whole response body,
but its STTResponse type declares only text, duration, and segments.Multi-speaker response format
transcribe-1-pro marks speaker changes inside the text string with <|speaker:N|> markers. The label N identifies the speaker for the text that follows, up to the next marker. A repeated label means that speaker is speaking again.
- Markers usually have a space on each side (
<|speaker:0|> 你好。 <|speaker:1|> ...). Do not depend on exact spacing. - Any text before the first marker belongs to the first turn.
- A transcript with no marker comes from a single speaker; treat it as speaker 0.
- Labels identify speakers within one response only. They are not names, and the same label in two requests is not the same person. Send a whole conversation as one file.
ignore_timestamps=true (the default), timestamps are skipped and speaker_turns is absent:
{
"text": "<|speaker:0|> 你好。 <|speaker:1|> [高兴]很开心认识你。 <|speaker:0|> 我也是。",
"duration": 6.4,
"segments": [],
"language_code": "zh",
"language": "Chinese",
"request_id": "5f0c7e2a-9b1d-4c3e-8f6a-2d4b7c9e1a03"
}
你好。, speaker 1 saying [高兴]很开心认识你。, and speaker 0 saying 我也是。. The [高兴] cue describes the delivery of speaker 1’s speech. There is no separate emotion field; the cues are part of the text.
Speaker turns
Withignore_timestamps=false, transcribe-1-pro also returns the turns as a structured speaker_turns array, unless you send diarize=false. When you request timestamps, prefer speaker_turns over parsing text.
speaker_turnslists the turns in the order they occur, and is[]when no speech was found. Each turn is{ "speaker": "speaker:N", "text": "...", "start": <seconds>, "end": <seconds> }.speakeris the stringspeaker:N, whereNmatches the<|speaker:N|>marker intext. Like the markers, labels identify speakers within one response only.textis that turn’s speech without speaker markers. It keeps emotion and vocal-event cues unlesstag_audio_events=false.startandendcome from the word timestamps. If word timing is unavailable (segmentsis empty), turn times are approximate and can cover the whole recording.- Consecutive turns can have the same speaker. Do not assume turns are contiguous or non-overlapping.
ignore_timestamps=false:
{
"text": "<|speaker:0|> 你好。 <|speaker:1|> [高兴]很开心认识你。 <|speaker:0|> 我也是。",
"duration": 6.4,
"segments": [
{ "text": "你", "start": 0.24, "end": 0.52 },
{ "text": "好", "start": 0.52, "end": 0.8 },
{ "text": "很", "start": 1.6, "end": 1.84 },
{ "text": "开", "start": 1.84, "end": 2.1 },
{ "text": "心", "start": 2.1, "end": 2.36 },
{ "text": "认", "start": 2.36, "end": 2.7 },
{ "text": "识", "start": 2.7, "end": 2.98 },
{ "text": "你", "start": 2.98, "end": 3.3 },
{ "text": "我", "start": 4.5, "end": 4.78 },
{ "text": "也", "start": 4.78, "end": 5.02 },
{ "text": "是", "start": 5.02, "end": 5.4 }
],
"speaker_turns": [
{ "speaker": "speaker:0", "text": "你好。", "start": 0.24, "end": 0.8 },
{
"speaker": "speaker:1",
"text": "[高兴]很开心认识你。",
"start": 1.6,
"end": 3.3
},
{ "speaker": "speaker:0", "text": "我也是。", "start": 4.5, "end": 5.4 }
],
"language_code": "zh",
"language": "Chinese",
"request_id": "5f0c7e2a-9b1d-4c3e-8f6a-2d4b7c9e1a03"
}
Segments
A segment is usually one word; in Chinese and Japanese it is usually one character, or a few. Segment text has no punctuation and no speaker markers or cues, and can be normalized (for example35 for 3.5), so it does not always match text character for character. A segment can have start equal to end. Segments are not speaker turns and carry no speaker ID.
Language
languageis optional. The language is detected automatically whether or not you send a hint, and a hint does not force the transcript into that language.- Use a lowercase ISO 639-1 code such as
en,zh, orja. Other forms, such asen-USorEnglish, may be rejected with 400. - If the language cannot be determined (for example, very short audio),
language_codereports your hint, which is not checked against the audio, andlanguagemay be absent. - Responses report one language. For recordings that switch languages,
languageandlanguage_codedo not list every language spoken.
Limits
- Audio length: up to 60 minutes per request. Longer audio returns 400 with code
audio_too_long. - Request size: for long recordings, send compressed audio such as MP3, Opus, or AAC. An hour of 128 kbps MP3 is about 55 MiB. A request that is too large returns 413.
- One conversation per request: send the whole recording in one request. Timestamps and speaker labels then cover the whole recording. Do not split a conversation, because labels are consistent only within one response.
transcribe-1: it is designed for short recordings; for recordings longer than a few minutes, use transcribe-1-pro. It accepts up to 50 MiB per request; keep MP3 and Opus files under 25 MiB. A request that exceeds the size limit returns 413 or 400. A long request can fail with 503 if it exceeds the processing-time limit. Retry, and if it keeps failing, use transcribe-1-pro or split the audio.
Processing time and timeouts
- Processing time grows with the length of the recording. Long
transcribe-1-prorecordings can take several minutes. - Set your HTTP client’s timeout accordingly. The official SDKs’ default timeout can be too short for long recordings, and many HTTP libraries default to even less (httpx defaults to 5 seconds). For long recordings, use a generous timeout, for example 15 minutes:
httpx.Timeout(900.0, connect=10.0)with httpx,RequestOptions(timeout=900)with the Python SDK, or{ timeoutInSeconds: 900 }with the JavaScript SDK. - In Node.js, the built-in
fetch, which the JavaScript SDK uses, also stops waiting for a response after 5 minutes, even if you set a longer timeout. To wait longer, callsetGlobalDispatcher(new Agent({ headersTimeout: 900_000, bodyTimeout: 900_000 }))from theundicipackage once at startup. See Processing time and timeouts. - If a long request fails with a 5xx error or the connection drops, retry it.
- Each request holds one of your account’s concurrent request slots until its response is returned.
Supported audio formats
transcribe-1-proaccepts WAV, MP3, AAC (including M4A/MP4), FLAC, Ogg (Opus or Vorbis), WebM/Matroska, and MOV, including browser recordings. For a video file, it uses the first audio track. Send the original file bytes; no conversion is needed.- Not supported: AIFF, CAF, WMA, AMR, AC-3, and raw (headerless) PCM. These return 400.
transcribe-1: it accepts WAV, MP3, AAC (including M4A/MP4), FLAC, and Ogg (Opus or Vorbis). Browser WebM recordings (for example, from MediaRecorder) may not be accepted; convert them to Ogg/Opus, MP3, or WAV first.
Errors
Error responses have one of these shapes:transcribe-1-proerrors are JSON withstatus,message,code, andrequest_id. Branch oncode, not onmessage.- Platform errors, such as authentication, credit, and concurrency errors, are returned before the request reaches a model. They are JSON with exactly
statusandmessage. - Errors from the network edge in front of the API, such as some 413 and 5xx responses, may not have a JSON body. Handle them by HTTP status.
transcribe-1-pro error body shows the shape:
{
"status": 400,
"message": "A human-readable description of the problem.",
"code": "invalid_parameter",
"request_id": "5f0c7e2a-9b1d-4c3e-8f6a-2d4b7c9e1a03"
}
| Status | code | Meaning | Retry? |
|---|---|---|---|
| 400 | invalid_request | Malformed request: the body could not be parsed, audio is missing, or more than one audio was sent. | No |
| 400 | invalid_parameter | A parameter has a value it does not accept, or parameters conflict. | No |
| 400 | invalid_audio | The audio could not be decoded, or the format is not supported. | No |
| 400 | audio_too_long | The audio is longer than 60 minutes. | No |
| 400 | audio_too_short | The audio is too short to transcribe (under about 0.08 seconds). | No |
| 400 | — | The request body could not be read. | No |
| 401 | — | Missing or invalid API key. | No |
| 402 | — | Insufficient API credit. | No |
| 413 | request_too_large | The request is too large. It may come from the network edge without a JSON body. | No |
| 415 | unsupported_media_type | Unsupported Content-Type. | No |
| 429 | — (rarely upstream_rejected) | Usually your account is at its concurrency limit. No Retry-After header is sent. | Yes, with backoff |
| Other 4xx (not 429) | upstream_rejected | Rare: the request was rejected during transcription. | No |
| 500 | internal_error | Internal error. | Yes, with backoff |
| 502 | — | The service could not be reached. | Yes, with backoff |
| 503 | upstream_unavailable | The transcription service is temporarily unavailable. | Yes, with backoff |
| 503 | upstream_timeout | Processing took too long. | Yes, with backoff |
| 503 | excessive_repetition | The transcription produced unusable repeated output. | Yes, with backoff |
| 503 | diarization_failed | Speaker identification failed for this recording. | Yes, with backoff |
| 504 | — | Timed out connecting to the service. | Yes, with backoff |
- A dash means the error has no
code; a 503 without acodealso means the service is temporarily unavailable. - New
codevalues may be added; handle unknown codes by HTTP status. - Retry 429 and 5xx responses with exponential backoff. Do not retry other 4xx responses; change the request first.
- Requests that return an error response are not billed.
- With the Python SDK, the raw error body is in the exception’s
bodyattribute; parse it withjson.loadsto readcodeandrequest_id.
transcribe-1: its errors are JSON with status and message. Rely only on these two fields and on the HTTP status.
See Errors for retry and SDK exception examples.Authorizations
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Headers
Specify which speech-to-text model to use.
transcribe-1, transcribe-1-pro Body
Audio file to be converted to text
Optional hint. The language is auto-detected regardless; the detected language is returned as language_code.
Whether to return precise timestamps in the text, this will increase the latency in audio shorter than 30 seconds
Response
Request fulfilled, document follows
Duration of the audio in seconds
Show child attributes
Show child attributes
Detected language as an ISO 639-1 code (e.g. en, ja). Omitted if no language is detected.
Detected language name (e.g. English). For display only; use language_code in code.
Was this page helpful?

