curl https://api.snapgen.org/v1/audio/transcriptions \
-H "Authorization: Bearer $SNAPGEN_API_KEY" \
-F model=scribe-v2 \
-F file=@interview.mp3 \
-F diarize=true \
-F subtitles=true
curl https://api.snapgen.org/v1/audio/transcriptions \
-H "Authorization: Bearer $SNAPGEN_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "scribe-v2",
"audio_url": "https://example.com/interview.mp3",
"diarize": true,
"subtitles": true
}'
import os
import requests
with open("interview.mp3", "rb") as audio:
response = requests.post(
"https://api.snapgen.org/v1/audio/transcriptions",
headers={"Authorization": f"Bearer {os.environ['SNAPGEN_API_KEY']}"},
data={"model": "scribe-v2", "diarize": "true", "subtitles": "true"},
files={"file": ("interview.mp3", audio, "audio/mpeg")},
timeout=600,
)
response.raise_for_status()
transcript = response.json()
print(transcript["text"])
print(transcript.get("srt"))
import { openAsBlob } from "node:fs";
const form = new FormData();
form.append("model", "scribe-v2");
form.append("file", await openAsBlob("interview.mp3"), "interview.mp3");
form.append("diarize", "true");
form.append("subtitles", "true");
const response = await fetch("https://api.snapgen.org/v1/audio/transcriptions", {
method: "POST",
headers: { Authorization: `Bearer ${process.env.SNAPGEN_API_KEY}` },
body: form,
});
if (!response.ok) throw new Error(await response.text());
const transcript = await response.json();
console.log(transcript.text, transcript.srt);
{
"text": "Welcome to the show. Thanks for having me.",
"language": "eng",
"language_probability": 0.99,
"duration_seconds": 12.34,
"srt": "1\n00:00:00,120 --> 00:00:01,480\nWelcome to the show.\n\n2\n00:00:01,900 --> 00:00:03,020\nThanks for having me.\n"
}
{
"text": "Welcome to the show.",
"language": "eng",
"language_probability": 0.99,
"duration_seconds": 1.6,
"words": [
{ "text": "Welcome", "type": "word", "start": 0.12, "end": 0.52, "speaker": "speaker_0" },
{ "text": " ", "type": "spacing", "start": 0.52, "end": 0.56, "speaker": "speaker_0" },
{ "text": "to", "type": "word", "start": 0.56, "end": 0.68, "speaker": "speaker_0" }
]
}
{
"error": {
"message": "Upload a file or send audio_url",
"type": "invalid_request_error",
"param": null,
"code": "audio_input_required"
}
}
Endpoint Reference
Speech to text
Transcribe an uploaded file or a public audio URL with Scribe v2, with word timings, speaker labels, and SRT subtitles.
POST
/
v1
/
audio
/
transcriptions
curl https://api.snapgen.org/v1/audio/transcriptions \
-H "Authorization: Bearer $SNAPGEN_API_KEY" \
-F model=scribe-v2 \
-F file=@interview.mp3 \
-F diarize=true \
-F subtitles=true
curl https://api.snapgen.org/v1/audio/transcriptions \
-H "Authorization: Bearer $SNAPGEN_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "scribe-v2",
"audio_url": "https://example.com/interview.mp3",
"diarize": true,
"subtitles": true
}'
import os
import requests
with open("interview.mp3", "rb") as audio:
response = requests.post(
"https://api.snapgen.org/v1/audio/transcriptions",
headers={"Authorization": f"Bearer {os.environ['SNAPGEN_API_KEY']}"},
data={"model": "scribe-v2", "diarize": "true", "subtitles": "true"},
files={"file": ("interview.mp3", audio, "audio/mpeg")},
timeout=600,
)
response.raise_for_status()
transcript = response.json()
print(transcript["text"])
print(transcript.get("srt"))
import { openAsBlob } from "node:fs";
const form = new FormData();
form.append("model", "scribe-v2");
form.append("file", await openAsBlob("interview.mp3"), "interview.mp3");
form.append("diarize", "true");
form.append("subtitles", "true");
const response = await fetch("https://api.snapgen.org/v1/audio/transcriptions", {
method: "POST",
headers: { Authorization: `Bearer ${process.env.SNAPGEN_API_KEY}` },
body: form,
});
if (!response.ok) throw new Error(await response.text());
const transcript = await response.json();
console.log(transcript.text, transcript.srt);
{
"text": "Welcome to the show. Thanks for having me.",
"language": "eng",
"language_probability": 0.99,
"duration_seconds": 12.34,
"srt": "1\n00:00:00,120 --> 00:00:01,480\nWelcome to the show.\n\n2\n00:00:01,900 --> 00:00:03,020\nThanks for having me.\n"
}
{
"text": "Welcome to the show.",
"language": "eng",
"language_probability": 0.99,
"duration_seconds": 1.6,
"words": [
{ "text": "Welcome", "type": "word", "start": 0.12, "end": 0.52, "speaker": "speaker_0" },
{ "text": " ", "type": "spacing", "start": 0.52, "end": 0.56, "speaker": "speaker_0" },
{ "text": "to", "type": "word", "start": 0.56, "end": 0.68, "speaker": "speaker_0" }
]
}
{
"error": {
"message": "Upload a file or send audio_url",
"type": "invalid_request_error",
"param": null,
"code": "audio_input_required"
}
}
POST /v1/audio/transcriptions transcribes speech with scribe-v2. Upload the file as multipart/form-data, the same way as the OpenAI transcription API, or send JSON with a public audio_url that the gateway downloads. The response is JSON with the transcript and, on request, word timings, speaker labels, and SRT subtitles.
| Model | Input | Price |
|---|---|---|
scribe-v2 | Up to 7,200 seconds (2 hours) and 50 MB | $0.000092 per second ($0.3312 per hour) |
string
required
Set to
scribe-v2.file
The audio file, in a multipart field named
file, up to 50 MB. Send file
or audio_url, not both.string
Public
http or https URL of the audio. The gateway downloads it over a
public-only connection, up to 50 MB. Send audio_url or file, not both.string
Language of the audio as an ISO 639 code of two or three lowercase letters,
such as
en. Omit it to detect the language automatically.boolean
Labels who speaks each word in
words[].speaker.integer
The most speakers in the audio, from
1 to 32. Helps speaker labeling.boolean
Tags sounds such as laughter or footsteps in the transcript.
string
none, word, or character. word and character add words to the
response with word-level timings; per-character timings aren’t returned.boolean
true adds SubRip (SRT) subtitles in srt. Subtitles need speaker labels
and word timings, so the gateway turns both on at the provider. To also get
words in the response, send timestamps: "word".number
Randomness, from
0 to 2. Higher values give more varied output.string
default:"json"
json or verbose_json. verbose_json adds words unless timestamps is
none. Both return the SnapGen JSON shape below, not OpenAI’s verbose schema.Send the audio
- Multipart upload: send exactly one
modelfield and one file in thefilefield. Send the other fields as text form fields;true,false, and numbers are converted for you. - JSON: send
Content-Type: application/jsonwithaudio_urland the other fields.
unsupported_audio_format before any balance is reserved.
Response
string
The full transcript.
string | null
The language code the provider detected or used, for example
eng.number | null
Confidence in the detected language, from 0 to 1.
number
Length of the audio in seconds.
object[]
Present when
timestamps is word or character, or with
response_format: "verbose_json".string
SubRip subtitles. Present when
subtitles is true.x-gateway-charge-microusd header.
Billing
The gateway bills the measured length of the audio at $0.000092 per second. It rounds up to whole seconds after a 0.1-second allowance for container padding, with a minimum of 1 second: a 12.34-second file bills 13 seconds ($0.001196). A one-hour recording costs $0.3312. Options such assubtitles
and diarize don’t change the price, and failed requests aren’t charged.
This endpoint rejects Idempotency-Key. If a request times out, check your
Console request logs before you retry.
Errors
| Status | error.code | Cause |
|---|---|---|
| 400 | audio_input_required | Neither file nor audio_url was sent. |
| 400 | invalid_request | A field is unknown or invalid, or both file and audio_url were sent. |
| 400 | invalid_multipart | The form has no model field, more than one file, a repeated field, or the file isn’t in the file field. |
| 400 | unsupported_audio_format | The format isn’t supported, or its length can’t be read. |
| 400 | audio_too_long | The audio is longer than 7,200 seconds. |
| 400 | audio_download_failed | The gateway couldn’t download audio_url. |
| 400 | idempotency_not_supported | The request has an Idempotency-Key header. |
| 400 | provider_rejected_request | The provider refused the audio or a parameter. |
| 402 | insufficient_funds | Your balance can’t cover the request. |
| 413 | audio_too_large, request_too_large | The downloaded file or the upload is larger than 50 MB. |
| 503 | no_eligible_route | The model doesn’t run on this endpoint, or no route is available right now. |
curl https://api.snapgen.org/v1/audio/transcriptions \
-H "Authorization: Bearer $SNAPGEN_API_KEY" \
-F model=scribe-v2 \
-F file=@interview.mp3 \
-F diarize=true \
-F subtitles=true
curl https://api.snapgen.org/v1/audio/transcriptions \
-H "Authorization: Bearer $SNAPGEN_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "scribe-v2",
"audio_url": "https://example.com/interview.mp3",
"diarize": true,
"subtitles": true
}'
import os
import requests
with open("interview.mp3", "rb") as audio:
response = requests.post(
"https://api.snapgen.org/v1/audio/transcriptions",
headers={"Authorization": f"Bearer {os.environ['SNAPGEN_API_KEY']}"},
data={"model": "scribe-v2", "diarize": "true", "subtitles": "true"},
files={"file": ("interview.mp3", audio, "audio/mpeg")},
timeout=600,
)
response.raise_for_status()
transcript = response.json()
print(transcript["text"])
print(transcript.get("srt"))
import { openAsBlob } from "node:fs";
const form = new FormData();
form.append("model", "scribe-v2");
form.append("file", await openAsBlob("interview.mp3"), "interview.mp3");
form.append("diarize", "true");
form.append("subtitles", "true");
const response = await fetch("https://api.snapgen.org/v1/audio/transcriptions", {
method: "POST",
headers: { Authorization: `Bearer ${process.env.SNAPGEN_API_KEY}` },
body: form,
});
if (!response.ok) throw new Error(await response.text());
const transcript = await response.json();
console.log(transcript.text, transcript.srt);
{
"text": "Welcome to the show. Thanks for having me.",
"language": "eng",
"language_probability": 0.99,
"duration_seconds": 12.34,
"srt": "1\n00:00:00,120 --> 00:00:01,480\nWelcome to the show.\n\n2\n00:00:01,900 --> 00:00:03,020\nThanks for having me.\n"
}
{
"text": "Welcome to the show.",
"language": "eng",
"language_probability": 0.99,
"duration_seconds": 1.6,
"words": [
{ "text": "Welcome", "type": "word", "start": 0.12, "end": 0.52, "speaker": "speaker_0" },
{ "text": " ", "type": "spacing", "start": 0.52, "end": 0.56, "speaker": "speaker_0" },
{ "text": "to", "type": "word", "start": 0.56, "end": 0.68, "speaker": "speaker_0" }
]
}
{
"error": {
"message": "Upload a file or send audio_url",
"type": "invalid_request_error",
"param": null,
"code": "audio_input_required"
}
}