AssemblyAI · ASR
Universal-3 Pro Batch
High-accuracy speech recognition for multilingual audio and long-form transcription.
universal-3-pro-batchCurrent model-specific price from our catalog. Requests require your project API key and sufficient balance. Availability is checked at execution time.
Input and capabilities
Upload a short audio or video file using multipart/form-data: at most 30 seconds and 20 MiB. Use WAV for the examples below. Container compatibility depends on the model.
- transcription
- diarization
- word timestamps
- segment timestamps
- hotwords
Automatic language detection: supported.
Supported language codes: en, es, de, fr, pt, it.
API endpoint
POST https://api.speechinfra.com/v1/inference/universal-3-pro-batch
Authenticate with your Speechinfra project key. Read the quickstart and billing guide.
cURL
curl --fail-with-body "https://api.speechinfra.com/v1/inference/universal-3-pro-batch" \
-H "Authorization: Bearer $SPEECH_API_KEY" \
-F "file=@clip.wav" \
--form-string "language=auto" \
--form-string "speaker_labels=false" \
--form-string "punctuate=true" \
--form-string "format_text=true" \
--form-string "filter_profanity=false" \
--form-string "disfluencies=false" \
--form-string "speech_threshold=0"Python
# pip install httpx
import os
import httpx
headers = {"Authorization": "Bearer " + os.environ["SPEECH_API_KEY"]}
with open("clip.wav", "rb") as audio:
response = httpx.post(
"https://api.speechinfra.com/v1/inference/universal-3-pro-batch",
headers=headers, files={"file": ("clip.wav", audio, "audio/wav")},
data={
"language": "auto",
"speaker_labels": "false",
"punctuate": "true",
"format_text": "true",
"filter_profanity": "false",
"disfluencies": "false",
"speech_threshold": "0"
}, timeout=90,
)
response.raise_for_status()
print(response.json())Node.js / fetch
// Node.js 22+, no SDK required.
import { readFile } from 'node:fs/promises';
const form = new FormData();
form.set('file', new Blob([await readFile('clip.wav')]), 'clip.wav');
for (const [key, value] of Object.entries({
"language": "auto",
"speaker_labels": "false",
"punctuate": "true",
"format_text": "true",
"filter_profanity": "false",
"disfluencies": "false",
"speech_threshold": "0"
})) form.set(key, value);
const response = await fetch("https://api.speechinfra.com/v1/inference/universal-3-pro-batch", {
method: 'POST', body: form, signal: AbortSignal.timeout(90000),
headers: { Authorization: 'Bearer ' + process.env.SPEECH_API_KEY },
});
if (!response.ok) throw new Error(await response.text());
console.log(await response.json());Request parameters
Generated from the API schema for this deployment. Conditional parameters apply only when their controlling setting is enabled.
| Parameter | Type | Details |
|---|---|---|
fileRequired | string | Short audio or video file. Direct API limit: 30 seconds, 20 MiB. |
language | string | Choose a language, auto-detect, or multi for native code-switching across en, es, de, fr, pt, it. enum: auto, en, es, de, fr, pt, it, multi · default: "auto" All accepted values[ "auto", "en", "es", "de", "fr", "pt", "it", "multi" ] |
prompt | string | max 2000 chars |
keyterms_prompt | array | max 1000 items |
speaker_labels | boolean | default: false |
min_speakers_expected | integer | min: 1 Applies when: |
max_speakers_expected | integer | min: 1 Applies when: |
punctuate | boolean | default: true |
format_text | boolean | default: true |
filter_profanity | boolean | default: false |
disfluencies | boolean | default: false |
speech_threshold | number | default: 0 · min: 0 · max: 1 |
Complete request schema
{
"type": "object",
"additionalProperties": false,
"required": [
"file"
],
"properties": {
"file": {
"type": "string",
"format": "binary",
"description": "Short audio or video file. Direct API limit: 30 seconds, 20 MiB."
},
"language": {
"type": "string",
"default": "auto",
"description": "Choose a language, auto-detect, or multi for native code-switching across en, es, de, fr, pt, it.",
"enum": [
"auto",
"en",
"es",
"de",
"fr",
"pt",
"it",
"multi"
]
},
"prompt": {
"type": "string",
"maxLength": 2000
},
"keyterms_prompt": {
"type": "array",
"items": {
"type": "string",
"minLength": 1,
"maxLength": 200,
"maxWords": 6
},
"maxItems": 1000
},
"speaker_labels": {
"type": "boolean",
"default": false
},
"min_speakers_expected": {
"type": "integer",
"minimum": 1,
"visibleWhen": {
"speaker_labels": true
}
},
"max_speakers_expected": {
"type": "integer",
"minimum": 1,
"visibleWhen": {
"speaker_labels": true
}
},
"punctuate": {
"type": "boolean",
"default": true
},
"format_text": {
"type": "boolean",
"default": true
},
"filter_profanity": {
"type": "boolean",
"default": false
},
"disfluencies": {
"type": "boolean",
"default": false
},
"speech_threshold": {
"type": "number",
"minimum": 0,
"maximum": 1,
"default": 0
}
}
}Response schema
The schema below describes the response; optional fields depend on model settings.
View complete response schema
{
"$defs": {
"Segment": {
"properties": {
"id": {
"title": "Id",
"type": "string"
},
"start": {
"title": "Start",
"type": "number"
},
"end": {
"title": "End",
"type": "number"
},
"text": {
"title": "Text",
"type": "string"
},
"speaker": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"default": null,
"title": "Speaker"
},
"language": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"default": null,
"title": "Language"
},
"words": {
"items": {
"$ref": "#/$defs/Word"
},
"title": "Words",
"type": "array"
}
},
"required": [
"id",
"start",
"end",
"text"
],
"title": "Segment",
"type": "object"
},
"Word": {
"properties": {
"word": {
"title": "Word",
"type": "string"
},
"start": {
"title": "Start",
"type": "number"
},
"end": {
"title": "End",
"type": "number"
},
"confidence": {
"anyOf": [
{
"type": "number"
},
{
"type": "null"
}
],
"default": null,
"title": "Confidence"
},
"speaker": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"default": null,
"title": "Speaker"
}
},
"required": [
"word",
"start",
"end"
],
"title": "Word",
"type": "object"
}
},
"properties": {
"text": {
"title": "Text",
"type": "string"
},
"language": {
"default": "und",
"title": "Language",
"type": "string"
},
"language_confidence": {
"anyOf": [
{
"type": "number"
},
{
"type": "null"
}
],
"default": null,
"title": "Language Confidence"
},
"duration": {
"default": 0,
"title": "Duration",
"type": "number"
},
"segments": {
"items": {
"$ref": "#/$defs/Segment"
},
"title": "Segments",
"type": "array"
},
"words": {
"items": {
"$ref": "#/$defs/Word"
},
"title": "Words",
"type": "array"
}
},
"required": [
"text"
],
"title": "PublicTranscriptionResult",
"type": "object"
}