NVIDIA · Speaker diarization
NVIDIA Nemotron 3 Diarization
Find who spoke when with up to eight speakers, overlapping speech and precise speaker turns. Returns timestamps and speaker labels without a transcript.
nemotron-3-diarizationCurrent model-specific price from our catalog. Requests require your project API key and sufficient balance. Availability is checked at execution time.
Input and capabilities
Upload a short audio or video file using multipart/form-data: at most 30 seconds and 20 MiB. Use WAV for the examples below. Container compatibility depends on the model.
- diarization
- segment timestamps
Identify up to eight speakers, including overlapping speech. This endpoint returns who spoke when; it does not produce a transcript or recognize a speaker’s identity. Labels belong to the current recording. Silence is a valid result with an empty speakers array.
No language or expected speaker count is required. Presets control how the model processes the recording internally; every preset uses a file upload and a complete response, not a realtime connection.
API endpoint
POST https://api.speechinfra.com/v1/inference/nemotron-3-diarization
Authenticate with your Speechinfra project key. Read the quickstart and billing guide.
The first request may wait while the model starts. Examples allow up to 16 minutes, including time to receive an error response; later requests can reuse the running worker. Avoid resubmitting while a request is pending.
cURL
curl --fail-with-body --max-time 960 "https://api.speechinfra.com/v1/inference/nemotron-3-diarization" \
-H "Authorization: Bearer $SPEECH_API_KEY" \
-F "file=@clip.wav" \
--form-string "preset=offline" \
--form-string "onset=0.5" \
--form-string "offset=0.5" \
--form-string "pad_onset=0" \
--form-string "pad_offset=0" \
--form-string "min_duration_on=0" \
--form-string "min_duration_off=0" \
--form-string "return_rttm=false"Python
# pip install httpx
import os
import httpx
headers = {"Authorization": "Bearer " + os.environ["SPEECH_API_KEY"]}
with open("clip.wav", "rb") as audio:
response = httpx.post(
"https://api.speechinfra.com/v1/inference/nemotron-3-diarization",
headers=headers, files={"file": ("clip.wav", audio, "audio/wav")},
data={
"preset": "offline",
"onset": "0.5",
"offset": "0.5",
"pad_onset": "0",
"pad_offset": "0",
"min_duration_on": "0",
"min_duration_off": "0",
"return_rttm": "false"
}, timeout=960,
)
response.raise_for_status()
print(response.json())Node.js / fetch
// Node.js 22+, no SDK required.
import { readFile } from 'node:fs/promises';
const form = new FormData();
form.set('file', new Blob([await readFile('clip.wav')]), 'clip.wav');
for (const [key, value] of Object.entries({
"preset": "offline",
"onset": "0.5",
"offset": "0.5",
"pad_onset": "0",
"pad_offset": "0",
"min_duration_on": "0",
"min_duration_off": "0",
"return_rttm": "false"
})) form.set(key, value);
const response = await fetch("https://api.speechinfra.com/v1/inference/nemotron-3-diarization", {
method: 'POST', body: form, signal: AbortSignal.timeout(960000),
headers: { Authorization: 'Bearer ' + process.env.SPEECH_API_KEY },
});
if (!response.ok) throw new Error(await response.text());
console.log(await response.json());Request parameters
Generated from the API schema for this deployment. Conditional parameters apply only when their controlling setting is enabled.
| Parameter | Type | Details |
|---|---|---|
fileRequired | string | Short audio or video file. Direct API limit: 30 seconds, 20 MiB. |
preset | string | Internal chunking preset. The endpoint always returns the complete recording; this is not a realtime stream. enum: offline, low_latency, very_low_latency, ultra_low_latency · default: "offline" All accepted values[ "offline", "low_latency", "very_low_latency", "ultra_low_latency" ] |
onset | number | Speech start probability threshold. default: 0.5 · min: 0 · max: 1 |
offset | number | Speech end threshold; must not exceed onset. default: 0.5 · min: 0 · max: 1 |
pad_onset | number | Seconds added before a speech segment. default: 0 · min: 0 · max: 1 |
pad_offset | number | Seconds added after a speech segment. default: 0 · min: 0 · max: 1 |
min_duration_on | number | Minimum retained speech segment duration in seconds. default: 0 · min: 0 · max: 10 |
min_duration_off | number | Fill speech gaps shorter than this many seconds. default: 0 · min: 0 · max: 10 |
return_rttm | boolean | Include standard RTTM speaker turns alongside JSON. default: false |
Complete request schema
{
"type": "object",
"additionalProperties": false,
"required": [
"file"
],
"properties": {
"file": {
"type": "string",
"format": "binary",
"description": "Short audio or video file. Direct API limit: 30 seconds, 20 MiB."
},
"preset": {
"type": "string",
"enum": [
"offline",
"low_latency",
"very_low_latency",
"ultra_low_latency"
],
"default": "offline",
"description": "Internal chunking preset. The endpoint always returns the complete recording; this is not a realtime stream."
},
"onset": {
"type": "number",
"minimum": 0,
"maximum": 1,
"default": 0.5,
"description": "Speech start probability threshold."
},
"offset": {
"type": "number",
"minimum": 0,
"maximum": 1,
"default": 0.5,
"description": "Speech end threshold; must not exceed onset."
},
"pad_onset": {
"type": "number",
"minimum": 0,
"maximum": 1,
"default": 0,
"description": "Seconds added before a speech segment."
},
"pad_offset": {
"type": "number",
"minimum": 0,
"maximum": 1,
"default": 0,
"description": "Seconds added after a speech segment."
},
"min_duration_on": {
"type": "number",
"minimum": 0,
"maximum": 10,
"default": 0,
"description": "Minimum retained speech segment duration in seconds."
},
"min_duration_off": {
"type": "number",
"minimum": 0,
"maximum": 10,
"default": 0,
"description": "Fill speech gaps shorter than this many seconds."
},
"return_rttm": {
"type": "boolean",
"default": false,
"description": "Include standard RTTM speaker turns alongside JSON."
}
}
}Response schema
duration, start and end are in seconds. speakers contains speaker turns, so one speaker may appear many times. Turns can overlap. Set return_rttm=true to also receive RTTM annotations.
Example speaker turns
{
"duration": 2.5,
"speakers": [
{
"speaker": "speaker_0",
"start": 0.2,
"end": 1.8,
"overlap": true
},
{
"speaker": "speaker_1",
"start": 1.5,
"end": 2.3,
"overlap": true
}
],
"model": "nemotron-3-diarization",
"provider": "runpod"
}View complete response schema
{
"$defs": {
"DiarizationTurn": {
"properties": {
"speaker": {
"title": "Speaker",
"type": "string"
},
"start": {
"title": "Start",
"type": "number"
},
"end": {
"title": "End",
"type": "number"
},
"overlap": {
"default": false,
"title": "Overlap",
"type": "boolean"
}
},
"required": [
"speaker",
"start",
"end"
],
"title": "DiarizationTurn",
"type": "object"
}
},
"properties": {
"duration": {
"title": "Duration",
"type": "number"
},
"speakers": {
"items": {
"$ref": "#/$defs/DiarizationTurn"
},
"title": "Speakers",
"type": "array"
},
"model": {
"title": "Model",
"type": "string"
},
"provider": {
"title": "Provider",
"type": "string"
},
"rttm": {
"anyOf": [
{
"type": "string"
},
{
"type": "null"
}
],
"default": null,
"title": "Rttm"
}
},
"required": [
"duration",
"speakers",
"model",
"provider"
],
"title": "DiarizationResult",
"type": "object"
}