Skip to content
speechinfraConsole

MOSI · ASR

MOSS Transcribe Diarize Pro

Speech transcription with speaker-aware segments for conversations and meetings.

moss-transcribe-diarize-pro
$0.005 / audio minuteTry in Console

Current model-specific price from our catalog. Requests require your project API key and sufficient balance. Availability is checked at execution time.

Input and capabilities

Upload a short audio or video file using multipart/form-data: at most 30 seconds and 20 MiB. Use WAV for the examples below. Container compatibility depends on the model.

  • transcription
  • diarization
  • segment timestamps
  • hotwords

Automatic language detection: supported.

No language-specific selection is declared for this model.

API endpoint

POST https://api.speechinfra.com/v1/inference/moss-transcribe-diarize-pro

Authenticate with your Speechinfra project key. Read the quickstart and billing guide.

cURL
curl --fail-with-body "https://api.speechinfra.com/v1/inference/moss-transcribe-diarize-pro" \
  -H "Authorization: Bearer $SPEECH_API_KEY" \
  -F "file=@clip.wav" \
  --form-string "diarize=true" \
  --form-string "response_format=diarized_json"
Python
# pip install httpx
import os
import httpx

headers = {"Authorization": "Bearer " + os.environ["SPEECH_API_KEY"]}
with open("clip.wav", "rb") as audio:
    response = httpx.post(
        "https://api.speechinfra.com/v1/inference/moss-transcribe-diarize-pro",
        headers=headers, files={"file": ("clip.wav", audio, "audio/wav")},
        data={
    "diarize": "true",
    "response_format": "diarized_json"
}, timeout=90,
    )
response.raise_for_status()
print(response.json())
Node.js / fetch
// Node.js 22+, no SDK required.
import { readFile } from 'node:fs/promises';
const form = new FormData();
form.set('file', new Blob([await readFile('clip.wav')]), 'clip.wav');
for (const [key, value] of Object.entries({
  "diarize": "true",
  "response_format": "diarized_json"
})) form.set(key, value);
const response = await fetch("https://api.speechinfra.com/v1/inference/moss-transcribe-diarize-pro", {
  method: 'POST', body: form, signal: AbortSignal.timeout(90000),
  headers: { Authorization: 'Bearer ' + process.env.SPEECH_API_KEY },
});
if (!response.ok) throw new Error(await response.text());
console.log(await response.json());

Request parameters

Generated from the API schema for this deployment. Conditional parameters apply only when their controlling setting is enabled.

ParameterTypeDetails
file

Required

string

Short audio or video file. Direct API limit: 30 seconds, 20 MiB.

diarizeboolean

Identify speakers in the uploaded recording. Disable for transcription without speaker labels.

default: true

response_formatstring

JSON returns structured transcription. Speaker labels require Separate speakers to be enabled. Plain text contains only the transcript.

enum: json, diarized_json, text · default: "diarized_json"

All accepted values
[
  "json",
  "diarized_json",
  "text"
]
keytermsarray

Optional names, brands or technical terms. In the playground, enter one per line. Up to 20 terms, 30 characters each. Leave empty for ordinary multi-speaker transcription.

max 20 items

Complete request schema
{
  "type": "object",
  "additionalProperties": false,
  "required": [
    "file"
  ],
  "properties": {
    "file": {
      "type": "string",
      "format": "binary",
      "description": "Short audio or video file. Direct API limit: 30 seconds, 20 MiB."
    },
    "diarize": {
      "type": "boolean",
      "default": true,
      "description": "Identify speakers in the uploaded recording. Disable for transcription without speaker labels."
    },
    "response_format": {
      "type": "string",
      "enum": [
        "json",
        "diarized_json",
        "text"
      ],
      "default": "diarized_json",
      "description": "JSON returns structured transcription. Speaker labels require Separate speakers to be enabled. Plain text contains only the transcript."
    },
    "keyterms": {
      "type": "array",
      "items": {
        "type": "string",
        "minLength": 1,
        "maxLength": 30
      },
      "maxItems": 20,
      "description": "Optional names, brands or technical terms. In the playground, enter one per line. Up to 20 terms, 30 characters each. Leave empty for ordinary multi-speaker transcription."
    }
  }
}

Response schema

The schema below describes the response; optional fields depend on model settings.

View complete response schema
{
  "anyOf": [
    {
      "properties": {
        "text": {
          "title": "Text",
          "type": "string"
        },
        "language": {
          "default": "und",
          "title": "Language",
          "type": "string"
        },
        "language_confidence": {
          "anyOf": [
            {
              "type": "number"
            },
            {
              "type": "null"
            }
          ],
          "default": null,
          "title": "Language Confidence"
        },
        "duration": {
          "default": 0,
          "title": "Duration",
          "type": "number"
        },
        "segments": {
          "items": {
            "$ref": "#/$defs/Segment"
          },
          "title": "Segments",
          "type": "array"
        },
        "words": {
          "items": {
            "$ref": "#/$defs/Word"
          },
          "title": "Words",
          "type": "array"
        }
      },
      "required": [
        "text"
      ],
      "title": "PublicTranscriptionResult",
      "type": "object"
    },
    {
      "type": "string",
      "description": "Plain transcript when response_format=text."
    }
  ],
  "$defs": {
    "Segment": {
      "properties": {
        "id": {
          "title": "Id",
          "type": "string"
        },
        "start": {
          "title": "Start",
          "type": "number"
        },
        "end": {
          "title": "End",
          "type": "number"
        },
        "text": {
          "title": "Text",
          "type": "string"
        },
        "speaker": {
          "anyOf": [
            {
              "type": "string"
            },
            {
              "type": "null"
            }
          ],
          "default": null,
          "title": "Speaker"
        },
        "language": {
          "anyOf": [
            {
              "type": "string"
            },
            {
              "type": "null"
            }
          ],
          "default": null,
          "title": "Language"
        },
        "words": {
          "items": {
            "$ref": "#/$defs/Word"
          },
          "title": "Words",
          "type": "array"
        }
      },
      "required": [
        "id",
        "start",
        "end",
        "text"
      ],
      "title": "Segment",
      "type": "object"
    },
    "Word": {
      "properties": {
        "word": {
          "title": "Word",
          "type": "string"
        },
        "start": {
          "title": "Start",
          "type": "number"
        },
        "end": {
          "title": "End",
          "type": "number"
        },
        "confidence": {
          "anyOf": [
            {
              "type": "number"
            },
            {
              "type": "null"
            }
          ],
          "default": null,
          "title": "Confidence"
        },
        "speaker": {
          "anyOf": [
            {
              "type": "string"
            },
            {
              "type": "null"
            }
          ],
          "default": null,
          "title": "Speaker"
        }
      },
      "required": [
        "word",
        "start",
        "end"
      ],
      "title": "Word",
      "type": "object"
    }
  }
}