# Transcribe — From audio file

```
POST 
/transcribe
```

Generate a transcript from an audio file. Only `audio/*` mime types are supported. The maximum duration is 10 minutes. If you have longer files, please use the [asynchronous equivalent](/core-api/reference/2026-03-16/server/transcribe-async.md).

## Request[​](#request "Direct link to request")

## Responses[​](#responses "Direct link to Responses")

* 200

Results of processing the audio file.

## Operation spec

```json
{
  "method": "post",
  "path": "/transcribe",
  "operationId": "transcribe",
  "requestBody": {
    "required": true,
    "content": {
      "multipart/form-data": {
        "schema": {
          "type": "object",
          "properties": {
            "request_parameters": {
              "type": "object",
              "description": "The object containing all the information needed along with the audio file to transcribe.",
              "properties": {
                "speech_locales": {
                  "description": "An array of up to two locales for transcription. The speech-to-text engine automatically detects the spoken locale from this list.\n\nSupport for Haitian Creole (`HAITIAN_HT`) is experimental and access is limited — please reach out to get access. Haitian Creole cannot be combined with any other language: when transcribing Haitian Creole, `speech_locales` must contain exactly one entry.",
                  "type": "array",
                  "items": {
                    "type": "string",
                    "enum": [
                      "ENGLISH_US",
                      "ENGLISH_UK",
                      "SPANISH_ES",
                      "SPANISH_MX",
                      "FRENCH_FR",
                      "ARABIC_EG",
                      "ARABIC_LB",
                      "ARABIC_MA",
                      "ARABIC_SA",
                      "ARMENIAN_AM",
                      "BENGALI_IN",
                      "CANTONESE_CN",
                      "CROATIAN_HR",
                      "FILIPINO_PH",
                      "GERMAN_DE",
                      "GREEK_GR",
                      "GUJARATI_IN",
                      "HEBREW_IL",
                      "HINDI_IN",
                      "ITALIAN_IT",
                      "JAPANESE_JP",
                      "KHMER_KH",
                      "KOREAN_KR",
                      "MANDARIN_CN",
                      "PERSIAN_IR",
                      "POLISH_PL",
                      "PORTUGUESE_PT",
                      "PUNJABI_IN",
                      "RUSSIAN_RU",
                      "SERBIAN_RS",
                      "TAMIL_IN",
                      "TELUGU_IN",
                      "THAI_TH",
                      "URDU_IN",
                      "VIETNAMESE_VN",
                      "HAITIAN_HT"
                    ],
                    "example": "ENGLISH_US",
                    "title": "speech_locale"
                  },
                  "minItems": 1,
                  "maxItems": 2,
                  "uniqueItems": true,
                  "title": "speech_locale_array"
                },
                "split_by_sentence": {
                  "type": "boolean",
                  "default": false,
                  "description": "Indicates whether to segment transcription results at sentence boundaries. Default is false, meaning that a single transcript item may encompass multiple sentences, provided they are not delineated by pauses (silence) in the audio."
                }
              },
              "required": [
                "speech_locales"
              ],
              "title": "transcribe_request"
            },
            "file": {
              "type": "string",
              "format": "binary"
            }
          },
          "required": [
            "request_parameters",
            "file"
          ]
        }
      }
    }
  },
  "responses": {
    "200": {
      "description": "Results of processing the audio file.",
      "content": {
        "application/json": {
          "schema": {
            "type": "object",
            "properties": {
              "transcript": {
                "type": "array",
                "description": "Transcript items from the audio file.",
                "items": {
                  "type": "object",
                  "description": "A portion of the transcribed consultation.",
                  "properties": {
                    "text": {
                      "type": "string",
                      "description": "The transcribed text.",
                      "example": "Also, I’m allergic to peanuts."
                    },
                    "speaker_type": {
                      "type": "string",
                      "enum": [
                        "DOCTOR",
                        "PATIENT",
                        "UNSPECIFIED"
                      ],
                      "description": "Who said the text in this transcript item.",
                      "example": "DOCTOR",
                      "title": "speaker"
                    },
                    "locale": {
                      "description": "Locale for this transcript item, detected by the speech-to-text engine from the list of locales in the input.",
                      "type": "string",
                      "enum": [
                        "ENGLISH_US",
                        "ENGLISH_UK",
                        "SPANISH_ES",
                        "SPANISH_MX",
                        "FRENCH_FR",
                        "ARABIC_EG",
                        "ARABIC_LB",
                        "ARABIC_MA",
                        "ARABIC_SA",
                        "ARMENIAN_AM",
                        "BENGALI_IN",
                        "CANTONESE_CN",
                        "CROATIAN_HR",
                        "FILIPINO_PH",
                        "GERMAN_DE",
                        "GREEK_GR",
                        "GUJARATI_IN",
                        "HEBREW_IL",
                        "HINDI_IN",
                        "ITALIAN_IT",
                        "JAPANESE_JP",
                        "KHMER_KH",
                        "KOREAN_KR",
                        "MANDARIN_CN",
                        "PERSIAN_IR",
                        "POLISH_PL",
                        "PORTUGUESE_PT",
                        "PUNJABI_IN",
                        "RUSSIAN_RU",
                        "SERBIAN_RS",
                        "TAMIL_IN",
                        "TELUGU_IN",
                        "THAI_TH",
                        "URDU_IN",
                        "VIETNAMESE_VN",
                        "HAITIAN_HT"
                      ],
                      "example": "ENGLISH_US",
                      "title": "speech_locale"
                    },
                    "start_offset_ms": {
                      "type": "integer",
                      "description": "Start time of this transcription item as the offset, in milliseconds, from the start of the audio file.",
                      "example": 65100
                    },
                    "end_offset_ms": {
                      "type": "integer",
                      "description": "End time of this transcription item as the offset, in milliseconds, from the start of the audio file. Equals the `start_time_ms` plus the duration of the related transcribed audio portion.",
                      "example": 69300
                    }
                  },
                  "required": [
                    "text",
                    "locale",
                    "start_offset_ms",
                    "end_offset_ms"
                  ],
                  "title": "transcript_item"
                },
                "title": "transcript"
              }
            },
            "required": [
              "transcript"
            ],
            "title": "transcribe_response"
          }
        }
      }
    }
  }
}
```
