Transcribe — From audio file
POST/transcribe
Generate a transcript from an audio file. Only audio/* mime types are supported. The maximum duration is 10 minutes. If you have longer files, please use the asynchronous equivalent.
Request
Responses
- 200
Results of processing the audio file.
Operation spec
{
"method": "post",
"path": "/transcribe",
"operationId": "transcribe",
"requestBody": {
"required": true,
"content": {
"multipart/form-data": {
"schema": {
"type": "object",
"properties": {
"request_parameters": {
"type": "object",
"description": "The object containing all the information needed along with the audio file to transcribe.",
"properties": {
"speech_locales": {
"description": "An array of up to two locales for transcription. The speech-to-text engine automatically detects the spoken locale from this list.\n\nSupport for Haitian Creole (`HAITIAN_HT`) is experimental and access is limited — please reach out to get access. Haitian Creole cannot be combined with any other language: when transcribing Haitian Creole, `speech_locales` must contain exactly one entry.",
"type": "array",
"items": {
"type": "string",
"enum": [
"ENGLISH_US",
"ENGLISH_UK",
"SPANISH_ES",
"SPANISH_MX",
"FRENCH_FR",
"ARABIC_EG",
"ARABIC_LB",
"ARABIC_MA",
"ARABIC_SA",
"ARMENIAN_AM",
"BENGALI_IN",
"CANTONESE_CN",
"CROATIAN_HR",
"FILIPINO_PH",
"GERMAN_DE",
"GREEK_GR",
"GUJARATI_IN",
"HEBREW_IL",
"HINDI_IN",
"ITALIAN_IT",
"JAPANESE_JP",
"KHMER_KH",
"KOREAN_KR",
"MANDARIN_CN",
"PERSIAN_IR",
"POLISH_PL",
"PORTUGUESE_PT",
"PUNJABI_IN",
"RUSSIAN_RU",
"SERBIAN_RS",
"TAMIL_IN",
"TELUGU_IN",
"THAI_TH",
"URDU_IN",
"VIETNAMESE_VN",
"HAITIAN_HT"
],
"example": "ENGLISH_US",
"title": "speech_locale"
},
"minItems": 1,
"maxItems": 2,
"uniqueItems": true,
"title": "speech_locale_array"
},
"split_by_sentence": {
"type": "boolean",
"default": false,
"description": "Indicates whether to segment transcription results at sentence boundaries. Default is false, meaning that a single transcript item may encompass multiple sentences, provided they are not delineated by pauses (silence) in the audio."
},
"specialty": {
"type": "string",
"description": "The medical specialty of the provider.",
"nullable": true,
"enum": [
"ADDICTION_MEDICINE",
"PAIN_MEDICINE",
"ALLERGY_AND_IMMUNOLOGY",
"ANESTHESIOLOGY",
"ONCOLOGY",
"CARDIOLOGY",
"SURGERY",
"DERMATOLOGY",
"DIABETOLOGY",
"ENDOCRINOLOGY",
"GENETICS",
"GERIATRICS",
"GYNECOLOGY",
"HEMATOLOGY",
"HEPATOLOGY",
"GASTROENTEROLOGY",
"SPORTS_MEDICINE",
"OCCUPATIONAL_MEDICINE",
"GENERAL_MEDICINE",
"FORENSIC_MEDICINE",
"PHYSICAL_MEDICINE_AND_REHABILITATION",
"NEPHROLOGY",
"NEUROLOGY",
"NUTRITION",
"DIETETICS",
"OPHTHALMOLOGY",
"ENT",
"PEDIATRICS",
"PULMONOLOGY",
"PSYCHIATRY",
"RHEUMATOLOGY",
"RADIOLOGY",
"IMMUNOLOGY",
"INFECTIOUS_DISEASE",
"SEXUAL_MEDICINE",
"TOXICOLOGY",
"UROLOGY",
"MIDWIFE",
"EMERGENCY_MEDICINE",
"NURSE",
"PSYCHOLOGY",
"PSYCHOTHERAPY",
"INTERNAL_MEDICINE",
"FAMILY_MEDICINE",
"DENTIST",
"VETERINARIAN",
"PHYSIOTHERAPY",
"CHIROPRACTIC",
"OSTEOPATHIC_MEDICINE",
"ORTHOPEDICS",
"OTHER",
"LACTATION_CONSULTANT",
"PODIATRY",
"GENERAL_PRACTICE",
"HOSPITAL_MEDICINE",
"BEHAVIORAL_HEALTH",
"MENTAL_HEALTH",
"SUBSTANCE_USE_DISORDER",
"VASCULAR_MEDICINE",
"LIFESTYLE_MEDICINE",
"PREVENTIVE_MEDICINE",
"PUBLIC_HEALTH",
"ADOLESCENT_MEDICINE",
"WOUND_CARE",
"NEUROSURGERY",
"PLASTIC_SURGERY",
"PALLIATIVE_CARE",
"TRANSPLANT_MEDICINE",
"OBESITY_MEDICINE",
"CASE_MANAGEMENT",
"CARE_MANAGEMENT",
"CARE_COORDINATION",
"SOCIAL_WORK",
"PATHOLOGY",
"RADIATION_ONCOLOGY",
"SLEEP_MEDICINE",
"AUDIOLOGY",
"REPRODUCTIVE_ENDOCRINOLOGY",
"PHARMACIST",
"OCCUPATIONAL_THERAPY",
"SPEECH_LANGUAGE_PATHOLOGY"
],
"example": "GENERAL_PRACTICE",
"title": "specialty_kind"
}
},
"required": [
"speech_locales"
],
"title": "transcribe_request"
},
"file": {
"type": "string",
"format": "binary"
}
},
"required": [
"request_parameters",
"file"
]
}
}
}
},
"responses": {
"200": {
"description": "Results of processing the audio file.",
"content": {
"application/json": {
"schema": {
"type": "object",
"properties": {
"transcript": {
"type": "array",
"description": "Transcript items from the audio file.",
"items": {
"type": "object",
"description": "A portion of the transcribed consultation.",
"properties": {
"text": {
"type": "string",
"description": "The transcribed text.",
"example": "Also, I’m allergic to peanuts."
},
"speaker_type": {
"type": "string",
"enum": [
"DOCTOR",
"PATIENT",
"UNSPECIFIED"
],
"description": "Who said the text in this transcript item.",
"example": "DOCTOR",
"title": "speaker"
},
"locale": {
"description": "Locale for this transcript item, detected by the speech-to-text engine from the list of locales in the input.",
"type": "string",
"enum": [
"ENGLISH_US",
"ENGLISH_UK",
"SPANISH_ES",
"SPANISH_MX",
"FRENCH_FR",
"ARABIC_EG",
"ARABIC_LB",
"ARABIC_MA",
"ARABIC_SA",
"ARMENIAN_AM",
"BENGALI_IN",
"CANTONESE_CN",
"CROATIAN_HR",
"FILIPINO_PH",
"GERMAN_DE",
"GREEK_GR",
"GUJARATI_IN",
"HEBREW_IL",
"HINDI_IN",
"ITALIAN_IT",
"JAPANESE_JP",
"KHMER_KH",
"KOREAN_KR",
"MANDARIN_CN",
"PERSIAN_IR",
"POLISH_PL",
"PORTUGUESE_PT",
"PUNJABI_IN",
"RUSSIAN_RU",
"SERBIAN_RS",
"TAMIL_IN",
"TELUGU_IN",
"THAI_TH",
"URDU_IN",
"VIETNAMESE_VN",
"HAITIAN_HT"
],
"example": "ENGLISH_US",
"title": "speech_locale"
},
"start_offset_ms": {
"type": "integer",
"description": "Start time of this transcription item as the offset, in milliseconds, from the start of the audio file.",
"example": 65100
},
"end_offset_ms": {
"type": "integer",
"description": "End time of this transcription item as the offset, in milliseconds, from the start of the audio file. Equals the `start_time_ms` plus the duration of the related transcribed audio portion.",
"example": 69300
}
},
"required": [
"text",
"locale",
"start_offset_ms",
"end_offset_ms"
],
"title": "transcript_item"
},
"title": "transcript"
}
},
"required": [
"transcript"
],
"title": "transcribe_response"
}
}
}
}
}
}