Create a transcription
curl --request POST \
--url https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions \
--header 'Content-Type: application/json' \
--header 'Idempotency-Key: <idempotency-key>' \
--header 'accessKey: <accesskey>' \
--data '
{
"uploadId": "<string>",
"mediaUrl": "<string>",
"audioTrackIndex": 123,
"languageHint": "<string>",
"title": "<string>",
"modelId": "<string>",
"options": {},
"waitForCompletion": true,
"waitTimeoutMs": 123
}
'import requests
url = "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions"
payload = {
"uploadId": "<string>",
"mediaUrl": "<string>",
"audioTrackIndex": 123,
"languageHint": "<string>",
"title": "<string>",
"modelId": "<string>",
"options": {},
"waitForCompletion": True,
"waitTimeoutMs": 123
}
headers = {
"accessKey": "<accesskey>",
"Idempotency-Key": "<idempotency-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {
accessKey: '<accesskey>',
'Idempotency-Key': '<idempotency-key>',
'Content-Type': 'application/json'
},
body: JSON.stringify({
uploadId: '<string>',
mediaUrl: '<string>',
audioTrackIndex: 123,
languageHint: '<string>',
title: '<string>',
modelId: '<string>',
options: {},
waitForCompletion: true,
waitTimeoutMs: 123
})
};
fetch('https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'uploadId' => '<string>',
'mediaUrl' => '<string>',
'audioTrackIndex' => 123,
'languageHint' => '<string>',
'title' => '<string>',
'modelId' => '<string>',
'options' => [
],
'waitForCompletion' => true,
'waitTimeoutMs' => 123
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"Idempotency-Key: <idempotency-key>",
"accessKey: <accesskey>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions"
payload := strings.NewReader("{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("accessKey", "<accesskey>")
req.Header.Add("Idempotency-Key", "<idempotency-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions")
.header("accessKey", "<accesskey>")
.header("Idempotency-Key", "<idempotency-key>")
.header("Content-Type", "application/json")
.body("{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["accessKey"] = '<accesskey>'
request["Idempotency-Key"] = '<idempotency-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}"
response = http.request(request)
puts response.read_body{
"code": 123,
"message": "<string>",
"data": {
"transcriptionId": "<string>",
"requestId": "<string>",
"status": "<string>",
"modelId": "<string>",
"billing": {}
}
}Speech to Text
Create a transcription
Submit an uploaded file or a public media URL, quote and reserve Characters, and start the task.
POST
/
sound_clone
/
api
/
v1
/
stt
/
transcriptions
Create a transcription
curl --request POST \
--url https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions \
--header 'Content-Type: application/json' \
--header 'Idempotency-Key: <idempotency-key>' \
--header 'accessKey: <accesskey>' \
--data '
{
"uploadId": "<string>",
"mediaUrl": "<string>",
"audioTrackIndex": 123,
"languageHint": "<string>",
"title": "<string>",
"modelId": "<string>",
"options": {},
"waitForCompletion": true,
"waitTimeoutMs": 123
}
'import requests
url = "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions"
payload = {
"uploadId": "<string>",
"mediaUrl": "<string>",
"audioTrackIndex": 123,
"languageHint": "<string>",
"title": "<string>",
"modelId": "<string>",
"options": {},
"waitForCompletion": True,
"waitTimeoutMs": 123
}
headers = {
"accessKey": "<accesskey>",
"Idempotency-Key": "<idempotency-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {
accessKey: '<accesskey>',
'Idempotency-Key': '<idempotency-key>',
'Content-Type': 'application/json'
},
body: JSON.stringify({
uploadId: '<string>',
mediaUrl: '<string>',
audioTrackIndex: 123,
languageHint: '<string>',
title: '<string>',
modelId: '<string>',
options: {},
waitForCompletion: true,
waitTimeoutMs: 123
})
};
fetch('https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'uploadId' => '<string>',
'mediaUrl' => '<string>',
'audioTrackIndex' => 123,
'languageHint' => '<string>',
'title' => '<string>',
'modelId' => '<string>',
'options' => [
],
'waitForCompletion' => true,
'waitTimeoutMs' => 123
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"Idempotency-Key: <idempotency-key>",
"accessKey: <accesskey>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions"
payload := strings.NewReader("{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("accessKey", "<accesskey>")
req.Header.Add("Idempotency-Key", "<idempotency-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions")
.header("accessKey", "<accesskey>")
.header("Idempotency-Key", "<idempotency-key>")
.header("Content-Type", "application/json")
.body("{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["accessKey"] = '<accesskey>'
request["Idempotency-Key"] = '<idempotency-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}"
response = http.request(request)
puts response.read_body{
"code": 123,
"message": "<string>",
"data": {
"transcriptionId": "<string>",
"requestId": "<string>",
"status": "<string>",
"modelId": "<string>",
"billing": {}
}
}Submits one transcription from an
Only relevant formats use each control;
uploadId or a public mediaUrl. The call quotes the work,
reserves Characters and starts the asynchronous task in a single request.
Header
string
required
API key for authentication.
string
required
Binds this request. The same key and the same body return the same task; the same key with a
different body is refused as
IDEMPOTENCY_CONFLICT.Body
string
An upload from
POST /uploads. Use exactly one of uploadId or mediaUrl.string
A public
http/https URL. The server fetches it over a real connection and re-checks every
redirect; private, loopback, link-local and cloud-metadata addresses are refused. There is no domain
or port allowlist.number
Which audio track to transcribe. Required when the source contains more than one audio track;
use the
index from the upload’s media.audioTracks.string
The spoken language, or omit for automatic detection.
string
A title for the History row.
string
Optional model identifier; defaults to
myvocal_stt_v1.object
Processing options:
separateSpeakers, maxSpeakerCount, alignmentLevel, entityCategories,
redactCategories, redactionStyle, rewriteInstruction, removeDisfluencies,
identifyConversationRoles, vocabularyHints, separateChannels, channelResultMode,
exportFormats, inputEncoding, processingContentStorage and more. Option combinations that the
model does not support are refused at submit with INPUT_INVALID rather than silently dropped.
Completion notifications are options too: options.notifyOnCompletion (true to enable),
options.notifyUrl (a public URL MyVocal posts the signed event to), options.clientMetadata
(opaque caller data echoed back) and options.notificationSigningSecret (the write-only HMAC
key). Polling stays the authoritative result.
See capabilities for the live list.boolean
When
true, this call waits for a terminal state inside the same request.number
The wait budget in milliseconds. If the task is still running at the budget, the request
returns the task with its identity; it is not cancelled.
Batch options
All fields below belong insideoptions. Omit fields you do not need.
| Field | Type and accepted values |
|---|---|
rewriteInstruction | String, at most 2000 characters |
includeSoundEvents | Boolean |
separateSpeakers | Boolean; default true for single-channel processing |
maxSpeakerCount | Integer, 1–32 |
speakerMergeSensitivity | Number, 0.1–0.4; requires speaker separation and no maxSpeakerCount |
alignmentLevel | none, word (default), character |
inputEncoding | other (default, prepared container audio), pcm_s16le_16 (converted 16 kHz mono PCM) |
samplingTemperature | Number, 0–2 |
randomSeed | Non-negative integer |
separateChannels | Boolean; preserve and transcribe channels separately |
channelResultMode | separate or combined |
entityCategories | Array, at most 32 strings; tested examples include pii, phi, pci, all, name |
redactCategories | Array, at most 32 strings; subset of entityCategories |
redactionStyle | redact, replace, mask; requires redactCategories |
removeDisfluencies | Boolean |
identifyConversationRoles | Boolean; requires speaker separation |
vocabularyHints | Array, at most 1000 strings, each at most 128 characters |
processingContentStorage | Boolean; see storage limitations |
exportFormats | Array of supported download formats |
exportOptions | Array of per-format objects, described below |
notifyOnCompletion | Boolean; enables completed-task notification |
notifyUrl | Public HTTP(S) URL |
clientMetadata | Object, at most two levels and 16 KB |
notificationSigningSecret | Write-only signing secret; see notifications |
Option combinations
rewriteInstructioncannot be combined with entity detection, redaction orseparateChannels.- Multi-channel processing does not separate speakers. Do not combine
separateChannelswithmaxSpeakerCount,speakerMergeSensitivity,identifyConversationRolesorpcm_s16le_16input. The selected audio track can have at most five channels for transcription. - Multi-channel
combinedresults require alignment and cannot be combined with entity detection or redaction. matchKnownSpeakersis unavailable. Do not request it.
exportOptions entry has a required format and optional controls:
| Field | Type / limit |
|---|---|
includeSpeakers, includeTimestamps | Boolean |
segmentOnSilenceLongerThanS, maxSegmentDurationS | Number greater than 0 and at most 3600 seconds |
maxSegmentChars | Integer, 1–100000 |
maxCharactersPerLine | Integer, 1–16384 |
json is the public transcription view. Without alignment,
SRT cannot produce timed cues. These options do not change recognition or billing.
Some invalid entity categories currently fail asynchronously as
MEDIA_UNREADABLE even for valid
audio. See known limitations before retrying.Response
Successful REST calls return{ code: 1, message, data }. The fields below are inside data;
JSON downloads are the documented exception and return the transcript view directly.
number
1 for success.string
Result message.
object
Hide properties
Hide properties
string
The MyVocal task id, also the History id.
string
The correlation id for this request.
string
QUEUED, PROCESSING, RECONCILING, COMPLETED, PARTIAL or FAILED.string
myvocal_stt_v1.object
state, planKey, rateVersion, ratePerMinute, reservedCharacters, settledCharacters,
releasedCharacters, billableCharacters and billableDurationMs. Characters and durations are
JSON strings; convert before comparing.