Create a transcription
curl --request POST \
--url https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions \
--header 'Content-Type: application/json' \
--header 'Idempotency-Key: <idempotency-key>' \
--header 'accessKey: <accesskey>' \
--data '
{
"uploadId": "<string>",
"mediaUrl": "<string>",
"audioTrackIndex": 123,
"languageHint": "<string>",
"title": "<string>",
"modelId": "<string>",
"options": {},
"waitForCompletion": true,
"waitTimeoutMs": 123
}
'import requests
url = "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions"
payload = {
"uploadId": "<string>",
"mediaUrl": "<string>",
"audioTrackIndex": 123,
"languageHint": "<string>",
"title": "<string>",
"modelId": "<string>",
"options": {},
"waitForCompletion": True,
"waitTimeoutMs": 123
}
headers = {
"accessKey": "<accesskey>",
"Idempotency-Key": "<idempotency-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {
accessKey: '<accesskey>',
'Idempotency-Key': '<idempotency-key>',
'Content-Type': 'application/json'
},
body: JSON.stringify({
uploadId: '<string>',
mediaUrl: '<string>',
audioTrackIndex: 123,
languageHint: '<string>',
title: '<string>',
modelId: '<string>',
options: {},
waitForCompletion: true,
waitTimeoutMs: 123
})
};
fetch('https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'uploadId' => '<string>',
'mediaUrl' => '<string>',
'audioTrackIndex' => 123,
'languageHint' => '<string>',
'title' => '<string>',
'modelId' => '<string>',
'options' => [
],
'waitForCompletion' => true,
'waitTimeoutMs' => 123
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"Idempotency-Key: <idempotency-key>",
"accessKey: <accesskey>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions"
payload := strings.NewReader("{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("accessKey", "<accesskey>")
req.Header.Add("Idempotency-Key", "<idempotency-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions")
.header("accessKey", "<accesskey>")
.header("Idempotency-Key", "<idempotency-key>")
.header("Content-Type", "application/json")
.body("{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["accessKey"] = '<accesskey>'
request["Idempotency-Key"] = '<idempotency-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}"
response = http.request(request)
puts response.read_body{
"code": 123,
"message": "<string>",
"data": {
"transcriptionId": "<string>",
"requestId": "<string>",
"status": "<string>",
"modelId": "<string>",
"billing": {}
}
}Speech to Text
Create a transcription
Submit an uploaded file or a public media URL, quote and reserve Characters, and start the task.
POST
/
sound_clone
/
api
/
v1
/
stt
/
transcriptions
Create a transcription
curl --request POST \
--url https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions \
--header 'Content-Type: application/json' \
--header 'Idempotency-Key: <idempotency-key>' \
--header 'accessKey: <accesskey>' \
--data '
{
"uploadId": "<string>",
"mediaUrl": "<string>",
"audioTrackIndex": 123,
"languageHint": "<string>",
"title": "<string>",
"modelId": "<string>",
"options": {},
"waitForCompletion": true,
"waitTimeoutMs": 123
}
'import requests
url = "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions"
payload = {
"uploadId": "<string>",
"mediaUrl": "<string>",
"audioTrackIndex": 123,
"languageHint": "<string>",
"title": "<string>",
"modelId": "<string>",
"options": {},
"waitForCompletion": True,
"waitTimeoutMs": 123
}
headers = {
"accessKey": "<accesskey>",
"Idempotency-Key": "<idempotency-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {
accessKey: '<accesskey>',
'Idempotency-Key': '<idempotency-key>',
'Content-Type': 'application/json'
},
body: JSON.stringify({
uploadId: '<string>',
mediaUrl: '<string>',
audioTrackIndex: 123,
languageHint: '<string>',
title: '<string>',
modelId: '<string>',
options: {},
waitForCompletion: true,
waitTimeoutMs: 123
})
};
fetch('https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'uploadId' => '<string>',
'mediaUrl' => '<string>',
'audioTrackIndex' => 123,
'languageHint' => '<string>',
'title' => '<string>',
'modelId' => '<string>',
'options' => [
],
'waitForCompletion' => true,
'waitTimeoutMs' => 123
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"Idempotency-Key: <idempotency-key>",
"accessKey: <accesskey>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions"
payload := strings.NewReader("{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("accessKey", "<accesskey>")
req.Header.Add("Idempotency-Key", "<idempotency-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions")
.header("accessKey", "<accesskey>")
.header("Idempotency-Key", "<idempotency-key>")
.header("Content-Type", "application/json")
.body("{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.myvocal.ai/sound_clone/api/v1/stt/transcriptions")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["accessKey"] = '<accesskey>'
request["Idempotency-Key"] = '<idempotency-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"uploadId\": \"<string>\",\n \"mediaUrl\": \"<string>\",\n \"audioTrackIndex\": 123,\n \"languageHint\": \"<string>\",\n \"title\": \"<string>\",\n \"modelId\": \"<string>\",\n \"options\": {},\n \"waitForCompletion\": true,\n \"waitTimeoutMs\": 123\n}"
response = http.request(request)
puts response.read_body{
"code": 123,
"message": "<string>",
"data": {
"transcriptionId": "<string>",
"requestId": "<string>",
"status": "<string>",
"modelId": "<string>",
"billing": {}
}
}Submits one transcription from an
Only relevant formats use each control;
uploadId or a public mediaUrl. The call quotes the work,
reserves Characters and starts the asynchronous task in a single request.
Header
string
required
API key for authentication.
string
required
Binds this request. The same key and the same body return the same task; the same key with a
different body is refused as
IDEMPOTENCY_CONFLICT.Body
string
An upload from
POST /uploads. Use exactly one of uploadId or mediaUrl.string
A public
http/https URL. The server fetches it over a real connection and re-checks every
redirect; private, loopback, link-local and cloud-metadata addresses are refused. There is no domain
or port allowlist. A failure reports details.reason (DESTINATION_NOT_ALLOWED, SOURCE_REFUSED,
SOURCE_UNAVAILABLE, SOURCE_UNREACHABLE, TOO_MANY_REDIRECTS or UNSUPPORTED_TYPE) and, when the
host answered, details.sourceStatus; see public media URLs.number
Which audio track to transcribe. Required when the source contains more than one audio track;
use the
index from the upload’s media.audioTracks.string
The spoken language, or omit for automatic detection. The hint is echoed as
languageHint; it is
never reported as the detected language.string
A title for the History row.
string
Optional model identifier; defaults to
myvocal_stt_v1.object
Processing options:
separateSpeakers, maxSpeakerCount, alignmentLevel, entityCategories,
redactCategories, redactionStyle, rewriteInstruction, removeDisfluencies,
identifyConversationRoles, vocabularyHints, separateChannels, channelResultMode,
exportFormats, inputEncoding, processingContentStorage and more. Option combinations that the
model does not support are refused at submit with INPUT_INVALID rather than silently dropped.
Completion notifications are options too: options.notifyOnCompletion (true to enable),
options.notifyUrl (a public URL MyVocal posts the signed event to), options.clientMetadata
(opaque caller data echoed back) and options.notificationSigningSecret (the write-only HMAC
key). Polling stays the authoritative result.
See capabilities for the live list.boolean
When
true, this call waits for a terminal state inside the same request.number
The wait budget in milliseconds. If the task is still running at the budget, the request
returns the task with its identity; it is not cancelled.
Batch options
All fields below belong insideoptions. Omit fields you do not need.
| Field | Type and accepted values |
|---|---|
rewriteInstruction | String, at most 2000 characters |
includeSoundEvents | Boolean |
separateSpeakers | Boolean; default true for single-channel processing |
maxSpeakerCount | Integer, 1–32 |
speakerMergeSensitivity | Number, 0.1–0.4; requires speaker separation and no maxSpeakerCount |
alignmentLevel | none, word (default), character |
inputEncoding | other (default, prepared container audio), pcm_s16le_16 (converted 16 kHz mono PCM) |
samplingTemperature | Number, 0–2 |
randomSeed | Non-negative integer |
separateChannels | Boolean; preserve and transcribe channels separately |
channelResultMode | separate or combined |
entityCategories | Array, at most 32 strings; tested examples include pii, phi, pci, all, name |
redactCategories | Array, at most 32 strings; subset of entityCategories |
redactionStyle | redact, replace, mask; requires redactCategories |
removeDisfluencies | Boolean |
identifyConversationRoles | Boolean; requires speaker separation |
vocabularyHints | Array, at most 1000 strings, each at most 128 characters |
processingContentStorage | Boolean; see storage limitations |
exportFormats | Array of supported download formats to pre-generate at completion; every format stays downloadable |
exportOptions | Array of per-format objects, described below |
notifyOnCompletion | Boolean; a notification is sent when notifyUrl is present unless this is false; true requires notifyUrl |
notifyUrl | Public HTTP(S) URL |
clientMetadata | Object, at most two levels and 16 KB |
notificationSigningSecret | Write-only signing secret; see notifications |
Option combinations
rewriteInstructioncannot be combined with entity detection, redaction orseparateChannels.- Multi-channel processing does not separate speakers. Do not combine
separateChannelswithmaxSpeakerCount,speakerMergeSensitivity,identifyConversationRolesorpcm_s16le_16input. The selected audio track can have at most five channels for transcription. - Multi-channel
combinedresults require alignment and cannot be combined with entity detection or redaction. matchKnownSpeakersis unavailable. Do not request it.
exportOptions entry has a required format and optional controls:
| Field | Type / limit |
|---|---|
includeSpeakers, includeTimestamps | Boolean, default true (the txt segment header shows both) |
segmentOnSilenceLongerThanS, maxSegmentDurationS | Number greater than 0 and at most 3600 seconds |
maxSegmentChars | Integer, 1–100000; counts the separators inserted when a segment is rebuilt |
maxCharactersPerLine | Integer, 1–16384; wraps at word boundaries (srt, txt) |
json is the public transcription view. An entry applies to
the pre-generated file and to later downloads of that format. Without timestamps (for example
alignmentLevel: none) SRT has no timed cues and is empty. These options do not change recognition
or billing.
An entity category the model rejects is detected when the task runs: the task ends
FAILED with
INPUT_INVALID. See parameter errors found by the model.Response
Successful REST calls return{ code: 1, message, data }. The fields below are inside data;
JSON downloads are the documented exception and return the transcript view directly.
number
1 for success.string
Result message.
object
Hide properties
Hide properties
string
The MyVocal task id, also the History id.
string
The correlation id for this request.
string
QUEUED, PROCESSING, RECONCILING, COMPLETED, PARTIAL or FAILED.string
myvocal_stt_v1.object
state, planKey, rateVersion, ratePerMinute, reservedCharacters, settledCharacters,
releasedCharacters, billableCharacters and billableDurationMs. Characters and durations are
JSON strings; convert before comparing.