Voice Cloned TTS Stream
curl --request POST \
--url https://api.vachana.ai/api/v1/tts/sse \
--header 'Content-Type: application/json' \
--header 'X-API-Key-ID: <api-key>' \
--data '
{
"text": "नमस्ते, आप कैसे हैं?",
"speaker_embedding": {
"embedding": "<base64-encoded embedding>",
"shape": [
1,
768
],
"dtype": "torch.bfloat16"
},
"audio_config": {
"sample_rate": 44100,
"num_channels": 1,
"sample_width": 2,
"encoding": "linear_pcm",
"container": "wav"
}
}
'import requests
url = "https://api.vachana.ai/api/v1/tts/sse"
payload = {
"text": "नमस्ते, आप कैसे हैं?",
"speaker_embedding": {
"embedding": "<base64-encoded embedding>",
"shape": [1, 768],
"dtype": "torch.bfloat16"
},
"audio_config": {
"sample_rate": 44100,
"num_channels": 1,
"sample_width": 2,
"encoding": "linear_pcm",
"container": "wav"
}
}
headers = {
"X-API-Key-ID": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'X-API-Key-ID': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
text: 'नमस्ते, आप कैसे हैं?',
speaker_embedding: {
embedding: '<base64-encoded embedding>',
shape: [1, 768],
dtype: 'torch.bfloat16'
},
audio_config: {
sample_rate: 44100,
num_channels: 1,
sample_width: 2,
encoding: 'linear_pcm',
container: 'wav'
}
})
};
fetch('https://api.vachana.ai/api/v1/tts/sse', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));"event: start\ndata: {\"status\": \"streaming_started\"}\n\nevent: chunk\ndata: {\"chunk_index\": 1, \"audio\": \"<base64>\", \"is_final\": false}\n\nevent: complete\ndata: {\"chunk_index\": 2, \"audio\": \"\", \"is_final\": true}"{
"success": false,
"error": {
"type": "INVALID_REQUEST_ERROR",
"message": "Invalid text or audio configuration."
}
}{
"success": false,
"error": {
"type": "RATE_LIMIT_ERROR",
"message": "Rate limit exceeded. Please try again later."
}
}{
"success": false,
"error": {
"type": "API_ERROR",
"message": "An unexpected error occurred while processing."
}
}{
"success": false,
"error": {
"type": "SERVICE_UNAVAILABLE",
"message": "Text-to-speech service is temporarily unavailable."
}
}Voice Cloning
Voice Cloned TTS (Streaming)
Stream cloned voice audio in chunks via Server-Sent Events.
POST
/
api
/
v1
/
tts
/
sse
Voice Cloned TTS Stream
curl --request POST \
--url https://api.vachana.ai/api/v1/tts/sse \
--header 'Content-Type: application/json' \
--header 'X-API-Key-ID: <api-key>' \
--data '
{
"text": "नमस्ते, आप कैसे हैं?",
"speaker_embedding": {
"embedding": "<base64-encoded embedding>",
"shape": [
1,
768
],
"dtype": "torch.bfloat16"
},
"audio_config": {
"sample_rate": 44100,
"num_channels": 1,
"sample_width": 2,
"encoding": "linear_pcm",
"container": "wav"
}
}
'import requests
url = "https://api.vachana.ai/api/v1/tts/sse"
payload = {
"text": "नमस्ते, आप कैसे हैं?",
"speaker_embedding": {
"embedding": "<base64-encoded embedding>",
"shape": [1, 768],
"dtype": "torch.bfloat16"
},
"audio_config": {
"sample_rate": 44100,
"num_channels": 1,
"sample_width": 2,
"encoding": "linear_pcm",
"container": "wav"
}
}
headers = {
"X-API-Key-ID": "<api-key>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'X-API-Key-ID': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
text: 'नमस्ते, आप कैसे हैं?',
speaker_embedding: {
embedding: '<base64-encoded embedding>',
shape: [1, 768],
dtype: 'torch.bfloat16'
},
audio_config: {
sample_rate: 44100,
num_channels: 1,
sample_width: 2,
encoding: 'linear_pcm',
container: 'wav'
}
})
};
fetch('https://api.vachana.ai/api/v1/tts/sse', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));"event: start\ndata: {\"status\": \"streaming_started\"}\n\nevent: chunk\ndata: {\"chunk_index\": 1, \"audio\": \"<base64>\", \"is_final\": false}\n\nevent: complete\ndata: {\"chunk_index\": 2, \"audio\": \"\", \"is_final\": true}"{
"success": false,
"error": {
"type": "INVALID_REQUEST_ERROR",
"message": "Invalid text or audio configuration."
}
}{
"success": false,
"error": {
"type": "RATE_LIMIT_ERROR",
"message": "Rate limit exceeded. Please try again later."
}
}{
"success": false,
"error": {
"type": "API_ERROR",
"message": "An unexpected error occurred while processing."
}
}{
"success": false,
"error": {
"type": "SERVICE_UNAVAILABLE",
"message": "Text-to-speech service is temporarily unavailable."
}
}Overview
1
Step 1 — Generate a voice embedding
Upload a reference audio clip to get a
speaker_embedding. See Voice Clone Embeddings.2
Step 2 — Synthesize with your cloned voice — this page
Pass the
speaker_embedding from Step 1 to this endpoint to stream cloned voice audio progressively.speaker_embedding from Voice Clone Embeddings to use your cloned voice. Reduces latency compared to Voice Cloned TTS REST. For the lowest latency, see Voice Cloned TTS Realtime.
Endpoint
POST https://api.vachana.ai/api/v1/tts/sse
Authentication
| Header | Required | Description | Example |
|---|---|---|---|
X-API-Key-ID | Yes | Your API key for authentication | your-api-key-id |
Content-Type | Yes | Must be application/json | application/json |
Request Body
string
required
The text to synthesize into speech
string
Voice cloning model. Use
vachana-vc-v1. Automatically set when speaker_embedding is provided — you may omit this field.object
required
Audio output configuration
Show properties
Show properties
object
required
Response
The server streams audio data via Server-Sent Events (SSE). Each event contains a chunk of audio data encoded in base64.Event Types
event
Sent when synthesis begins.
event: start
data: {"status": "streaming_started", "text": "नमस्ते"}
event
Contains base64-encoded audio in the
audio field.event: chunk
data: {"chunk_index": 1, "audio": "<base64>", "is_final": false}
event
Signals the end of the audio stream.
event: complete
data: {"chunk_index": 2, "audio": "", "is_final": true}
Example Request
curl -X POST https://api.vachana.ai/api/v1/tts/sse \
-H "X-API-Key-ID: your-api-key-id" \
-H "Content-Type: application/json" \
-H "Accept: text/event-stream" \
-N \
-d '{
"text": "नमस्ते, आप कैसे हैं?",
"model": "vachana-vc-v1",
"audio_config": {
"sample_rate": 44100,
"encoding": "linear_pcm",
"container": "wav"
},
"speaker_embedding": {
"embedding": "your-embedding-string",
"shape": [1, 768],
"dtype": "torch.bfloat16"
}
}'
import requests
import base64
url = "https://api.vachana.ai/api/v1/tts/sse"
headers = {
"X-API-Key-ID": "your-api-key-id",
"Content-Type": "application/json",
"Accept": "text/event-stream"
}
payload = {
"text": "नमस्ते, आप कैसे हैं?",
"model": "vachana-vc-v1",
"audio_config": {
"sample_rate": 44100,
"encoding": "linear_pcm",
"container": "wav"
},
"speaker_embedding": {
"embedding": "your-embedding-string",
"shape": [1, 768],
"dtype": "torch.bfloat16"
}
}
response = requests.post(url, headers=headers, json=payload, stream=True)
audio_chunks = []
for line in response.iter_lines():
if line:
line = line.decode('utf-8')
if line.startswith('data: '):
data = line[6:]
if data.startswith('{'):
# Completed event
print("Stream completed")
else:
# Audio chunk
audio_chunks.append(base64.b64decode(data))
# Combine and save audio
with open("audio.wav", "wb") as f:
for chunk in audio_chunks:
f.write(chunk)
print("Audio saved successfully")
const url = "https://api.vachana.ai/api/v1/tts/sse";
const headers = {
"X-API-Key-ID": "your-api-key-id",
"Content-Type": "application/json",
"Accept": "text/event-stream"
};
const payload = {
text: "नमस्ते, आप कैसे हैं?",
model: "vachana-vc-v1",
audio_config: {
sample_rate: 44100,
encoding: "linear_pcm",
container: "wav"
},
speaker_embedding: {
embedding: "your-embedding-string",
shape: [1, 768],
dtype: "torch.bfloat16"
}
};
const eventSource = new EventSource(url);
const audioChunks = [];
eventSource.addEventListener("chunk", (event) => {
const payload = JSON.parse(event.data);
if (payload.audio) {
const audioData = atob(payload.audio);
const bytes = new Uint8Array(audioData.length);
for (let i = 0; i < audioData.length; i++) {
bytes[i] = audioData.charCodeAt(i);
}
audioChunks.push(bytes);
}
});
eventSource.addEventListener("complete", (event) => {
console.log("Stream complete");
// Combine chunks and create blob
const blob = new Blob(audioChunks, { type: "audio/wav" });
const url = window.URL.createObjectURL(blob);
const a = document.createElement("a");
a.href = url;
a.download = "audio.wav";
a.click();
eventSource.close();
});
eventSource.onerror = (error) => {
console.error("SSE Error:", error);
eventSource.close();
};
// Send the request
fetch(url, {
method: "POST",
headers: headers,
body: JSON.stringify(payload)
});
Error Responses
Bad Request
Invalid text or audio configuration
{
"success": false,
"error": {
"type": "INVALID_REQUEST_ERROR",
"message": "Invalid text or audio configuration."
}
}
Too Many Requests
Rate limit exceeded
{
"success": false,
"error": {
"type": "RATE_LIMIT_ERROR",
"message": "Rate limit exceeded. Please try again later."
}
}
Internal Server Error
Unexpected error occurred
{
"success": false,
"error": {
"type": "API_ERROR",
"message": "An unexpected error occurred while processing."
}
}
Authorizations
Headers
Body
application/json
Request body for voice cloned TTS inference.
The text to synthesize into speech.
Voice clone embedding from the Voice Clone Embeddings endpoint.
Show child attributes
Show child attributes
Audio output configuration.
Show child attributes
Show child attributes
Voice cloning model. Automatically set to vachana-vc-v1 when speaker_embedding is provided.
Available options:
vachana-vc-v1 Response
Successful Server-Sent Events stream
The response is of type string.
Was this page helpful?