curl --request POST \
--url https://api.goenhance.ai/api/v1/audio/generations \
--header 'Content-Type: application/json' \
--data '
{
"model": "qwen3-tts-voice-clone",
"text": "This voice was cloned from a short recording.",
"reference_audio_url": "https://your-cdn.com/voice-sample.mp3",
"language": "english"
}
'import requests
url = "https://api.goenhance.ai/api/v1/audio/generations"
payload = {
"model": "qwen3-tts-voice-clone",
"text": "This voice was cloned from a short recording.",
"reference_audio_url": "https://your-cdn.com/voice-sample.mp3",
"language": "english"
}
headers = {"Content-Type": "application/json"}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'qwen3-tts-voice-clone',
text: 'This voice was cloned from a short recording.',
reference_audio_url: 'https://your-cdn.com/voice-sample.mp3',
language: 'english'
})
};
fetch('https://api.goenhance.ai/api/v1/audio/generations', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.goenhance.ai/api/v1/audio/generations",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => 'qwen3-tts-voice-clone',
'text' => 'This voice was cloned from a short recording.',
'reference_audio_url' => 'https://your-cdn.com/voice-sample.mp3',
'language' => 'english'
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.goenhance.ai/api/v1/audio/generations"
payload := strings.NewReader("{\n \"model\": \"qwen3-tts-voice-clone\",\n \"text\": \"This voice was cloned from a short recording.\",\n \"reference_audio_url\": \"https://your-cdn.com/voice-sample.mp3\",\n \"language\": \"english\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.goenhance.ai/api/v1/audio/generations")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"qwen3-tts-voice-clone\",\n \"text\": \"This voice was cloned from a short recording.\",\n \"reference_audio_url\": \"https://your-cdn.com/voice-sample.mp3\",\n \"language\": \"english\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.goenhance.ai/api/v1/audio/generations")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"qwen3-tts-voice-clone\",\n \"text\": \"This voice was cloned from a short recording.\",\n \"reference_audio_url\": \"https://your-cdn.com/voice-sample.mp3\",\n \"language\": \"english\"\n}"
response = http.request(request)
puts response.read_body{
"code": 0,
"msg": "Success",
"data": {
"img_uuid": "c12b656c-747a-44fd-9c80-add79b0c52d5",
"cost": 1
}
}Qwen3 TTS Voice Clone
Clone a voice from a short recording and have it read any text. The reference and the text don’t need to share a language — an English recording can read Chinese, and vice versa. Speaks Chinese, English, Japanese, Korean, German, French, Russian, Portuguese, Spanish and Italian. Returns an mp3 (default) or WAV URL (24 kHz, mono).
There is no enrollment step and no voice ID to manage: send the reference recording with each request.
For the best clone, use 5–20 seconds of clear speech from a single speaker, without background music or noise — whatever is in the recording (other voices, music, room echo) tends to carry over. The pace follows the reference, so there is no speed parameter.
Pricing — billed by text length, not by audio duration:
| Unit | Tokens | USD |
|---|---|---|
| every 200 characters (rounded up) | 1 | $0.02 |
A 350-character request is billed as 2 units (2 tokens = $0.04). Anything from 1 to 200 characters costs 1 unit.
Returns an img_uuid; poll GET /api/v1/jobs/detail (or use custom_callback_url) to get the generated audio. The result arrives as a json array item with type: "audio" and the audio URL in value. If generation fails, the task ends as failed and the tokens are refunded automatically.
curl --request POST \
--url https://api.goenhance.ai/api/v1/audio/generations \
--header 'Content-Type: application/json' \
--data '
{
"model": "qwen3-tts-voice-clone",
"text": "This voice was cloned from a short recording.",
"reference_audio_url": "https://your-cdn.com/voice-sample.mp3",
"language": "english"
}
'import requests
url = "https://api.goenhance.ai/api/v1/audio/generations"
payload = {
"model": "qwen3-tts-voice-clone",
"text": "This voice was cloned from a short recording.",
"reference_audio_url": "https://your-cdn.com/voice-sample.mp3",
"language": "english"
}
headers = {"Content-Type": "application/json"}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'qwen3-tts-voice-clone',
text: 'This voice was cloned from a short recording.',
reference_audio_url: 'https://your-cdn.com/voice-sample.mp3',
language: 'english'
})
};
fetch('https://api.goenhance.ai/api/v1/audio/generations', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.goenhance.ai/api/v1/audio/generations",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => 'qwen3-tts-voice-clone',
'text' => 'This voice was cloned from a short recording.',
'reference_audio_url' => 'https://your-cdn.com/voice-sample.mp3',
'language' => 'english'
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.goenhance.ai/api/v1/audio/generations"
payload := strings.NewReader("{\n \"model\": \"qwen3-tts-voice-clone\",\n \"text\": \"This voice was cloned from a short recording.\",\n \"reference_audio_url\": \"https://your-cdn.com/voice-sample.mp3\",\n \"language\": \"english\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.goenhance.ai/api/v1/audio/generations")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"qwen3-tts-voice-clone\",\n \"text\": \"This voice was cloned from a short recording.\",\n \"reference_audio_url\": \"https://your-cdn.com/voice-sample.mp3\",\n \"language\": \"english\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.goenhance.ai/api/v1/audio/generations")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"qwen3-tts-voice-clone\",\n \"text\": \"This voice was cloned from a short recording.\",\n \"reference_audio_url\": \"https://your-cdn.com/voice-sample.mp3\",\n \"language\": \"english\"\n}"
response = http.request(request)
puts response.read_body{
"code": 0,
"msg": "Success",
"data": {
"img_uuid": "c12b656c-747a-44fd-9c80-add79b0c52d5",
"cost": 1
}
}Headers
Body
Model name. Must be qwen3-tts-voice-clone.
qwen3-tts-voice-clone Text to synthesize. Required. Max 5000 characters. Billing is based on this length. Long texts are read in sentence-sized parts and joined into one file, so a single request can cover a whole article.
1 - 5000Required. Public HTTPS URL of the voice to clone. Max 50MB. Common audio formats (mp3, wav, m4a, aac, ogg, flac) and video files (the audio track is used) are accepted. Needs at least 2 seconds of speech. Unreachable or oversized links are rejected before any tokens are charged; if the file turns out not to be audio or contains no speech, the task fails and is refunded.
"https://your-cdn.com/voice-sample.mp3"
Optional. The exact words spoken in the reference recording. When given, it must match the recording word for word, and the whole recording is used (up to 60 seconds). When omitted, the first 20 seconds or less (cut at a natural pause) are used and transcribed automatically.
1000Language of text. Optional — defaults to auto. Case-insensitive; the two-letter codes are accepted too. Supported: Chinese, English, Japanese, Korean, German, French, Russian, Portuguese, Spanish and Italian. Setting it explicitly when you know the language gives the most stable result.
auto, chinese, english, japanese, korean, german, french, russian, portuguese, spanish, italian, zh, en, ja, ko, de, fr, ru, pt, es, it Audio format. Optional — mp3 (default, 128 kbps) or wav (16-bit). Both are 24 kHz mono.
mp3, wav Optional. A publicly accessible HTTPS URL. When the task status changes (processing / success / failed), GoEnhance sends a POST request to this URL. The request body is identical to the response of GET /api/v1/jobs/detail. If your server does not respond with HTTP 200, the notification is retried up to 3 times, with a 3-second timeout per attempt.
"https://your-server.com/goenhance/callback"
