curl --request POST \
--url https://api.goenhance.ai/api/v1/videos/generations \
--header 'Content-Type: application/json' \
--data '
{
"model": "ai-talking-avatar-video",
"video_url": "https://example.com/speaker.mp4",
"audio_url": "https://example.com/speech.mp3",
"prompt": "a man speaking calmly to the camera",
"resolution": "540p",
"video_mode": "normal"
}
'import requests
url = "https://api.goenhance.ai/api/v1/videos/generations"
payload = {
"model": "ai-talking-avatar-video",
"video_url": "https://example.com/speaker.mp4",
"audio_url": "https://example.com/speech.mp3",
"prompt": "a man speaking calmly to the camera",
"resolution": "540p",
"video_mode": "normal"
}
headers = {"Content-Type": "application/json"}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'ai-talking-avatar-video',
video_url: 'https://example.com/speaker.mp4',
audio_url: 'https://example.com/speech.mp3',
prompt: 'a man speaking calmly to the camera',
resolution: '540p',
video_mode: 'normal'
})
};
fetch('https://api.goenhance.ai/api/v1/videos/generations', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.goenhance.ai/api/v1/videos/generations",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => 'ai-talking-avatar-video',
'video_url' => 'https://example.com/speaker.mp4',
'audio_url' => 'https://example.com/speech.mp3',
'prompt' => 'a man speaking calmly to the camera',
'resolution' => '540p',
'video_mode' => 'normal'
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.goenhance.ai/api/v1/videos/generations"
payload := strings.NewReader("{\n \"model\": \"ai-talking-avatar-video\",\n \"video_url\": \"https://example.com/speaker.mp4\",\n \"audio_url\": \"https://example.com/speech.mp3\",\n \"prompt\": \"a man speaking calmly to the camera\",\n \"resolution\": \"540p\",\n \"video_mode\": \"normal\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.goenhance.ai/api/v1/videos/generations")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"ai-talking-avatar-video\",\n \"video_url\": \"https://example.com/speaker.mp4\",\n \"audio_url\": \"https://example.com/speech.mp3\",\n \"prompt\": \"a man speaking calmly to the camera\",\n \"resolution\": \"540p\",\n \"video_mode\": \"normal\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.goenhance.ai/api/v1/videos/generations")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"ai-talking-avatar-video\",\n \"video_url\": \"https://example.com/speaker.mp4\",\n \"audio_url\": \"https://example.com/speech.mp3\",\n \"prompt\": \"a man speaking calmly to the camera\",\n \"resolution\": \"540p\",\n \"video_mode\": \"normal\"\n}"
response = http.request(request)
puts response.read_body{
"code": 0,
"msg": "Success",
"data": {
"img_uuid": "c12b656c-747a-44fd-9c80-add79b0c52d5",
"cost": 5.82
}
}AI Talking Avatar (Video)
AI Talking Avatar from a video. How it works — one video of a person plus one audio track: the person in the video speaks the audio. The whole frame is regenerated with your video as the reference, so head movement and facial expressions follow the speech — not just the lips. To only redraw the mouth and keep every other pixel of your video, use lipsync-video instead; to start from a single photo, use ai-talking-avatar.
Output length — set by video_mode when the video and the audio have different lengths:
| video_mode | output length |
|---|---|
normal (default) | the shorter of the video and the audio |
loop | the audio length; a shorter video is played forward and backward until it covers the audio |
Before any tokens are deducted, GoEnhance transfers both files to its own storage and measures their real durations — that measurement is what you are billed for, and it is also what enforces the 3-60s limit. Because the files are measured and then sent onward from GoEnhance’s storage, swapping the URLs afterwards has no effect.
To make a shorter video, set the optional duration (whole seconds, 3-60): only the first duration seconds are used, and you are billed for duration seconds. It cannot exceed the output length above. With duration set, the video and the audio themselves may be longer than 60 seconds.
Video requirements — up to 300MB. One person, face clearly visible and roughly front-facing for most of the clip. The output keeps the framing of your video.
Audio requirements — up to 50MB. Clear speech with little background noise gives the best lip sync.
Pricing — per second of output, in tokens (USD at $0.02/token):
| resolution | tokens/s | USD/s |
|---|---|---|
| 540p | 1 | $0.02 |
| 720p | 1.5 | $0.03 |
A 30-second clip therefore costs 30 tokens at 540p (= 0.60).At720pthesameclipcosts45tokens(=0.90).
Returns an img_uuid; poll GET /api/v1/jobs/detail (or use custom_callback_url) to get the generated video.
curl --request POST \
--url https://api.goenhance.ai/api/v1/videos/generations \
--header 'Content-Type: application/json' \
--data '
{
"model": "ai-talking-avatar-video",
"video_url": "https://example.com/speaker.mp4",
"audio_url": "https://example.com/speech.mp3",
"prompt": "a man speaking calmly to the camera",
"resolution": "540p",
"video_mode": "normal"
}
'import requests
url = "https://api.goenhance.ai/api/v1/videos/generations"
payload = {
"model": "ai-talking-avatar-video",
"video_url": "https://example.com/speaker.mp4",
"audio_url": "https://example.com/speech.mp3",
"prompt": "a man speaking calmly to the camera",
"resolution": "540p",
"video_mode": "normal"
}
headers = {"Content-Type": "application/json"}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'ai-talking-avatar-video',
video_url: 'https://example.com/speaker.mp4',
audio_url: 'https://example.com/speech.mp3',
prompt: 'a man speaking calmly to the camera',
resolution: '540p',
video_mode: 'normal'
})
};
fetch('https://api.goenhance.ai/api/v1/videos/generations', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.goenhance.ai/api/v1/videos/generations",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'model' => 'ai-talking-avatar-video',
'video_url' => 'https://example.com/speaker.mp4',
'audio_url' => 'https://example.com/speech.mp3',
'prompt' => 'a man speaking calmly to the camera',
'resolution' => '540p',
'video_mode' => 'normal'
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.goenhance.ai/api/v1/videos/generations"
payload := strings.NewReader("{\n \"model\": \"ai-talking-avatar-video\",\n \"video_url\": \"https://example.com/speaker.mp4\",\n \"audio_url\": \"https://example.com/speech.mp3\",\n \"prompt\": \"a man speaking calmly to the camera\",\n \"resolution\": \"540p\",\n \"video_mode\": \"normal\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.goenhance.ai/api/v1/videos/generations")
.header("Content-Type", "application/json")
.body("{\n \"model\": \"ai-talking-avatar-video\",\n \"video_url\": \"https://example.com/speaker.mp4\",\n \"audio_url\": \"https://example.com/speech.mp3\",\n \"prompt\": \"a man speaking calmly to the camera\",\n \"resolution\": \"540p\",\n \"video_mode\": \"normal\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.goenhance.ai/api/v1/videos/generations")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Content-Type"] = 'application/json'
request.body = "{\n \"model\": \"ai-talking-avatar-video\",\n \"video_url\": \"https://example.com/speaker.mp4\",\n \"audio_url\": \"https://example.com/speech.mp3\",\n \"prompt\": \"a man speaking calmly to the camera\",\n \"resolution\": \"540p\",\n \"video_mode\": \"normal\"\n}"
response = http.request(request)
puts response.read_body{
"code": 0,
"msg": "Success",
"data": {
"img_uuid": "c12b656c-747a-44fd-9c80-add79b0c52d5",
"cost": 5.82
}
}Headers
Body
Model name. Must be ai-talking-avatar-video.
ai-talking-avatar-video Video of the person who should speak (up to 300MB). Required. Keep one person in frame, with the face clearly visible for most of the clip.
Speech audio to lip-sync to (up to 50MB). Required.
Optional description of the desired performance, e.g. a man speaking calmly to the camera.
Output resolution. 720p costs 1.5 times as much per second as 540p.
540p, 720p What to do when the video and the audio have different lengths. normal: the output is as long as the shorter of the two. loop: the output follows the audio, and a shorter video is played forward and backward until it covers the audio.
normal, loop Optional. Output length in whole seconds (3-60). Only the first duration seconds are used, and you are billed for duration seconds. Must not exceed the output length set by video_mode. Omit it to use the full length.
3 <= x <= 6015
Optional. A publicly accessible HTTPS URL. When the task status changes (processing / success / failed), GoEnhance sends a POST request to this URL. The request body is identical to the response of GET /api/v1/jobs/detail. If your server does not respond with HTTP 200, the notification is retried up to 3 times, with a 3-second timeout per attempt.
"https://your-server.com/goenhance/callback"
