Lip sync a video
Match a speaker’s mouth to new audio: a translated voiceover, a corrected line, or a different voice. Send the video and the audio, get back a re-synced result.
The flow
POST /v1/workspaces/{workspaceId}/library ← upload video + audio
POST /v1/ai/lipsync ← start the lip-sync job
GET /v1/jobs/{id} ← poll until completed
Send workspace_id and the result lands in that workspace’s Library. Send project_id instead and it lands in that project’s uploads. Both are optional: omit them and the result goes to your Library, as long as the account has a single workspace.
Lip sync is billed per second of video, and the price is fixed before the job is queued. With a project upload the API reads the stored duration. With a Library asset, or a URL, it does not, so pass the video’s length in seconds as top-level duration. Leave it out and the request fails instead of guessing.
The script
The polling helpers used below
const API_KEY = "YOUR_API_KEY";
const API = "https://api.rendley.com/v1";
const headers = {
"Authorization": `Bearer ${API_KEY}`,
"Content-Type": "application/json",
};
// Both endpoints share these three terminal statuses. They differ only in
// their in-progress names: queued/processing vs pending/running.
const TERMINAL = ["completed", "failed", "canceled"];
const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
// Generation and export jobs. The finished job carries output.url.
async function waitForJob(jobId, interval = 5000) {
let job = null;
while (job === null || !TERMINAL.includes(job.status)) {
await sleep(interval);
const response = await fetch(`${API}/jobs/${jobId}`, { headers });
if (!response.ok) {
throw new Error("Job lookup failed: " + response.status);
}
const body = await response.json();
job = body.data;
console.log("Job status:", job.status);
}
if (job.status !== "completed") {
throw new Error("Job " + job.status + ": " + (job.error || ""));
}
return job;
}
// Agent jobs. The finished job carries project_id, thread_id and
// last_message, but no output: export the project to get a file.
// onPause runs when an interactive run stops to ask something.
async function waitForAgentJob(jobId, interval = 5000, onPause) {
let job = null;
while (job === null || !TERMINAL.includes(job.status)) {
// This endpoint long-polls. Sleep anyway, so a fast response
// cannot turn this into a tight request loop.
await sleep(interval);
const response = await fetch(`${API}/agent/jobs/${jobId}`, { headers });
if (!response.ok) {
throw new Error("Job lookup failed: " + response.status);
}
const body = await response.json();
job = body.data;
console.log("Edit status:", job.status);
// Only interactive runs reach this; unattended runs never pause.
if (job.status === "waiting_input" && onPause) {
await onPause(job);
}
}
if (job.status !== "completed") {
throw new Error("The edit did not finish: " + (job.error || job.reason));
}
return job;
}import time
import requests
API_KEY = "YOUR_API_KEY"
API = "https://api.rendley.com/v1"
HEADERS = {"Authorization": f"Bearer {API_KEY}"}
# Both endpoints share these three terminal statuses. They differ only in
# their in-progress names: queued/processing vs pending/running.
TERMINAL = {"completed", "failed", "canceled"}
def wait_for_job(job_id, interval=5):
"""Generation and export jobs. The finished job carries output.url."""
job = None
while job is None or job["status"] not in TERMINAL:
time.sleep(interval)
response = requests.get(f"{API}/jobs/{job_id}", headers=HEADERS)
response.raise_for_status()
job = response.json()["data"]
print("Job status:", job["status"])
if job["status"] != "completed":
raise RuntimeError(f"Job {job['status']}: {job.get('error', '')}")
return job
def wait_for_agent_job(job_id, interval=5, on_pause=None):
"""Agent jobs. The finished job carries project_id, thread_id and
last_message, but no output: export the project to get a file.
on_pause runs when an interactive run stops to ask something.
"""
job = None
while job is None or job["status"] not in TERMINAL:
# This endpoint long-polls. Sleep anyway, so a fast response
# cannot turn this into a tight request loop.
time.sleep(interval)
response = requests.get(f"{API}/agent/jobs/{job_id}", headers=HEADERS)
response.raise_for_status()
job = response.json()["data"]
print("Edit status:", job["status"])
# Only interactive runs reach this; unattended runs never pause.
if job["status"] == "waiting_input" and on_pause:
on_pause(job)
if job["status"] != "completed":
raise RuntimeError("The edit did not finish: " + (job.get("error") or job.get("reason", "")))
return jobimport { readFile } from "node:fs/promises";
import { basename } from "node:path";
const VIDEO = "presenter.mp4";
const AUDIO = "spanish-voiceover.mp3";
// Length of the video in seconds. Lip sync is billed per second, and the
// API does not probe Library assets, so the request has to state it.
const VIDEO_SECONDS = 12;
// Files live inside a workspace, so the upload needs one. An account
// always has at least one, and the first is the default.
async function firstWorkspaceId() {
const response = await fetch(`${API}/workspaces`, { headers });
if (!response.ok) {
throw new Error("Could not list workspaces: " + response.status);
}
const body = await response.json();
return body.data[0].id;
}
// Push the raw bytes into the media library. The response carries the
// file_hash the AI endpoints take as input.
async function upload(path, workspaceId) {
const bytes = await readFile(path);
const query = new URLSearchParams({
file_name: basename(path),
mime_type: path.endsWith(".mp3") ? "audio/mpeg" : "video/mp4",
});
const url = `${API}/workspaces/${workspaceId}/library?${query}`;
const response = await fetch(url, {
method: "POST",
headers: {
...headers,
"Content-Type": "application/octet-stream",
},
body: bytes,
});
if (!response.ok) {
throw new Error("Upload failed: " + response.status);
}
const body = await response.json();
return body.data;
}
// Start the lip-sync job. This returns straight away with a job id.
async function startLipSync(videoHash, audioHash, workspaceId) {
const response = await fetch(`${API}/ai/lipsync`, {
method: "POST",
headers: {
...headers,
"Content-Type": "application/json",
},
// To skip an upload, drop its hash and send a public https URL as
// video_file_url or audio_file_url at the top level, next to params.
body: JSON.stringify({
workspace_id: workspaceId,
duration: VIDEO_SECONDS,
params: {
video_file_hash: videoHash,
audio_file_hash: audioHash,
},
}),
});
if (!response.ok) {
throw new Error("Could not start the job: " + response.status);
}
const body = await response.json();
return body.data.job_id;
}
const workspaceId = await firstWorkspaceId();
// Lip sync needs both sides: the footage and the audio to match it to.
const video = await upload(VIDEO, workspaceId);
const audio = await upload(AUDIO, workspaceId);
const jobId = await startLipSync(video.file_hash, audio.file_hash, workspaceId);
const job = await waitForJob(jobId);
// The signed URL is the result. Fetch it, pipe it to your storage,
// or hand it to the browser.
console.log(job.output.url);import os
import mimetypes
import requests
VIDEO = "presenter.mp4"
AUDIO = "spanish-voiceover.mp3"
# Length of the video in seconds. Lip sync is billed per second, and the
# API does not probe Library assets, so the request has to state it.
VIDEO_SECONDS = 12
def first_workspace_id():
"""Files live inside a workspace. An account always has at least one,
and the first is the default."""
response = requests.get(f"{API}/workspaces", headers=HEADERS)
response.raise_for_status()
return response.json()["data"][0]["id"]
def upload(path, workspace_id):
"""Push the raw bytes into the media library. The response carries the
file_hash the AI endpoints take as input."""
mime = mimetypes.guess_type(path)[0] or "application/octet-stream"
with open(path, "rb") as f:
response = requests.post(
f"{API}/workspaces/{workspace_id}/library",
headers={**HEADERS, "Content-Type": "application/octet-stream"},
params={"file_name": os.path.basename(path), "mime_type": mime},
data=f,
)
response.raise_for_status()
return response.json()["data"]
def start_lip_sync(video_hash, audio_hash, workspace_id):
"""Start the lip-sync job. Returns straight away with a job id."""
response = requests.post(
f"{API}/ai/lipsync",
headers=HEADERS,
# To skip an upload, drop its hash and send a public https URL as
# video_file_url or audio_file_url at the top level, next to params.
json={
"workspace_id": workspace_id,
"duration": VIDEO_SECONDS,
"params": {
"video_file_hash": video_hash,
"audio_file_hash": audio_hash,
},
},
)
response.raise_for_status()
return response.json()["data"]["job_id"]
workspace_id = first_workspace_id()
# Lip sync needs both sides: the footage and the audio to match it to.
video = upload(VIDEO, workspace_id)
audio = upload(AUDIO, workspace_id)
job_id = start_lip_sync(video["file_hash"], audio["file_hash"], workspace_id)
job = wait_for_job(job_id)
# The signed URL is the result. Download it, or hand it straight to
# whatever consumes the video next.
result = requests.get(job["output"]["url"])
with open("presenter-spanish.mp4", "wb") as f:
f.write(result.content)
print("Saved presenter-spanish.mp4")Common use cases
- Multilingual dubbing, generate a translated voiceover with text-to-speech and lip-sync it to the original footage.
- Script corrections, re-record one line and sync the speaker’s mouth to the fix.
- AI avatars, pair a stock presenter video with generated speech for personalized outreach.
- Accessibility, re-voice content in a clearer voice while keeping the visual presentation.
End-to-end dubbing pipeline
Dub a video into another language in four steps:
- Transcribe the original audio.
- Translate the transcript (your own logic or an LLM).
- Generate the translated voiceover with text-to-speech.
- Lip-sync the original video to the new audio (this page).