Rendley docs

Lip sync a video

Match a speaker’s mouth to new audio: a translated voiceover, a corrected line, or a different voice. Send the video and the audio, get back a re-synced result.

The flow

POST /v1/workspaces/{workspaceId}/library   ← upload video + audio
POST /v1/ai/lipsync                          ← start the lip-sync job
GET  /v1/jobs/{id}                           ← poll until completed

Send workspace_id and the result lands in that workspace’s Library. Send project_id instead and it lands in that project’s uploads. Both are optional: omit them and the result goes to your Library, as long as the account has a single workspace.

Lip sync is billed per second of video, and the price is fixed before the job is queued. With a project upload the API reads the stored duration. With a Library asset, or a URL, it does not, so pass the video’s length in seconds as top-level duration. Leave it out and the request fails instead of guessing.

The script

The polling helpers used below
const API_KEY = "YOUR_API_KEY";
  const API = "https://api.rendley.com/v1";

  const headers = {
    "Authorization": `Bearer ${API_KEY}`,
    "Content-Type": "application/json",
  };

  // Both endpoints share these three terminal statuses. They differ only in
  // their in-progress names: queued/processing vs pending/running.
  const TERMINAL = ["completed", "failed", "canceled"];

  const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));


  // Generation and export jobs. The finished job carries output.url.
  async function waitForJob(jobId, interval = 5000) {
    let job = null;

    while (job === null || !TERMINAL.includes(job.status)) {
      await sleep(interval);

      const response = await fetch(`${API}/jobs/${jobId}`, { headers });

      if (!response.ok) {
        throw new Error("Job lookup failed: " + response.status);
      }

      const body = await response.json();

      job = body.data;
      console.log("Job status:", job.status);
    }

    if (job.status !== "completed") {
      throw new Error("Job " + job.status + ": " + (job.error || ""));
    }

    return job;
  }


  // Agent jobs. The finished job carries project_id, thread_id and
  // last_message, but no output: export the project to get a file.
  // onPause runs when an interactive run stops to ask something.
  async function waitForAgentJob(jobId, interval = 5000, onPause) {
    let job = null;

    while (job === null || !TERMINAL.includes(job.status)) {
      // This endpoint long-polls. Sleep anyway, so a fast response
      // cannot turn this into a tight request loop.
      await sleep(interval);

      const response = await fetch(`${API}/agent/jobs/${jobId}`, { headers });

      if (!response.ok) {
        throw new Error("Job lookup failed: " + response.status);
      }

      const body = await response.json();

      job = body.data;
      console.log("Edit status:", job.status);

      // Only interactive runs reach this; unattended runs never pause.
      if (job.status === "waiting_input" && onPause) {
        await onPause(job);
      }
    }

    if (job.status !== "completed") {
      throw new Error("The edit did not finish: " + (job.error || job.reason));
    }

    return job;
  }
import time
  import requests

  API_KEY = "YOUR_API_KEY"
  API = "https://api.rendley.com/v1"

  HEADERS = {"Authorization": f"Bearer {API_KEY}"}

  # Both endpoints share these three terminal statuses. They differ only in
  # their in-progress names: queued/processing vs pending/running.
  TERMINAL = {"completed", "failed", "canceled"}


  def wait_for_job(job_id, interval=5):
      """Generation and export jobs. The finished job carries output.url."""
      job = None

      while job is None or job["status"] not in TERMINAL:
          time.sleep(interval)

          response = requests.get(f"{API}/jobs/{job_id}", headers=HEADERS)
          response.raise_for_status()

          job = response.json()["data"]
          print("Job status:", job["status"])

      if job["status"] != "completed":
          raise RuntimeError(f"Job {job['status']}: {job.get('error', '')}")

      return job


  def wait_for_agent_job(job_id, interval=5, on_pause=None):
      """Agent jobs. The finished job carries project_id, thread_id and
      last_message, but no output: export the project to get a file.

      on_pause runs when an interactive run stops to ask something.
      """
      job = None

      while job is None or job["status"] not in TERMINAL:
          # This endpoint long-polls. Sleep anyway, so a fast response
          # cannot turn this into a tight request loop.
          time.sleep(interval)

          response = requests.get(f"{API}/agent/jobs/{job_id}", headers=HEADERS)
          response.raise_for_status()

          job = response.json()["data"]
          print("Edit status:", job["status"])

          # Only interactive runs reach this; unattended runs never pause.
          if job["status"] == "waiting_input" and on_pause:
              on_pause(job)

      if job["status"] != "completed":
          raise RuntimeError("The edit did not finish: " + (job.get("error") or job.get("reason", "")))

      return job
import { readFile } from "node:fs/promises";
import { basename } from "node:path";

const VIDEO = "presenter.mp4";
const AUDIO = "spanish-voiceover.mp3";
// Length of the video in seconds. Lip sync is billed per second, and the
// API does not probe Library assets, so the request has to state it.
const VIDEO_SECONDS = 12;


// Files live inside a workspace, so the upload needs one. An account
// always has at least one, and the first is the default.
async function firstWorkspaceId() {
  const response = await fetch(`${API}/workspaces`, { headers });

  if (!response.ok) {
    throw new Error("Could not list workspaces: " + response.status);
  }

  const body = await response.json();
  return body.data[0].id;
}


// Push the raw bytes into the media library. The response carries the
// file_hash the AI endpoints take as input.
async function upload(path, workspaceId) {
  const bytes = await readFile(path);

  const query = new URLSearchParams({
    file_name: basename(path),
    mime_type: path.endsWith(".mp3") ? "audio/mpeg" : "video/mp4",
  });

  const url = `${API}/workspaces/${workspaceId}/library?${query}`;

  const response = await fetch(url, {
    method: "POST",
    headers: {
      ...headers,
      "Content-Type": "application/octet-stream",
    },
    body: bytes,
  });

  if (!response.ok) {
    throw new Error("Upload failed: " + response.status);
  }

  const body = await response.json();
  return body.data;
}


// Start the lip-sync job. This returns straight away with a job id.
async function startLipSync(videoHash, audioHash, workspaceId) {
  const response = await fetch(`${API}/ai/lipsync`, {
    method: "POST",
    headers: {
      ...headers,
      "Content-Type": "application/json",
    },
    // To skip an upload, drop its hash and send a public https URL as
    // video_file_url or audio_file_url at the top level, next to params.
    body: JSON.stringify({
      workspace_id: workspaceId,
      duration: VIDEO_SECONDS,
      params: {
        video_file_hash: videoHash,
        audio_file_hash: audioHash,
      },
    }),
  });

  if (!response.ok) {
    throw new Error("Could not start the job: " + response.status);
  }

  const body = await response.json();
  return body.data.job_id;
}


const workspaceId = await firstWorkspaceId();

// Lip sync needs both sides: the footage and the audio to match it to.
const video = await upload(VIDEO, workspaceId);
const audio = await upload(AUDIO, workspaceId);

const jobId = await startLipSync(video.file_hash, audio.file_hash, workspaceId);
const job = await waitForJob(jobId);

// The signed URL is the result. Fetch it, pipe it to your storage,
// or hand it to the browser.
console.log(job.output.url);
import os
import mimetypes
import requests

VIDEO = "presenter.mp4"
AUDIO = "spanish-voiceover.mp3"
# Length of the video in seconds. Lip sync is billed per second, and the
# API does not probe Library assets, so the request has to state it.
VIDEO_SECONDS = 12


def first_workspace_id():
    """Files live inside a workspace. An account always has at least one,
    and the first is the default."""
    response = requests.get(f"{API}/workspaces", headers=HEADERS)
    response.raise_for_status()

    return response.json()["data"][0]["id"]


def upload(path, workspace_id):
    """Push the raw bytes into the media library. The response carries the
    file_hash the AI endpoints take as input."""
    mime = mimetypes.guess_type(path)[0] or "application/octet-stream"

    with open(path, "rb") as f:
        response = requests.post(
            f"{API}/workspaces/{workspace_id}/library",
            headers={**HEADERS, "Content-Type": "application/octet-stream"},
            params={"file_name": os.path.basename(path), "mime_type": mime},
            data=f,
        )

    response.raise_for_status()

    return response.json()["data"]


def start_lip_sync(video_hash, audio_hash, workspace_id):
    """Start the lip-sync job. Returns straight away with a job id."""
    response = requests.post(
        f"{API}/ai/lipsync",
        headers=HEADERS,
        # To skip an upload, drop its hash and send a public https URL as
        # video_file_url or audio_file_url at the top level, next to params.
        json={
            "workspace_id": workspace_id,
            "duration": VIDEO_SECONDS,
            "params": {
                "video_file_hash": video_hash,
                "audio_file_hash": audio_hash,
            },
        },
    )
    response.raise_for_status()

    return response.json()["data"]["job_id"]


workspace_id = first_workspace_id()

# Lip sync needs both sides: the footage and the audio to match it to.
video = upload(VIDEO, workspace_id)
audio = upload(AUDIO, workspace_id)

job_id = start_lip_sync(video["file_hash"], audio["file_hash"], workspace_id)
job = wait_for_job(job_id)

# The signed URL is the result. Download it, or hand it straight to
# whatever consumes the video next.
result = requests.get(job["output"]["url"])

with open("presenter-spanish.mp4", "wb") as f:
    f.write(result.content)

print("Saved presenter-spanish.mp4")

Common use cases

  • Multilingual dubbing, generate a translated voiceover with text-to-speech and lip-sync it to the original footage.
  • Script corrections, re-record one line and sync the speaker’s mouth to the fix.
  • AI avatars, pair a stock presenter video with generated speech for personalized outreach.
  • Accessibility, re-voice content in a clearer voice while keeping the visual presentation.

End-to-end dubbing pipeline

Dub a video into another language in four steps:

  1. Transcribe the original audio.
  2. Translate the transcript (your own logic or an LLM).
  3. Generate the translated voiceover with text-to-speech.
  4. Lip-sync the original video to the new audio (this page).