diff --git a/examples/rime-multilingual-dub/.env.example b/examples/rime-multilingual-dub/.env.example new file mode 100644 index 0000000..4ccb9da --- /dev/null +++ b/examples/rime-multilingual-dub/.env.example @@ -0,0 +1,14 @@ +# Your Shotstack API key: https://dashboard.shotstack.io +SHOTSTACK_API_KEY= + +# Your Rime API key: https://rime.ai +RIME_API_KEY= + +# A public link to the video you want to dub +SOURCE_VIDEO= + +# Optional. Another Rime model: https://docs.rime.ai +RIME_MODEL= + +# Optional. Set to stage for free watermarked test renders. +SHOTSTACK_ENV= diff --git a/examples/rime-multilingual-dub/.gitignore b/examples/rime-multilingual-dub/.gitignore new file mode 100644 index 0000000..eb2f977 --- /dev/null +++ b/examples/rime-multilingual-dub/.gitignore @@ -0,0 +1,2 @@ +.env +output/ diff --git a/examples/rime-multilingual-dub/README.md b/examples/rime-multilingual-dub/README.md new file mode 100644 index 0000000..fa9c52d --- /dev/null +++ b/examples/rime-multilingual-dub/README.md @@ -0,0 +1,59 @@ +# Rime multilingual dub + +Make one video speak several languages. The script reads `translations.csv`, +asks Rime for a voice-over in each language, then renders the same source video +with the original audio silenced, the new voice in its place, and captions +burned in. You get one MP4 per language in `output/`. + +You supply the translated lines. The script does not translate anything, so +nothing is guessed on your behalf. + +## Requirements + +- A [Shotstack account](https://dashboard.shotstack.io/register) and your + **production** API key +- A [Rime account](https://rime.ai) and an API key +- A public link to the video you want to dub +- Node.js 20 or later + +Each row generates one voice-over at Rime and one Shotstack render. The captions +are transcribed from the new voice-over, which consumes generation credits. +Remove the first track in `template.json` to render without captions. + +## Setup + +```bash +git clone https://github.com/shotstack/shotstack-cookbook.git +cd shotstack-cookbook/examples/rime-multilingual-dub +cp .env.example .env +``` + +Load the file into your shell. Do this in each new terminal: + +```bash +set -a +source .env +set +a +``` + +Set `SOURCE_VIDEO` to your video. Edit `translations.csv`: one row per language, +with the Rime `speaker` voice and the translated `text`. The voices are listed +in the [Rime documentation](https://docs.rime.ai). + +## Run + +```bash +node multilingual-dub.mjs +``` + +## What happens + +For each row the script sends the text to Rime and receives an MP3. It uploads +the MP3 to the Shotstack Ingest API and waits until the source is ready. It puts +the audio URL and the source video into the merge fields of `template.json`, +then renders a 1080p video where the original audio is set to zero volume and +the new voice plays over it. Each MP4 is downloaded to `output/`. + +Keep the voice-over close to the length of the original video. The voice track +is capped to the length of the source video, so a line that runs long is cut +off at the end of the picture rather than extending the render. diff --git a/examples/rime-multilingual-dub/multilingual-dub.mjs b/examples/rime-multilingual-dub/multilingual-dub.mjs new file mode 100644 index 0000000..00ec19d --- /dev/null +++ b/examples/rime-multilingual-dub/multilingual-dub.mjs @@ -0,0 +1,303 @@ +import { readFile, writeFile, mkdir } from 'node:fs/promises'; +import { setTimeout as delay } from 'node:timers/promises'; +import { createHash } from 'node:crypto'; + +// The production environment. For free watermarked test renders, set +// SHOTSTACK_ENV=stage in .env and use your sandbox API key. +const STAGE = process.env.SHOTSTACK_ENV === 'stage' ? 'stage' : 'v1'; +const EDIT_URL = `https://api.shotstack.io/edit/${STAGE}`; +const INGEST_URL = `https://api.shotstack.io/ingest/${STAGE}`; +const POLL_INTERVAL_MS = 5_000; +const MAX_WAIT_MS = 15 * 60 * 1_000; + +function requireEnv(name, hint) { + const value = process.env[name]; + + if (!value) { + console.error(`Set the ${name} environment variable first. ${hint}`); + process.exit(1); + } + + return value; +} + +async function request(url, options, service) { + let response; + + try { + response = await fetch(url, { + signal: AbortSignal.timeout(120_000), + ...options + }); + } catch (error) { + throw new Error( + `Could not reach the ${service} API. ` + + 'Check your network connection and try again.', + { cause: error } + ); + } + + if (!response.ok) { + const detail = (await response.text()).slice(0, 300); + throw new Error(`${service} returned ${response.status}: ${detail}`); + } + + return response; +} + +// A small CSV reader. It accepts quoted fields and commas inside quotes. +function parseCsv(text) { + const rows = []; + let row = []; + let field = ''; + let quoted = false; + + for (let i = 0; i < text.length; i += 1) { + const char = text[i]; + + if (quoted) { + if (char === '"' && text[i + 1] === '"') { + field += '"'; + i += 1; + } else if (char === '"') { + quoted = false; + } else { + field += char; + } + } else if (char === '"') { + quoted = true; + } else if (char === ',') { + row.push(field.trim()); + field = ''; + } else if (char === '\n' || char === '\r') { + if (field !== '' || row.length > 0) { + row.push(field.trim()); + rows.push(row); + row = []; + field = ''; + } + } else { + field += char; + } + } + + if (field !== '' || row.length > 0) { + row.push(field.trim()); + rows.push(row); + } + + const header = rows.shift(); + + if (!header) { + throw new Error('The input file is empty.'); + } + + return rows.map(cells => + Object.fromEntries(header.map((name, i) => [name, cells[i] ?? ''])) + ); +} + +function fingerprint(value) { + return createHash('sha1') + .update(JSON.stringify(value)) + .digest('hex') + .slice(0, 12); +} + +// The manifest makes the script resumable. Work already done is skipped, and +// media already generated is reused, so a second run costs nothing. +async function readManifest(path) { + try { + return JSON.parse(await readFile(path, 'utf8')); + } catch { + return { jobs: {}, cache: {} }; + } +} + +async function writeManifest(path, manifest) { + await writeFile(path, `${JSON.stringify(manifest, null, 2)}\n`); +} + +async function uploadToShotstack(key, body, contentType, label) { + const signed = await request( + `${INGEST_URL}/upload`, + { method: 'POST', headers: { 'x-api-key': key } }, + 'Shotstack Ingest' + ); + const { data } = await signed.json(); + await request( + data.attributes.url, + { method: 'PUT', headers: { 'Content-Type': contentType }, body }, + 'Shotstack Ingest' + ); + const started = Date.now(); + + while (Date.now() - started < MAX_WAIT_MS) { + const poll = await request( + `${INGEST_URL}/sources/${data.id}`, + { headers: { 'x-api-key': key } }, + 'Shotstack Ingest' + ); + const source = (await poll.json()).data.attributes; + + if (source.status === 'ready') { + return source.source; + } + + if (source.status === 'failed') { + throw new Error(`Shotstack could not read the ${label}.`); + } + + await delay(POLL_INTERVAL_MS); + } + + throw new Error(`Shotstack did not finish reading the ${label} in time.`); +} + +async function renderEdit(key, edit, label) { + const submitted = await request( + `${EDIT_URL}/render`, + { + method: 'POST', + headers: { 'x-api-key': key, 'Content-Type': 'application/json' }, + body: JSON.stringify(edit) + }, + 'Shotstack' + ); + const renderId = (await submitted.json()).response.id; + const started = Date.now(); + + while (Date.now() - started < MAX_WAIT_MS) { + await delay(POLL_INTERVAL_MS); + const poll = await request( + `${EDIT_URL}/render/${renderId}`, + { headers: { 'x-api-key': key } }, + 'Shotstack' + ); + const status = (await poll.json()).response; + + if (status.status === 'done') { + return status.url; + } + + if (status.status === 'failed') { + throw new Error( + `The ${label} render failed: ${status.error ?? 'no reason given'}` + ); + } + } + + throw new Error(`The ${label} render did not finish in time.`); +} + +async function download(url, path) { + const response = await request(url, {}, 'Shotstack CDN'); + await writeFile(path, Buffer.from(await response.arrayBuffer())); +} + +function mergeFields(values) { + return Object.entries(values).map(([find, replace]) => ({ + find, + replace: String(replace) + })); +} + +const RIME_URL = 'https://users.rime.ai/v1/rime-tts'; +const RIME_MODEL = process.env.RIME_MODEL || 'arcana'; + +const shotstackKey = requireEnv( + 'SHOTSTACK_API_KEY', + 'Get it from https://dashboard.shotstack.io.' +); +const rimeKey = requireEnv('RIME_API_KEY', 'Get it from https://rime.ai.'); +const sourceVideo = requireEnv( + 'SOURCE_VIDEO', + 'It must be a public link to the video you want to dub.' +); + +async function generateVoiceover(text, speaker) { + const response = await request( + RIME_URL, + { + method: 'POST', + headers: { + Authorization: `Bearer ${rimeKey}`, + Accept: 'audio/mpeg', + 'Content-Type': 'application/json' + }, + body: JSON.stringify({ speaker, text, modelId: RIME_MODEL }) + }, + 'Rime' + ); + + return Buffer.from(await response.arrayBuffer()); +} + +async function main() { + const rows = parseCsv(await readFile('translations.csv', 'utf8')); + + if (rows.length === 0) { + throw new Error('translations.csv has a header but no rows.'); + } + + await mkdir('output', { recursive: true }); + const manifestPath = 'output/manifest.json'; + const manifest = await readManifest(manifestPath); + const template = JSON.parse(await readFile('template.json', 'utf8')); + + for (const row of rows) { + if (manifest.jobs[row.locale]?.status === 'done') { + console.log(`${row.locale}: already rendered, skipping`); + continue; + } + + const cacheKey = fingerprint([row.text, row.speaker, RIME_MODEL]); + let dubUrl = manifest.cache[cacheKey]; + + if (dubUrl) { + console.log(`${row.locale}: reusing the voice-over from an earlier run`); + } else { + console.log(`${row.locale}: generating the voice-over with Rime`); + const audio = await generateVoiceover(row.text, row.speaker); + dubUrl = await uploadToShotstack( + shotstackKey, + audio, + 'audio/mpeg', + 'voice-over' + ); + manifest.cache[cacheKey] = dubUrl; + await writeManifest(manifestPath, manifest); + } + + const edit = structuredClone(template); + edit.merge = mergeFields({ + DUB_URL: dubUrl, + SOURCE_VIDEO: sourceVideo, + LOCALE_LABEL: row.label || row.locale, + ACCENT: process.env.ACCENT || '#0CB5B2' + }); + + console.log(`${row.locale}: rendering`); + const url = await renderEdit(shotstackKey, edit, row.locale); + const file = `output/${row.locale}.mp4`; + await download(url, file); + manifest.jobs[row.locale] = { + status: 'done', + file, + url, + speaker: row.speaker, + renderedAt: new Date().toISOString() + }; + await writeManifest(manifestPath, manifest); + console.log(`${row.locale}: saved to ${file}`); + } + + console.log('\nDone. One dubbed video per locale in output/.'); +} + +try { + await main(); +} catch (error) { + console.error(error instanceof Error ? error.message : error); + process.exit(1); +} diff --git a/examples/rime-multilingual-dub/template.json b/examples/rime-multilingual-dub/template.json new file mode 100644 index 0000000..72b2af8 --- /dev/null +++ b/examples/rime-multilingual-dub/template.json @@ -0,0 +1,116 @@ +{ + "timeline": { + "fonts": [ + { + "src": "https://fonts.gstatic.com/s/inter/v20/UcCo3FwrK3iLTfvlaQc78lA2.ttf" + } + ], + "background": "#000000", + "tracks": [ + { + "clips": [ + { + "asset": { + "type": "rich-caption", + "src": "alias://dub", + "font": { + "family": "UcCo3FwrK3iLTfvlaQc78lA2", + "size": 40, + "color": "#FFFFFF", + "weight": "700" + }, + "background": { + "color": "#000000", + "opacity": 0.55, + "borderRadius": 8, + "padding": 12 + }, + "align": { + "horizontal": "center", + "vertical": "bottom" + } + }, + "start": 0, + "length": "alias://dub", + "fit": "none", + "width": 1000, + "height": 320, + "position": "bottom", + "offset": { + "x": 0, + "y": 0.08 + } + } + ] + }, + { + "clips": [ + { + "asset": { + "type": "rich-text", + "text": "{{LOCALE_LABEL}}", + "font": { + "family": "UcCo3FwrK3iLTfvlaQc78lA2", + "size": 28, + "color": "#0B0B0B", + "weight": "700" + }, + "background": { + "color": "{{ACCENT}}", + "opacity": 1, + "borderRadius": 4, + "padding": 10 + }, + "align": { + "horizontal": "center", + "vertical": "middle" + } + }, + "start": 0.5, + "length": 3, + "position": "topRight", + "offset": { + "x": -0.04, + "y": -0.05 + }, + "transition": { + "in": "fade", + "out": "fadeFast" + } + } + ] + }, + { + "clips": [ + { + "asset": { + "type": "audio", + "src": "{{DUB_URL}}" + }, + "alias": "dub", + "start": 0, + "length": "alias://source" + } + ] + }, + { + "clips": [ + { + "asset": { + "type": "video", + "src": "{{SOURCE_VIDEO}}", + "volume": 0 + }, + "alias": "source", + "start": 0, + "length": "auto" + } + ] + } + ] + }, + "output": { + "format": "mp4", + "resolution": "1080" + } +} diff --git a/examples/rime-multilingual-dub/translations.csv b/examples/rime-multilingual-dub/translations.csv new file mode 100644 index 0000000..b931ee9 --- /dev/null +++ b/examples/rime-multilingual-dub/translations.csv @@ -0,0 +1,4 @@ +locale,label,speaker,text +en-US,English,astra,"Every render starts as a script. This is what the finished version sounds like." +es-ES,Espanol,celeste,"Cada video empieza como un guion. Asi suena la version terminada." +fr-FR,Francais,marcel,"Chaque video commence par un texte. Voici la version terminee."