#!/usr/bin/env python3
"""Extract the original captioned product-demo kit. No packages, downloads or commands run.
Usage: python3 captioned-product-demo-workflow.txt NEW_DIRECTORY
Guide: https://www.murmurtts.com/blog/resources/captioned-product-demo-local-voiceover-mac
"""
import json
import sys
from pathlib import Path
FILES = json.loads('{"package.json": "{\\n \\"name\\": \\"captioned-local-demo-prototype\\",\\n \\"version\\": \\"1.0.0\\",\\n \\"private\\": true,\\n \\"dependencies\\": {\\n \\"@remotion/cli\\": \\"4.0.534\\",\\n \\"@remotion/media\\": \\"4.0.534\\",\\n \\"remotion\\": \\"4.0.534\\",\\n \\"react\\": \\"19.3.0\\",\\n \\"react-dom\\": \\"19.3.0\\"\\n },\\n \\"type\\": \\"module\\"\\n}\\n", "prepare.mjs": "import fs from \'node:fs\';\\nimport path from \'node:path\';\\nimport {createHash} from \'node:crypto\';\\nimport {execFileSync} from \'node:child_process\';\\nimport {pathToFileURL} from \'node:url\';\\nconst FPS = 30;\\nexport const digest = file => createHash(\'sha256\').update(fs.readFileSync(file)).digest(\'hex\');\\nfunction regular(file) { const real = fs.realpathSync(file); if (!fs.statSync(real).isFile()) throw new Error(\'Input must be a local regular file.\'); return real; }\\nfunction duration(file, type) {\\n const data = JSON.parse(execFileSync(\'ffprobe\', [\'-v\', \'error\', \'-select_streams\', type + \':0\', \'-show_entries\', \'stream=codec_type:format=duration\', \'-of\', \'json\', file], {encoding: \'utf8\'}));\\n const value = Number(data.format?.duration);\\n if (!data.streams?.length || !Number.isFinite(value) || value <= 0) throw new Error(\'Missing stream or invalid duration: \' + file);\\n return value;\\n}\\nexport function validateInputs(videoPath, audioPath, captionsPath, fixture = false) {\\n const video = regular(videoPath), audio = regular(audioPath), captionFile = regular(captionsPath);\\n const audioDuration = duration(audio, \'a\'), videoDuration = duration(video, \'v\');\\n const bundle = JSON.parse(fs.readFileSync(captionFile, \'utf8\'));\\n if (bundle.narrationSha256 !== digest(audio) || !Number.isFinite(bundle.narrationDurationSeconds) || Math.abs(bundle.narrationDurationSeconds - audioDuration) > 0.000001) throw new Error(\'Caption review is stale: narration hash or duration changed. Re-time and review cues; do not stretch them automatically.\');\\n const fixtureOnly = fixture && bundle.review?.state === \'fixture-only\';\\n if (!fixtureOnly && (bundle.review?.state !== \'approved\' || typeof bundle.review?.reviewer !== \'string\' || !bundle.review.reviewer.trim())) throw new Error(\'Explicit caption review is required. Fixture-only cues need --fixture and are watermarked.\');\\n if (!Array.isArray(bundle.cues) || !bundle.cues.length) throw new Error(\'Supply explicit caption cues.\');\\n let previousEnd = 0;\\n const cues = bundle.cues.map((cue, index) => {\\n const {startSeconds, endSeconds, text} = cue;\\n if (!Number.isFinite(startSeconds) || !Number.isFinite(endSeconds) || startSeconds < 0 || endSeconds <= startSeconds || endSeconds > audioDuration + 0.000001 || startSeconds < previousEnd) throw new Error(\'Invalid, overlapping or out-of-range cue \' + index);\\n if (typeof text !== \'string\' || !text.trim() || text.length > 140 || text.split(\'\\\\n\').length > 2) throw new Error(\'Cue text must contain 1-140 characters on at most two lines.\');\\n const startFrame = Math.ceil(startSeconds * FPS), endFrame = Math.ceil(endSeconds * FPS);\\n if (startFrame >= endFrame) throw new Error(\'Cue is shorter than its visible frame interval.\');\\n previousEnd = endSeconds;\\n return {text, startFrame, endFrame};\\n });\\n const durationInFrames = Math.ceil(audioDuration * FPS);\\n if (videoDuration + 0.000001 < durationInFrames / FPS) throw new Error(\'Screen clip is shorter than the rounded narration timeline. Choose another clip; no silent freeze/loop.\');\\n return {video, audio, captionFile, videoDuration, audioDuration, durationInFrames, cues, fixtureOnly};\\n}\\nexport function prepare(videoPath, audioPath, captionsPath, outputDirectory, fixture = false) {\\n const checked = validateInputs(videoPath, audioPath, captionsPath, fixture);\\n const target = path.resolve(outputDirectory);\\n fs.mkdirSync(target); // Existing paths fail, before any copy or props write.\\n const publicDir = path.join(target, \'public\'); fs.mkdirSync(publicDir);\\n fs.copyFileSync(checked.video, path.join(publicDir, \'screen.mp4\'), fs.constants.COPYFILE_EXCL);\\n fs.copyFileSync(checked.audio, path.join(publicDir, \'narration.wav\'), fs.constants.COPYFILE_EXCL);\\n fs.copyFileSync(checked.captionFile, path.join(target, \'captions.json\'), fs.constants.COPYFILE_EXCL);\\n // Validate the copied snapshot: originals can change while the copies are made.\\n const snapshot = validateInputs(path.join(publicDir, \'screen.mp4\'), path.join(publicDir, \'narration.wav\'), path.join(target, \'captions.json\'), fixture);\\n const props = {videoFile: \'screen.mp4\', audioFile: \'narration.wav\', durationSeconds: snapshot.audioDuration, fps: FPS, cues: snapshot.cues, prototypeOnly: snapshot.fixtureOnly};\\n const propsBytes = JSON.stringify(props, null, 2) + \'\\\\n\';\\n const manifest = {videoDuration: snapshot.videoDuration, durationInFrames: snapshot.durationInFrames, sha256: {video: digest(snapshot.video), audio: digest(snapshot.audio), captions: digest(snapshot.captionFile), props: createHash(\'sha256\').update(propsBytes).digest(\'hex\')}};\\n fs.writeFileSync(path.join(target, \'props.json\'), propsBytes, {flag: \'wx\'});\\n fs.writeFileSync(path.join(target, \'manifest.json\'), JSON.stringify(manifest, null, 2) + \'\\\\n\', {flag: \'wx\'});\\n return {directory: target, ...manifest};\\n}\\nif (process.argv[1] && import.meta.url === pathToFileURL(fs.realpathSync(process.argv[1])).href) {\\n try {\\n const [video, audio, captions, directory, flag] = process.argv.slice(2);\\n if (!directory || (flag && flag !== \'--fixture\')) throw new Error(\'Usage: node prepare.mjs video.mp4 narration.wav captions.json NEW_RUN_DIR [--fixture]\');\\n console.log(JSON.stringify(prepare(video, audio, captions, directory, flag === \'--fixture\'), null, 2));\\n } catch (failure) { console.error(failure.message); process.exitCode = 1; }\\n}\\n", "render.mjs": "import fs from \'node:fs\';\\nimport path from \'node:path\';\\nimport {fileURLToPath} from \'node:url\';\\nimport {execFileSync} from \'node:child_process\';\\nimport {digest} from \'./prepare.mjs\';\\nconst root = path.dirname(fileURLToPath(import.meta.url));\\nconst [runArg, outputArg] = process.argv.slice(2);\\nlet stage, installedOutput;\\ntry {\\n if (!runArg || !outputArg) throw new Error(\'Usage: node render.mjs RUN_DIR NEW_OUTPUT.mp4\');\\n const run = path.resolve(runArg), output = path.resolve(outputArg), reportPath = output + \'.json\';\\n for (const target of [output, reportPath]) { try { fs.lstatSync(target); throw new Error(\'Output or report already exists: \' + target); } catch (error) { if (error.code !== \'ENOENT\') throw error; } }\\n const browser = process.env.REMOTION_BROWSER_EXECUTABLE;\\n if (!browser || !fs.existsSync(browser)) throw new Error(\'Set REMOTION_BROWSER_EXECUTABLE to an existing cached Chrome Headless Shell; no browser download is performed.\');\\n const manifest = JSON.parse(fs.readFileSync(path.join(run, \'manifest.json\'), \'utf8\'));\\n if (digest(path.join(run, \'props.json\')) !== manifest.sha256.props || digest(path.join(run, \'captions.json\')) !== manifest.sha256.captions) throw new Error(\'Prepared props or captions changed; prepare and review again.\');\\n if (digest(path.join(run, \'public/screen.mp4\')) !== manifest.sha256.video || digest(path.join(run, \'public/narration.wav\')) !== manifest.sha256.audio) throw new Error(\'Prepared media changed; prepare and review again.\');\\n stage = fs.mkdtempSync(path.join(path.dirname(output), \'.captioned-demo-\'));\\n const movie = path.join(stage, \'candidate.mp4\');\\n const log = fs.openSync(path.join(stage, \'render.log\'), \'wx\');\\n try { execFileSync(path.join(root, \'node_modules/.bin/remotion\'), [\'render\', \'src/index.tsx\', \'CaptionedDemo\', movie, \'--props=\' + path.join(run, \'props.json\'), \'--public-dir=\' + path.join(run, \'public\'), \'--browser-executable=\' + browser, \'--codec=h264\', \'--audio-codec=aac\', \'--pixel-format=yuv420p\', \'--concurrency=2\', \'--overwrite=false\'], {cwd: root, stdio: [\'ignore\', log, log]}); } finally { fs.closeSync(log); }\\n execFileSync(\'ffmpeg\', [\'-hide_banner\', \'-loglevel\', \'error\', \'-i\', movie, \'-f\', \'null\', \'-\'], {stdio: [\'ignore\', \'pipe\', \'pipe\']});\\n const probe = JSON.parse(execFileSync(\'ffprobe\', [\'-v\', \'error\', \'-count_frames\', \'-show_entries\', \'stream=codec_name,codec_type,width,height,avg_frame_rate,nb_read_frames,duration,sample_rate,channels:format=duration,size\', \'-of\', \'json\', movie], {encoding: \'utf8\'}));\\n const video = probe.streams.find(item => item.codec_type === \'video\');\\n if (Number(video?.nb_read_frames) !== manifest.durationInFrames || !probe.streams.some(item => item.codec_type === \'audio\')) throw new Error(\'Unexpected output frame count or missing audio.\');\\n const report = {fullDecode: \'passed\', probe, outputSha256: digest(movie), renderLog: path.join(stage, \'render.log\')};\\n fs.linkSync(movie, output); // Exclusive install: a late collision cannot overwrite.\\n installedOutput = output;\\n fs.writeFileSync(reportPath, JSON.stringify(report, null, 2) + \'\\\\n\', {flag: \'wx\'});\\n fs.unlinkSync(movie); // Keep the render log for review.\\n console.log(JSON.stringify({output, report: reportPath, ...report}, null, 2));\\n} catch (failure) { console.error(\'Stopped: \' + failure.message + (stage ? \'\\\\nRender stage: \' + stage : \'\') + (installedOutput ? \'\\\\nVideo installed at \' + installedOutput + \'; inspect the video and sidecar before retrying.\' : \'\')); process.exitCode = 1; }\\n", "src/index.tsx": "import React from \'react\';\\nimport {Audio, Video} from \'@remotion/media\';\\nimport {AbsoluteFill, Composition, registerRoot, staticFile, useCurrentFrame} from \'remotion\';\\n\\ntype Cue = {text: string; startFrame: number; endFrame: number};\\ntype Props = {audioFile: string; videoFile: string; durationSeconds: number; fps: number; cues: Cue[]; prototypeOnly: boolean};\\nconst Demo = ({audioFile, videoFile, cues, prototypeOnly}: Props) => {\\n const frame = useCurrentFrame();\\n const cue = cues.find(item => frame >= item.startFrame && frame < item.endFrame);\\n return \\n ;\\n};\\nconst Root = () => {\\n if (!Number.isFinite(props.durationSeconds) || props.durationSeconds <= 0 || props.fps !== 30) throw new Error(\'Prepare valid media metadata first.\');\\n return {durationInFrames: Math.ceil(props.durationSeconds * props.fps)};\\n }} />;\\nregisterRoot(Root);\\n", "caption-review.example.json": "{\\n \\"narrationSha256\\": \\"REPLACE_WITH_SHA256_OF_REVIEWED_NARRATION\\",\\n \\"narrationDurationSeconds\\": 7.12,\\n \\"review\\": {\\n \\"state\\": \\"pending\\",\\n \\"reviewer\\": \\"\\",\\n \\"notes\\": \\"Listen to the exact narration and verify every start/end time and word before approving.\\"\\n },\\n \\"cues\\": [\\n {\\n \\"startSeconds\\": 0.0,\\n \\"endSeconds\\": 2.4,\\n \\"text\\": \\"Replace with reviewed transcript text.\\"\\n }\\n ]\\n}\\n", ".gitignore": "/node_modules/\\n/run-*/\\n/.captioned-demo-*/\\n/*.mp4\\n/*.mp4.json\\n/*.mov\\n/*.wav\\n/*.m4a\\n/captions.json\\n", "START_HERE.md": "# Captioned product demo: local source kit\\n\\nThis kit combines your screen clip, final WAV and explicit JSON captions. It does not record, generate speech, transcribe, infer timing or upload a movie. All source files are readable. The extractor only creates a new folder; it refuses an existing destination and runs no installers.\\n\\n## Set up\\n\\nSeparately install Python 3 (for extraction), Node/npm, FFmpeg/ffprobe and Murmur with local automation enabled. The worked render used Node 22.23.1, FFmpeg 9.0.2 and the pinned Remotion 4.0.534/React 19.3.0 packages. Initial dependency, speech-model, ASR-model and browser downloads may require network access. A hosted assistant may receive your prompt even when media processing stays local. Review Remotion\'s terms for your organization: https://www.remotion.dev/docs/license/pricing\\n\\nFrom this new kit folder:\\n\\n npm install --registry=https://registry.npmjs.org --no-audit --no-fund\\n npx remotion browser ensure --chrome-mode=headless-shell\\n\\nKeep the generated package-lock.json; later use npm ci. Set REMOTION_BROWSER_EXECUTABLE to the actual Chrome Headless Shell executable reported/installed on your machine. Do not paste a path from another person\'s machine. The rendering helper requires that existing executable and never downloads a browser itself.\\n\\n export REMOTION_BROWSER_EXECUTABLE=\'/absolute/path/to/chrome-headless-shell\'\\n\\n## Prepare narration and footage\\n\\nUse a fresh script/output name. Select a model and voice that your own catalog reports ready:\\n\\n murmur status --json\\n murmur models --json\\n murmur voices --model qwen3-base --json\\n murmur generate --input script.txt --model qwen3-base --voice Ryan \\\\\\n --language EN-US --speed 1.0 --output narration.wav --json\\n\\nThe worked sentence is: Open the model selector to see which speech engine is active.\\nIt describes only opening a selector. It does not imply a model change or generation on screen. The guide records the measured development-helper result; generation duration and installed models vary. Wait for the job to succeed and inspect its real output. Murmur is a paid product; local generation is bounded by your hardware and the selected model\'s terms.\\n\\nRecord or use your own reviewed product footage. Exclude customer content, notifications and unrelated history before capture. Choose the actual range and crop for your recording. Here is the tested cut for our 1920x1206 source; it is not a universal crop:\\n\\n ffmpeg -hide_banner -loglevel error -ss 6 -i screen-source.mp4 -t 5 \\\\\\n -vf \'crop=960:540:960:0,scale=1280:720,fps=30\' -an \\\\\\n -c:v libx264 -crf 18 -pix_fmt yuv420p -n screen.mp4\\n\\nKeep originals. This five-second source covers the short narration at normal speed. The kit mutes source audio, fits the supplied clip without additional cropping, and ends at ceil(narrationSeconds*30)/30. Short clips are refused; it does not loop or freeze frames.\\n\\n## Make and review captions\\n\\nCopy caption-review.example.json to a fresh captions.json. Start with a local transcript or manually timed words. The optional separate whisper.cpp tool can supply a draft, but its words and timestamps can be wrong. It is not a dependency of this kit, and ASR output does not directly match this kit\'s JSON schema. Convert chosen phrase ranges explicitly.\\n\\n shasum -a 256 narration.wav\\n ffprobe -v error -show_entries format=duration \\\\\\n -of default=noprint_wrappers=1:nokey=1 narration.wav\\n\\nEnter that exact hash and duration. Set startSeconds/endSeconds and literal text for each cue. Use finite ordered non-overlapping times within the audio duration, at most 140 characters and two explicit lines. Gaps are allowed. A cue starts on ceil(startSeconds*30), inclusive, and stops on ceil(endSeconds*30), exclusive.\\n\\nListen to the final take, compare words, and inspect caption starts/ends and the completed movie before approving your promotional asset. Set review.state to approved only for your explicitly documented review and name its reviewer. Describe the method and limits in review.notes. A supplied flag is not independent proof that a human listened, that every ASR timestamp is right, or that accessibility requirements are met. The worked guide distinguishes automated transcript matching and visual inspection from human listening.\\n\\nAny narration edit changes the hash; rebuild timing and review instead of stretching old cues. Fixture-only labels require --fixture and display a conspicuous watermark. Do not use fixture labels as speech captions.\\n\\n## Prepare and render\\n\\n node prepare.mjs screen.mp4 narration.wav captions.json run-01\\n node render.mjs run-01 captioned-demo.mp4\\n\\nPreparation creates a new folder, copies inputs, revalidates the copied snapshot, and binds media, caption and props hashes. An existing run is refused. Render rechecks hashes and renders to a new stage before decoding the whole candidate and checking frames/audio. It installs the movie and JSON sidecar exclusively, refusing existing names. A late failure can leave an installed movie; read the error and inspect it before a new attempt.\\n\\nOutput is 1280x720 at 30 fps, H.264/yuv420p with AAC. Captions are burned into the image, not a selectable subtitle track. AAC/container duration can slightly exceed the video timeline. Inspect every cue boundary, small-screen readability, UI coverage, audio pronunciation and destination loudness before release. Successful decoding is only a technical check.\\n\\nThe public/ folder inside a prepared run is a renderer asset directory, not a privacy boundary. Do not publish that folder or raw recordings without review. No external upload, cloud-rendering service, assistant-model execution or connectivity-disabled test is performed by these scripts.\\n\\nReferences: https://www.remotion.dev/docs/media/video ; https://www.remotion.dev/docs/captions/ ; https://github.com/ggml-org/whisper.cpp ; https://www.w3.org/WAI/media/av/captions/\\n\\n## Measured worked example (October 9, 2026)\\n\\nOne development-helper 1.0.18 job with Qwen3-TTS Base/Ryan produced a 3.68-second mono 24 kHz PCM take. Unprompted local whisper.cpp 1.9.4 base.en matched all 11 normalized script words. We used two estimated phrase cues: 0.19–1.33 seconds and 1.33–3.60 seconds. These are specific to that take, not universal timings. At 30 fps they occupy frames 6–39 and 40–107 inclusive. We inspected each boundary in the final movie. The example has 111 frames, 3.700 seconds of video, 3.712 seconds container duration and 48 kHz stereo AAC. Full decoding passed. Final MP4 SHA-256: 03f1830428765ca986cbd8957664235d6acb064b3789e3015874f3f5ac75fa06.\\n\\nThe supplied review record for that technical example explicitly names automated transcript comparison and visual engineering review. It is not human listening, guaranteed alignment, a released-binary parity test, a vertical-phone readability test, or accessibility certification. The recording’s exact app build is unknown. Menu labels reflect the recording; check your current catalog. The article includes a descriptive account of the visible action.\\n"}')
def main():
if len(sys.argv) != 2:
raise SystemExit(__doc__)
destination = Path(sys.argv[1]).absolute()
destination.mkdir() # Refuse existing files, folders and symlinks.
for relative, content in FILES.items():
target = destination / relative
target.parent.mkdir(parents=True, exist_ok=True)
with target.open('x', encoding='utf-8', newline='\n') as handle:
handle.write(content)
print('Created ' + str(destination) + '; read START_HERE.md before installing dependencies.')
if __name__ == '__main__':
main()