فهرست منبع

fix(movie): make narration acceptance govern assembly

Repair Task 1 from the 2026-09-11 movie committee plan. Narration now atomically withdraws acceptance before it mutates accepted bytes, stages failed takes as retained evidence, and publishes a manifest entry only after transcript gates and duration measurement. It preflights ffprobe, preserves bounded retries and cache identity, and distinguishes unsupported token comparison from cache identity.

Assembly now validates accepted manifest narration before it starts encoding, ignores generated narration for movie scenes, selects manifest WAV paths, maps source or synthetic movie audio explicitly, fits wide movies within the requested inner rectangle, and escapes only literal directory percent signs in sequence paths. The focused regression suite mocks every media boundary; real-media fixture declarations are updated but not executed. Prompt constraints prohibit real media generation, probing, inspection, synthesis, ASR, browser, checker, or full media suites.
Drew Ritter 3 هفته پیش
والد
کامیت
5e02913144

+ 28 - 8
skills/proving-it-works-with-a-movie/scripts/assemble

@@ -35,7 +35,8 @@ from pathlib import Path
 
 import yaml
 from browser_tools import find_browser, render_card
-from media_paths import ffconcat_entry, stage_frames
+from media_paths import ffconcat_entry, sequence_pattern, stage_frames
+from narration_contract import accepted_narration
 
 CARD_HTML = """<!doctype html><meta charset="utf-8">
 <style>
@@ -69,6 +70,15 @@ def dur(path):
     return float(r.stdout.strip())
 
 
+def has_audio_stream(path):
+    result = run(["ffprobe", "-v", "error", "-show_streams", "-of", "json", str(path)])
+    try:
+        streams = json.loads(result.stdout).get("streams", [])
+    except json.JSONDecodeError:
+        die(f"ffprobe returned invalid stream data for {path}")
+    return any(stream.get("codec_type") == "audio" for stream in streams)
+
+
 def make_card(scene, png, w, h, browser):
     if not browser:
         die("a `card` scene needs a browser (Chrome/Chromium) to render text; "
@@ -108,6 +118,10 @@ def main():
     W, H = int(res.get("width", 1920)), int(res.get("height", 1080))
     FPS = int(doc.get("fps", 30))
     narration = args.narration or (base / "narration")
+    try:
+        accepted = accepted_narration(narration, doc["scenes"])
+    except ValueError as error:
+        die(str(error))
     work = args.work or (base / "segments")
     work.mkdir(parents=True, exist_ok=True)
     browser = find_browser(args.browser)
@@ -120,18 +134,24 @@ def main():
         sid = sc["id"]
         kind = sc.get("kind", "frames")
         seg = work / f"{sid}.mp4"
-        nar = narration / f"{sid}.wav"
-        nard = dur(nar) if nar.exists() else 0.0
+        nar = accepted.get(sid)
+        nard = dur(nar) if nar is not None else 0.0
 
         if kind == "movie":
             src = base / sc["src"]
             target = dur(src)
             inner_h = int(sc.get("height", int(H * 0.82)))
-            run(["ffmpeg", "-nostdin", "-y", "-v", "error", "-i", str(src),
-                 "-vf", f"scale=-2:{inner_h},pad={W}:{H}:(ow-iw)/2:(oh-ih)/2:"
+            source_audio = has_audio_stream(src)
+            ain = ([] if source_audio
+                   else ["-f", "lavfi", "-i", "anullsrc=r=44100:cl=stereo"])
+            audio_map = "0:a:0" if source_audio else "1:a:0"
+            run(["ffmpeg", "-nostdin", "-y", "-v", "error", "-i", str(src), *ain,
+                 "-vf", f"scale={W}:{inner_h}:force_original_aspect_ratio=decrease,"
+                        f"pad={W}:{H}:(ow-iw)/2:(oh-ih)/2:"
                         f"color=#101014,setsar=1",
                  "-af", f"volume={sc.get('gain_db', 0)}dB,apad",
                  "-r", str(FPS), "-t", f"{target:.3f}",
+                 "-map", "0:v:0", "-map", audio_map,
                  "-c:v", "libx264", "-preset", "medium", "-pix_fmt", "yuv420p",
                  "-c:a", "aac", "-ar", "44100", "-ac", "2", str(seg)])
         else:
@@ -156,7 +176,7 @@ def main():
                 vis = n / rate
                 target = max(nard, vis)
                 vin = ["-framerate", str(rate), "-start_number", "0",
-                       "-i", str(staged[0].parent / "frame-%08d.png")]
+                       "-i", sequence_pattern(staged[0].parent, "frame-%08d.png")]
                 # freeze the last frame when narration outlasts the action
                 vf = fit + f",tpad=stop_mode=clone:stop_duration={max(0.0, target - vis):.3f}"
             else:
@@ -173,7 +193,7 @@ def main():
                 vin = ["-loop", "1", "-i", str(img)]
                 vf = fit
 
-            ain = (["-i", str(nar)] if nar.exists()
+            ain = (["-i", str(nar)] if nar is not None
                    else ["-f", "lavfi", "-i", "anullsrc=r=44100:cl=stereo"])
             try:
                 run(["ffmpeg", "-nostdin", "-y", "-v", "error", *vin, *ain,
@@ -188,7 +208,7 @@ def main():
         actual = dur(seg)
         # only scenes that speak get a subtitle offset; a movie played as
         # itself carries its own subtitles already
-        if nar.exists() and kind != "movie":
+        if nar is not None and kind != "movie":
             offsets[sid] = round(clock, 3)
         clock += actual
         concat_lines.append(ffconcat_entry(seg))

+ 3 - 1
skills/proving-it-works-with-a-movie/scripts/check-movie

@@ -34,6 +34,7 @@ import sys
 from pathlib import Path
 
 from PIL import Image
+from media_paths import sequence_pattern
 
 THUMB_W = 320          # sampling width; the metric is a pixel fraction, so scale-free
 PIXEL_DELTA = 8        # per-pixel grey delta that counts as "this pixel moved"
@@ -63,7 +64,8 @@ def sample_picture(movie, workdir):
         old.unlink()
     out = subprocess.run(
         ["ffmpeg", "-nostdin", "-v", "error", "-i", str(movie),
-         "-vf", f"fps=1,scale={THUMB_W}:-1", "-f", "image2", str(frames / "s%05d.png")],
+         "-vf", f"fps=1,scale={THUMB_W}:-1", "-f", "image2",
+         sequence_pattern(frames, "s%05d.png")],
         capture_output=True, text=True, encoding="utf-8", errors="replace")
     if out.returncode != 0:
         die(f"frame sampling failed: {out.stderr.strip()[:200]}")

+ 5 - 0
skills/proving-it-works-with-a-movie/scripts/media_paths.py

@@ -22,3 +22,8 @@ def ffconcat_entry(path: Path) -> str:
     if "\n" in value or "\r" in value:
         raise ValueError("FFconcat paths cannot contain line breaks")
     return "file '" + value.replace("'", "'\\''") + "'\n"
+
+
+def sequence_pattern(directory: Path, filename_pattern: str) -> str:
+    """Keep FFmpeg's frame placeholder while escaping literal directory percent signs."""
+    return directory.as_posix().replace("%", "%%") + "/" + filename_pattern

+ 137 - 34
skills/proving-it-works-with-a-movie/scripts/narrate

@@ -25,11 +25,13 @@ import base64
 import difflib
 import json
 import os
-import re
+import shutil
 import subprocess
 import sys
 import tempfile
+import unicodedata
 import urllib.request
+import uuid
 import wave
 from pathlib import Path
 
@@ -59,8 +61,31 @@ def openai_key():
     return None
 
 
+SEGMENTATION_SCRIPTS = (
+    (0x3040, 0x30FF),  # Hiragana and Katakana
+    (0x3400, 0x4DBF),  # CJK Extension A
+    (0x4E00, 0x9FFF),  # CJK Unified Ideographs
+    (0x0E00, 0x0E7F),  # Thai
+)
+
+
 def norm(s):
-    return re.sub(r"[^a-z0-9 ]+", "", s.lower()).split()
+    """Return comparable words without changing narration/cache identity."""
+    words = []
+    for token in unicodedata.normalize("NFKC", s).casefold().split():
+        word = "".join(
+            char for char in token
+            if unicodedata.category(char)[0] in {"L", "N", "M"}
+        )
+        if word:
+            words.append(word)
+    return words
+
+
+def requires_segmentation(s):
+    return any(start <= ord(char) <= end
+               for char in unicodedata.normalize("NFKC", s)
+               for start, end in SEGMENTATION_SCRIPTS)
 
 
 ASR_SNIPPET = """
@@ -111,7 +136,13 @@ def structural_drift(text, heard):
     the voice skipped, or a preamble it invented. Returns
     (length_delta_fraction, longest_run_of_missing_added_or_changed_words).
     """
+    if requires_segmentation(text) or requires_segmentation(heard):
+        return None
     want, got = norm(text), norm(heard)
+    if not want:
+        return None
+    if not got:
+        return 1.0, len(want)
     delta = abs(len(got) - len(want)) / max(1, len(want))
     ops = difflib.SequenceMatcher(a=want, b=got).get_opcodes()
     worst = max((max(i2 - i1, j2 - j1) for tag, i1, i2, j1, j2 in ops if tag != "equal"),
@@ -173,6 +204,26 @@ def duration(path):
     return round(float(out.stdout.strip()), 3)
 
 
+def accepted_wav(outdir, entry):
+    wav = entry.get("wav") if isinstance(entry, dict) else None
+    if not isinstance(wav, str) or not wav:
+        return None
+    candidate = Path(wav)
+    if candidate.is_absolute():
+        return None
+    return outdir / candidate
+
+
+def write_manifest(manifest_path, entries):
+    with tempfile.NamedTemporaryFile(dir=manifest_path.parent, delete=False) as handle:
+        temporary = Path(handle.name)
+    try:
+        temporary.write_text(json.dumps(entries, indent=2), encoding="utf-8")
+        temporary.replace(manifest_path)
+    finally:
+        temporary.unlink(missing_ok=True)
+
+
 def main():
     for stream in (sys.stdout, sys.stderr):
         if hasattr(stream, "reconfigure"):
@@ -180,7 +231,11 @@ def main():
     if len(sys.argv) == 4 and sys.argv[1] == "--drift-check":
         script = Path(sys.argv[2]).read_text(encoding="utf-8-sig")
         heard = Path(sys.argv[3]).read_text(encoding="utf-8-sig")
-        delta, worst = structural_drift(script, heard)
+        drift = structural_drift(script, heard)
+        if drift is None:
+            print("comparison unavailable -> MISMATCH")
+            return 1
+        delta, worst = drift
         bad = delta > 0.15 or worst >= 4
         print(f"length change {delta:.0%}, worst run {worst} -> "
               f"{'MISMATCH' if bad else 'ok'}")
@@ -204,6 +259,8 @@ def main():
     scenes = [s for s in doc.get("scenes", []) if (s.get("narration") or "").strip()]
     if not scenes:
         die("no scenes with narration")
+    if not shutil.which("ffprobe"):
+        die("ffprobe not on PATH (narration durations cannot be measured)")
 
     key = openai_key()
     engine = args.engine
@@ -234,32 +291,63 @@ def main():
                      json.loads(prior_path.read_text(encoding="utf-8-sig"))}
         except Exception:  # noqa: BLE001 - a corrupt manifest just means no cache
             prior = {}
-    manifest, failures = [], []
+    current_ids = {sc["id"] for sc in scenes}
+    manifest = {sid: entry for sid, entry in prior.items() if sid in current_ids}
+    write_manifest(prior_path, list(manifest.values()))
+    failures = []
+
+    def withdraw(sid):
+        if sid in manifest:
+            del manifest[sid]
+        write_manifest(prior_path, list(manifest.values()))
+
+    def accept(entry):
+        manifest[entry["id"]] = entry
+        write_manifest(prior_path, list(manifest.values()))
 
     for sc in scenes:
         sid = sc["id"]
         text = " ".join((sc["narration"] or "").split())
-        wav = args.outdir / f"{sid}.wav"
         previous = prior.get(sid, {})
-        cached = (wav.exists() and not args.force
+        previous_wav = accepted_wav(args.outdir, previous)
+        cached = (previous_wav is not None and previous_wav.exists() and not args.force
                   and previous.get("text") == text
                   and previous.get("synthesis") == synthesis)
         if cached:
             print(f"{sid}: cached")
-        elif wav.exists() and not args.force and sid in prior:
+        elif previous_wav is not None and previous_wav.exists() and not args.force and sid in prior:
             print(f"{sid}: text or synthesis settings changed - redoing")
+        if engine == "openai-chat" and structural_drift(text, text) is None:
+            print(f"{sid}: chat transcript comparison unavailable", file=sys.stderr)
+            withdraw(sid)
+            failures.append(sid)
+            continue
+        if not cached:
+            withdraw(sid)
+        accepted = False
         for attempt in ((1,) if cached else (1, 2)):
             claimed = None
+            candidate = previous_wav if cached else args.outdir / (
+                f".{sid}.attempt-{uuid.uuid4().hex}.wav"
+            )
             if not cached:
-                if engine == "openai":
-                    claimed = say_openai(key, text, wav, voice)
-                elif engine == "openai-chat":
-                    claimed = say_openai_chat(key, text, wav, voice)
-                else:
-                    claimed = say_piper(text, wav, voice)
+                try:
+                    if engine == "openai":
+                        claimed = say_openai(key, text, candidate, voice)
+                    elif engine == "openai-chat":
+                        claimed = say_openai_chat(key, text, candidate, voice)
+                    else:
+                        claimed = say_piper(text, candidate, voice)
+                except Exception as error:  # rejected bytes remain as evidence
+                    print(f"{sid}: synthesis failed (attempt {attempt}: {error})", file=sys.stderr)
+                    continue
 
             # Preserve the engine transcript gate and the ASR drift thresholds.
             if claimed is not None:
+                drift_result = structural_drift(text, claimed)
+                if drift_result is None:
+                    print(f"{sid}: chat transcript comparison unavailable", file=sys.stderr)
+                    break
                 want, got = norm(text), norm(claimed)
                 drift = abs(len(want) - len(got)) + sum(
                     1 for a, b in zip(want, got) if a != b)
@@ -267,36 +355,51 @@ def main():
                     print(f"{sid}: engine ad-libbed (attempt {attempt}, drift {drift})")
                     continue
             if verify:
-                heard = transcribe_local(wav, args.asr_model)
+                heard = transcribe_local(candidate, args.asr_model)
                 if heard is None:
                     if args.verify == "on":
                         print(f"{sid}: required verification unavailable", file=sys.stderr)
-                        failures.append(sid)
+                        break
                     else:
                         print(f"{sid}: verification unavailable (no local ASR)")
-                    break
-                delta, worst = structural_drift(text, heard)
-                if delta > 0.15 or worst >= 4:
-                    print(f"{sid}: what came out does not match the script "
-                          f"(attempt {attempt}: {delta:.0%} length change, "
-                          f"{worst} words in a row wrong)")
-                    print(f"       heard: {heard[:120]}")
-                    continue
-                print(f"{sid}: ok (verified by ear: {delta:.0%} length "
-                      f"change, worst run {worst})")
+                else:
+                    drift_result = structural_drift(text, heard)
+                    if drift_result is None:
+                        if args.verify == "on":
+                            print(f"{sid}: required verification unavailable", file=sys.stderr)
+                            break
+                        print(f"{sid}: verification unavailable (unsupported script)")
+                    else:
+                        delta, worst = drift_result
+                        if delta > 0.15 or worst >= 4:
+                            print(f"{sid}: what came out does not match the script "
+                                  f"(attempt {attempt}: {delta:.0%} length change, "
+                                  f"{worst} words in a row wrong)")
+                            print(f"       heard: {heard[:120]}")
+                            continue
+                        print(f"{sid}: ok (verified by ear: {delta:.0%} length "
+                              f"change, worst run {worst})")
+            if not verify:
+                print(f"{sid}: ok")
+            if not cached:
+                candidate.replace(args.outdir / f"{sid}.wav")
+                candidate = args.outdir / f"{sid}.wav"
+            try:
+                clip_duration = duration(candidate)
+            except Exception as error:  # unaccepted bytes remain as evidence
+                print(f"{sid}: duration failed ({error})", file=sys.stderr)
                 break
-            print(f"{sid}: ok")
+            accept({"id": sid, "text": text, "wav": candidate.name,
+                    "duration": clip_duration, "synthesis": synthesis})
+            accepted = True
             break
-        else:
+        if not accepted:
+            withdraw(sid)
             failures.append(sid)
-        if sid in failures:
-            continue
-        manifest.append({"id": sid, "text": text, "wav": wav.name,
-                         "duration": duration(wav), "synthesis": synthesis})
 
-    (args.outdir / "manifest.json").write_text(json.dumps(manifest, indent=2), encoding="utf-8")
-    total = sum(m["duration"] for m in manifest)
-    print(f"\n{len(manifest)} clips, {total:.1f}s total -> {args.outdir}/manifest.json")
+    entries = list(manifest.values())
+    total = sum(m["duration"] for m in entries)
+    print(f"\n{len(entries)} clips, {total:.1f}s total -> {args.outdir}/manifest.json")
     if engine == "piper":
         print("local voice: it mispronounces unusual names rather than dropping "
               "them - listen to one clip before you commit to a voice.")

+ 41 - 0
skills/proving-it-works-with-a-movie/scripts/narration_contract.py

@@ -0,0 +1,41 @@
+"""Accepted narration shared by rendering and assembly."""
+
+import json
+from pathlib import Path
+
+
+def normalized_text(text):
+    return " ".join((text or "").split())
+
+
+def accepted_narration(narration_dir, scenes):
+    """Return accepted WAVs for narrated non-movie scenes, or explain the gap."""
+    required = [scene for scene in scenes
+                if scene.get("kind", "frames") != "movie"
+                and normalized_text(scene.get("narration"))]
+    if not required:
+        return {}
+    manifest_path = narration_dir / "manifest.json"
+    if not manifest_path.is_file():
+        raise ValueError(f"missing narration manifest: {manifest_path}")
+    try:
+        entries = json.loads(manifest_path.read_text(encoding="utf-8-sig"))
+    except (OSError, ValueError, json.JSONDecodeError) as error:
+        raise ValueError(f"invalid narration manifest: {error}") from error
+    by_id = {entry.get("id"): entry for entry in entries if isinstance(entry, dict)}
+    accepted = {}
+    for scene in required:
+        sid = scene["id"]
+        entry = by_id.get(sid)
+        if entry is None:
+            raise ValueError(f"scene {sid}: missing accepted narration entry")
+        if entry.get("text") != normalized_text(scene.get("narration")):
+            raise ValueError(f"scene {sid}: accepted narration text changed")
+        wav_name = entry.get("wav")
+        if not isinstance(wav_name, str) or not wav_name or Path(wav_name).is_absolute():
+            raise ValueError(f"scene {sid}: invalid accepted narration WAV")
+        wav = narration_dir / wav_name
+        if not wav.is_file():
+            raise ValueError(f"scene {sid}: missing accepted narration WAV: {wav}")
+        accepted[sid] = wav
+    return accepted

+ 1 - 0
tests/proving-it-works-with-a-movie/fixtures.py

@@ -164,6 +164,7 @@ scenes:
     kind: frames
     src: shots
     rate: 1.0
+    narration: one two three four five six seven eight nine ten
 """,
         encoding="utf-8",
     )

+ 10 - 0
tests/proving-it-works-with-a-movie/test_assembly.py

@@ -214,6 +214,10 @@ class AssemblyRegression(unittest.TestCase):
             make_tone(narration / "image.wav", 0.8, 660, cwd=launch)
             make_tone(narration / "frames.wav", 1.4, 770, cwd=launch)
             make_tone(narration / "movie.wav", 1.6, 880, cwd=launch)
+            (narration / "manifest.json").write_text(json.dumps([
+                {"id": "image", "text": "Image narration", "wav": "image.wav", "duration": 0.8},
+                {"id": "frames", "text": "Frames narration", "wav": "frames.wav", "duration": 1.4},
+            ]), encoding="utf-8")
 
             nested_srt = subtitles / "captions.srt"
             nested_srt.write_bytes(
@@ -227,10 +231,12 @@ scenes:
     kind: image
     src: assets/still image.png
     duration: 0.3
+    narration: Image narration
   - id: frames
     kind: frames
     src: {frames.resolve().as_posix()}
     rate: 2.0
+    narration: Frames narration
   - id: movie
     kind: movie
     src: assets/source movie.mp4
@@ -326,6 +332,9 @@ scenes:
             narration = root / "narration"
             narration.mkdir()
             make_tone(narration / "card.wav", 0.8, 550, cwd=root)
+            (narration / "manifest.json").write_text(json.dumps([
+                {"id": "card", "text": "Card narration", "wav": "card.wav", "duration": 0.8}
+            ]), encoding="utf-8")
             scenes = root / "scenes.yaml"
             scenes.write_bytes(
                 b"\xef\xbb\xbfresolution: { width: 160, height: 90 }\r\n"
@@ -336,6 +345,7 @@ scenes:
                 b"    title: Portable\r\n"
                 b"    subtitle: paths\r\n"
                 b"    duration: 0.3\r\n"
+                b"    narration: Card narration\r\n"
             )
             work = root / "work"
             result = fixtures.run_tool(

+ 4 - 2
tests/proving-it-works-with-a-movie/test_narration.py

@@ -148,11 +148,13 @@ class NarrationDriftRegression(unittest.TestCase):
                         (output / "manifest.json").read_text(encoding="utf-8")
                     )
                     self.assertEqual([entry["id"] for entry in manifest], ["accepted"])
-                    rejected_renders.append((output / "rejected.wav").read_bytes())
+                    rejected_renders.append(sorted(
+                        path.read_bytes() for path in output.glob(".rejected.attempt-*.wav")
+                    ))
 
             self.assertEqual(calls.count("Read this sentence exactly."), 1)
             self.assertEqual(calls.count("Keep this evidence out of the manifest."), 4)
-            self.assertNotEqual(rejected_renders[0], rejected_renders[1])
+            self.assertLess(len(rejected_renders[0]), len(rejected_renders[1]))
 
 class TranscriptionProtocolRegression(unittest.TestCase):
     def test_owned_json_is_used_instead_of_library_stdout(self):

+ 333 - 0
tests/proving-it-works-with-a-movie/test_narration_contract.py

@@ -0,0 +1,333 @@
+import io
+import json
+import subprocess
+import sys
+import tempfile
+import unittest
+from contextlib import redirect_stderr, redirect_stdout
+from pathlib import Path
+from unittest.mock import patch
+
+import fixtures
+
+
+class NarrationPublicationContract(unittest.TestCase):
+    def setUp(self):
+        self.module = fixtures.load_script("narrate")
+        self.directory = tempfile.TemporaryDirectory()
+        self.addCleanup(self.directory.cleanup)
+        self.root = Path(self.directory.name)
+        self.scenes = self.root / "scenes.yaml"
+        self.output = self.root / "narration"
+
+    def run_narrate(self, scenes, synthesize, *, verify="off", expected=0):
+        self.scenes.write_text(json.dumps({"scenes": scenes}), encoding="utf-8")
+        argv = ["narrate", str(self.scenes), str(self.output), "--engine", "piper",
+                "--verify", verify]
+        stderr = io.StringIO()
+        with patch.object(sys, "argv", argv), \
+             patch.object(self.module.shutil, "which", return_value="ffprobe"), \
+             patch.object(self.module, "openai_key", return_value=None), \
+             patch.object(self.module, "say_piper", side_effect=synthesize), \
+             patch.object(self.module, "duration", return_value=1.25), \
+             patch.object(self.module, "transcribe_local", return_value=None), \
+             redirect_stdout(io.StringIO()), redirect_stderr(stderr):
+            self.assertEqual(self.module.main(), expected, stderr.getvalue())
+        manifest = self.output / "manifest.json"
+        return json.loads(manifest.read_text(encoding="utf-8")) if manifest.exists() else []
+
+    def test_acceptance_is_withdrawn_before_replacing_accepted_bytes(self):
+        old_wav = self.output / "first.wav"
+        self.output.mkdir()
+        old_wav.write_bytes(b"accepted bytes")
+        synthesis = {"engine": "piper", "voice": self.module.PIPER_VOICE,
+                     "model": self.module.PIPER_VOICE}
+        (self.output / "manifest.json").write_text(json.dumps([{
+            "id": "first", "text": "Old words", "wav": "first.wav",
+            "duration": 1.0, "synthesis": synthesis,
+        }]), encoding="utf-8")
+
+        def synthesize(text, wav, voice):
+            published = json.loads((self.output / "manifest.json").read_text())
+            self.assertNotIn("first", [entry["id"] for entry in published])
+            wav.write_bytes(b"replacement bytes")
+
+        manifest = self.run_narrate(
+            [{"id": "first", "narration": "New words"}], synthesize
+        )
+
+        self.assertEqual(manifest[0]["text"], "New words")
+        self.assertEqual(old_wav.read_bytes(), b"replacement bytes")
+
+    def test_rejected_and_exceptional_takes_are_not_accepted_on_a_rerun(self):
+        attempts = []
+
+        def synthesize(text, wav, voice):
+            attempts.append((text, wav))
+            wav.write_bytes(f"{text}:{len(attempts)}".encode())
+            if text == "Reject this":
+                return "invented words that do not match this script at all"
+            if text == "Raise here":
+                raise RuntimeError("synthesis interrupted")
+
+        scenes = [{"id": "reject", "narration": "Reject this"},
+                  {"id": "raise", "narration": "Raise here"}]
+        first = self.run_narrate(scenes, synthesize, expected=1)
+        second = self.run_narrate(scenes, synthesize, expected=1)
+
+        self.assertEqual(first, [])
+        self.assertEqual(second, [])
+        rejected_paths = [path for text, path in attempts if text == "Reject this"]
+        self.assertEqual(len({path.name for path in rejected_paths}), len(rejected_paths))
+        self.assertTrue(all(path.exists() for path in rejected_paths))
+
+    def test_duration_failure_leaves_only_prior_accepted_scenes_published(self):
+        scenes = [{"id": "accepted", "narration": "Accepted words"},
+                  {"id": "broken", "narration": "Broken words"}]
+
+        def synthesize(text, wav, voice):
+            wav.write_bytes(text.encode())
+
+        self.scenes.write_text(json.dumps({"scenes": scenes}), encoding="utf-8")
+        argv = ["narrate", str(self.scenes), str(self.output), "--engine", "piper",
+                "--verify", "off"]
+        durations = iter((1.0, RuntimeError("ffprobe failed")))
+        with patch.object(sys, "argv", argv), \
+             patch.object(self.module.shutil, "which", return_value="ffprobe"), \
+             patch.object(self.module, "openai_key", return_value=None), \
+             patch.object(self.module, "say_piper", side_effect=synthesize), \
+             patch.object(self.module, "duration", side_effect=durations), \
+             redirect_stdout(io.StringIO()), redirect_stderr(io.StringIO()):
+            self.assertEqual(self.module.main(), 1)
+
+        manifest = json.loads((self.output / "manifest.json").read_text())
+        self.assertEqual([entry["id"] for entry in manifest], ["accepted"])
+
+    def test_two_accepted_scenes_are_published_together(self):
+        def synthesize(text, wav, voice):
+            wav.write_bytes(text.encode())
+
+        manifest = self.run_narrate([
+            {"id": "first", "narration": "First accepted scene"},
+            {"id": "second", "narration": "Second accepted scene"},
+        ], synthesize)
+        self.assertEqual([entry["id"] for entry in manifest], ["first", "second"])
+
+    def test_strict_unavailable_verification_withdraws_cached_acceptance(self):
+        def synthesize(text, wav, voice):
+            wav.write_bytes(b"accepted")
+
+        self.run_narrate([{"id": "clip", "narration": "Read these words"}], synthesize)
+        manifest = self.run_narrate(
+            [{"id": "clip", "narration": "Read these words"}], synthesize,
+            verify="on", expected=1,
+        )
+        self.assertEqual(manifest, [])
+
+    def test_unsupported_asr_comparison_obeys_auto_on_and_off_modes(self):
+        def synthesize(text, wav, voice):
+            wav.write_bytes(b"clip")
+
+        scenes = [{"id": "clip", "narration": "\u77ed\u6587"}]
+        for mode, expected in (("auto", 0), ("on", 1), ("off", 0)):
+            with self.subTest(mode=mode):
+                manifest = self.run_narrate(scenes, synthesize, verify=mode,
+                                             expected=expected)
+                self.assertEqual(bool(manifest), expected == 0)
+
+    def test_missing_ffprobe_stops_before_synthesis(self):
+        self.scenes.write_text(json.dumps({"scenes": [{"id": "clip", "narration": "Words"}]}),
+                              encoding="utf-8")
+        with patch.object(sys, "argv", ["narrate", str(self.scenes), str(self.output)]), \
+             patch.object(self.module.shutil, "which", return_value=None), \
+             patch.object(self.module, "say_piper", side_effect=AssertionError("synthesized")), \
+             redirect_stderr(io.StringIO()):
+            with self.assertRaises(SystemExit):
+                self.module.main()
+
+    def test_cached_unsupported_chat_transcript_withdraws_acceptance_without_asr(self):
+        self.output.mkdir()
+        (self.output / "clip.wav").write_bytes(b"cached bytes")
+        synthesis = {"engine": "openai-chat", "voice": "nova",
+                     "model": self.module.OPENAI_CHAT_MODEL}
+        (self.output / "manifest.json").write_text(json.dumps([{
+            "id": "clip", "text": "\u77ed\u6587", "wav": "clip.wav", "duration": 1.0,
+            "synthesis": synthesis,
+        }]), encoding="utf-8")
+        self.scenes.write_text(json.dumps({"scenes": [
+            {"id": "clip", "narration": "\u77ed\u6587"}
+        ]}), encoding="utf-8")
+        argv = ["narrate", str(self.scenes), str(self.output), "--engine", "openai-chat",
+                "--verify", "off"]
+        with patch.object(sys, "argv", argv), \
+             patch.object(self.module.shutil, "which", return_value="ffprobe"), \
+             patch.object(self.module, "openai_key", return_value="key"), \
+             patch.object(self.module, "say_openai_chat", side_effect=AssertionError("cached")), \
+             patch.object(self.module, "transcribe_local", side_effect=AssertionError("ASR")), \
+             redirect_stdout(io.StringIO()), redirect_stderr(io.StringIO()):
+            self.assertEqual(self.module.main(), 1)
+        self.assertEqual(json.loads((self.output / "manifest.json").read_text()), [])
+
+
+class NarrationComparisonContract(unittest.TestCase):
+    def setUp(self):
+        self.module = fixtures.load_script("narrate")
+
+    def test_comparison_marks_scripts_requiring_segmentation_or_no_words_unavailable(self):
+        for script, heard in (
+            ("你好世界", "こんにちは世界"),
+            ("短文", "短文"),
+            ("mixed 日本語 words", "mixed 日本語 words"),
+            ("*** !!!", "*** !!!"),
+        ):
+            with self.subTest(script=script):
+                self.assertIsNone(self.module.structural_drift(script, heard))
+
+    def test_comparison_accepts_multiline_accented_latin_and_spaced_cyrillic(self):
+        for script, heard in (
+            ("Caf\u00e9\nna\u00efve", "CAF\u00c9 na\u00efve"),
+            ("\u041f\u0440\u0438\u0432\u0435\u0442 \u043c\u0438\u0440", "\u043f\u0440\u0438\u0432\u0435\u0442 \u043c\u0438\u0440"),
+            ("\uc548\ub155 \uc138\uacc4", "\uc548\ub155 \uc138\uacc4"),
+        ):
+            with self.subTest(script=script):
+                self.assertEqual(self.module.structural_drift(script, heard), (0.0, 0))
+
+    def test_drift_check_reports_unavailable_comparison_as_nonzero(self):
+        with tempfile.TemporaryDirectory() as directory:
+            root = Path(directory)
+            script = root / "script.txt"
+            heard = root / "heard.txt"
+            script.write_text("\u77ed\u6587", encoding="utf-8")
+            heard.write_text("\u77ed\u6587", encoding="utf-8")
+            output = io.StringIO()
+            with patch.object(sys, "argv", ["narrate", "--drift-check", str(script), str(heard)]), \
+                 redirect_stdout(output):
+                self.assertEqual(self.module.main(), 1)
+            self.assertIn("comparison unavailable", output.getvalue())
+
+
+class AssemblyNarrationContract(unittest.TestCase):
+    def setUp(self):
+        self.module = fixtures.load_script("assemble")
+        self.directory = tempfile.TemporaryDirectory()
+        self.addCleanup(self.directory.cleanup)
+        self.root = Path(self.directory.name)
+        self.scenes = self.root / "scenes.yaml"
+        self.narration = self.root / "narration"
+        self.narration.mkdir()
+
+    def assemble(self, scenes, *, manifest=None, run_side_effect=None):
+        self.scenes.write_text(json.dumps({"scenes": scenes}), encoding="utf-8")
+        if manifest is not None:
+            (self.narration / "manifest.json").write_text(json.dumps(manifest), encoding="utf-8")
+        calls = []
+
+        def run(command, **kwargs):
+            calls.append(command)
+            if run_side_effect is not None:
+                return run_side_effect(command)
+            return subprocess.CompletedProcess(command, 0, "1.0", "")
+
+        argv = ["assemble", str(self.scenes), str(self.root / "out.mp4"),
+                "--narration", str(self.narration), "--work", str(self.root / "work")]
+        with patch.object(sys, "argv", argv), \
+             patch.object(self.module.shutil, "which", return_value="tool"), \
+             patch.object(self.module, "find_browser", return_value=None), \
+             patch.object(self.module, "run", side_effect=run), \
+             redirect_stdout(io.StringIO()), redirect_stderr(io.StringIO()):
+            return self.module.main(), calls
+
+    def test_required_narration_contract_fails_before_encoding(self):
+        with self.assertRaises(SystemExit):
+            self.assemble([{"id": "spoken", "kind": "image", "src": "still.png",
+                            "narration": "Expected words"}],
+                          run_side_effect=lambda command: (_ for _ in ()).throw(
+                              AssertionError("encoding started")
+                          ))
+
+    def test_missing_entry_wav_or_matching_text_fails_before_encoding(self):
+        (self.root / "still.png").write_bytes(b"image sentinel")
+        cases = (
+            ([], "missing entry"),
+            ([{"id": "spoken", "text": "Expected words", "wav": "missing.wav"}], "missing WAV"),
+            ([{"id": "spoken", "text": "Changed words", "wav": "accepted.wav"}], "changed text"),
+        )
+        for manifest, label in cases:
+            with self.subTest(label=label):
+                (self.narration / "manifest.json").unlink(missing_ok=True)
+                with self.assertRaises(SystemExit):
+                    self.assemble([{"id": "spoken", "kind": "image", "src": "still.png",
+                                    "narration": "Expected words"}], manifest=manifest,
+                                  run_side_effect=lambda command: (_ for _ in ()).throw(
+                                      AssertionError("encoding started")
+                                  ))
+
+    def test_manifest_wav_name_is_authoritative_and_removed_narration_ignores_leftover_wav(self):
+        selected = self.narration / "accepted-name.wav"
+        selected.write_bytes(b"accepted")
+        leftover = self.narration / "silent.wav"
+        leftover.write_bytes(b"leftover")
+        (self.root / "still.png").write_bytes(b"image sentinel")
+        manifest = [{"id": "spoken", "text": "Expected words",
+                     "wav": selected.name, "duration": 1.0, "synthesis": {}}]
+        status, calls = self.assemble([
+            {"id": "spoken", "kind": "image", "src": "still.png", "narration": "Expected  words"},
+            {"id": "silent", "kind": "image", "src": "still.png"},
+        ], manifest=manifest)
+        self.assertEqual(status, 0)
+        encoded = [call for call in calls if call and call[0] == "ffmpeg"]
+        self.assertTrue(any(str(selected) in call for call in encoded))
+        self.assertFalse(any(str(leftover) in call for call in encoded))
+        offsets = json.loads((self.root / "work" / "offsets.json").read_text())
+        self.assertEqual(set(offsets), {"spoken"})
+
+    def test_2560x1080_movie_uses_source_audio_and_silent_source_gets_anullsrc(self):
+        source = self.root / "wide-2560x1080.mp4"
+        source.write_bytes(b"source")
+        status, calls = self.assemble(
+            [{"id": "movie", "kind": "movie", "src": source.name,
+              "narration": "Ignore this", "height": 800}],
+            run_side_effect=lambda command: subprocess.CompletedProcess(
+                command, 0,
+                json.dumps({"streams": [{"codec_type": "video", "width": 2560,
+                                           "height": 1080}]}), ""
+            ) if "-show_streams" in command else subprocess.CompletedProcess(command, 0, "1.0", ""),
+        )
+        self.assertEqual(status, 0)
+        movie_encode = next(call for call in calls if call and call[0] == "ffmpeg")
+        self.assertIn("anullsrc=r=44100:cl=stereo", movie_encode)
+        self.assertIn("0:v:0", movie_encode)
+        self.assertIn("1:a:0", movie_encode)
+        self.assertIn("scale=1920:800:force_original_aspect_ratio=decrease",
+                      movie_encode[movie_encode.index("-vf") + 1])
+        offsets = json.loads((self.root / "work" / "offsets.json").read_text())
+        self.assertEqual(offsets, {})
+
+
+class PercentPathContract(unittest.TestCase):
+    def test_sequence_pattern_escapes_only_directory_percents(self):
+        module = fixtures.load_script("media_paths")
+        pattern = module.sequence_pattern(Path("folder%name") / "frames", "frame-%08d.png")
+        self.assertEqual(pattern, "folder%%name/frames/frame-%08d.png")
+
+    def test_checker_sampling_escapes_only_output_directory_percents(self):
+        module = fixtures.load_script("check-movie")
+        with tempfile.TemporaryDirectory() as directory:
+            work = Path(directory) / "proof%take"
+            work.mkdir()
+            command = []
+
+            def run(argv, **kwargs):
+                command.extend(argv)
+                return subprocess.CompletedProcess(argv, 0, "", "")
+
+            with patch.object(module.subprocess, "run", side_effect=run), redirect_stdout(io.StringIO()):
+                with self.assertRaises(SystemExit):
+                    module.sample_picture(Path("movie.mp4"), work)
+            output = command[-1]
+            self.assertIn("proof%%take", output)
+            self.assertTrue(output.endswith("s%05d.png"))
+
+
+if __name__ == "__main__":
+    unittest.main()