Explorar el Código

fix(movie): refine subtitle chunks using allocated durations

Address Task 2 review round 1: initial character budgets count spaces that disappear between chunks, so proportional timing can exceed --max-secs even when a further word-boundary split is feasible. Reallocate after splitting an over-target multiword cue, checking actual integer millisecond intervals each time.

Coalescing runs once before refinement, and refinement stops at the available millisecond count or unsplittable words. This keeps tiny impossible readability targets bounded while preserving every word and the full measured scene interval.

Added a failing six-second unequal-chunk regression and a tiny-target termination/positive-interval guard. The RED emitted 3273 ms for a cue with a 3000 ms target. GREEN: 43 safe contract tests and six covering existing offset/BOM tests pass. All verification remained text/data or mocked media boundaries.
Drew Ritter hace 3 semanas
padre
commit
e48cb57bf1

+ 25 - 14
skills/proving-it-works-with-a-movie/scripts/make-subtitles

@@ -68,20 +68,31 @@ def scene_cues(text, start, duration, max_chars, max_secs):
         chunks = [" ".join(chunks[i * len(chunks) // available:
                                   (i + 1) * len(chunks) // available])
                   for i in range(available)]
-    total_chars = sum(len(chunk) for chunk in chunks) or 1
-    elapsed_chars, previous = 0, start_ms
-    cues = []
-    for i, chunk in enumerate(chunks):
-        elapsed_chars += len(chunk)
-        remaining = len(chunks) - i - 1
-        boundary = start_ms + round(available * elapsed_chars / total_chars)
-        # Reserve one millisecond per remaining cue, even for very uneven text.
-        boundary = min(end_ms - remaining, max(previous + 1, boundary))
-        if not remaining:
-            boundary = end_ms
-        cues.append((previous / 1000, boundary / 1000, wrap(chunk)))
-        previous = boundary
-    return cues
+    while True:
+        total_chars = sum(len(chunk) for chunk in chunks) or 1
+        elapsed_chars, previous = 0, start_ms
+        cues, split = [], None
+        for i, chunk in enumerate(chunks):
+            elapsed_chars += len(chunk)
+            remaining = len(chunks) - i - 1
+            boundary = start_ms + round(available * elapsed_chars / total_chars)
+            # Reserve one millisecond per remaining cue, even for very uneven text.
+            boundary = min(end_ms - remaining, max(previous + 1, boundary))
+            if not remaining:
+                boundary = end_ms
+            if (split is None and boundary - previous > max_secs * 1000
+                    and len(chunk.split()) > 1):
+                split = i
+            cues.append((previous / 1000, boundary / 1000, wrap(chunk)))
+            previous = boundary
+        if split is None or len(chunks) == available:
+            return cues
+        # Splitting removes a space from the allocation weights, so remeasure all
+        # cues until every splittable chunk fits or milliseconds limit the count.
+        words = chunks[split].split()
+        midpoint = len(words) // 2
+        chunks[split:split + 1] = [" ".join(words[:midpoint]),
+                                   " ".join(words[midpoint:])]
 
 
 def wrap(line, width=42):

+ 17 - 0
tests/proving-it-works-with-a-movie/test_subtitle_contract.py

@@ -84,6 +84,15 @@ class SubtitleTimingContract(unittest.TestCase):
         self.assertGreater(len(cues), 1)
         self.assertTrue(all(b - a <= 3000 for a, b, _ in cues))
 
+    def test_max_seconds_refines_unequal_chunks_against_allocated_time(self):
+        text = "ab cde f ghi"
+        cues, report = self.subtitles([
+            {"id": "unequal", "duration": 6, "text": text},
+        ], "--max-secs", "3")
+        self.assert_scene(cues, 0, 6000, text)
+        self.assertTrue(all(b - a <= 3000 for a, b, _ in cues), cues)
+        self.assertIn("ends at 00:00:06,000", report)
+
     def test_chunks_coalesce_to_fit_representable_milliseconds(self):
         text = "one two six ten red"
         cues, report = self.subtitles([{"id": "tiny", "duration": 0.002, "text": text}], "--max-chars", "3")
@@ -91,6 +100,14 @@ class SubtitleTimingContract(unittest.TestCase):
         self.assertLessEqual(len(cues), 2)
         self.assertIn("ends at 00:00:00,002", report)
 
+    def test_unrepresentable_max_seconds_preserves_positive_cues_and_all_words(self):
+        text = "one two six ten red"
+        cues, _ = self.subtitles([
+            {"id": "tiny", "duration": 0.002, "text": text},
+        ], "--max-secs", "0.0001")
+        self.assert_scene(cues, 0, 2, text)
+        self.assertLessEqual(len(cues), 2)
+
     def test_submillisecond_scene_can_use_its_rounded_interval(self):
         cues, _ = self.subtitles([{"id": "tiny", "duration": 0.0008, "text": "one two"}])
         self.assert_scene(cues, 0, 1, "one two")