Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion pixi.toml
Original file line number Diff line number Diff line change
Expand Up @@ -11,4 +11,4 @@ test = "mojo run -I src test/test_captions.mojo"
demo = "mojo run -I src examples/clip_transcript.mojo test/data/sample.vtt 0 20000"

[dependencies]
mojo = ">=1.0.0b3,<2"
mojo = ">=1.0.0b3.dev0,<2"
34 changes: 31 additions & 3 deletions src/captions/captions.mojo
Original file line number Diff line number Diff line change
Expand Up @@ -176,6 +176,24 @@ def _parse_timing(line: String, mut start_ms: Int, mut end_ms: Int) raises:
end_ms = _parse_ts(right)


def _is_timing_line(line: String) -> Bool:
"""Whether `line` is a genuine `timestamp --> timestamp` timing line.

Cue text may legitimately contain `-->` (e.g. "map --> filter", or a
Unicode-arrow gloss), so a bare `-->` substring is not enough to mark
a line as a cue boundary; both sides must parse as timestamps.
"""
if line.find("-->") == -1:
return False
var start_ms = 0
var end_ms = 0
try:
_parse_timing(line, start_ms, end_ms)
except:
return False
return True


def _strip_voice_tags(text: String, mut speaker: String) -> String:
"""Remove `<v ...>` / `</v>` markup; record the first voice's name."""
if text.find("<v") == -1:
Expand Down Expand Up @@ -285,7 +303,7 @@ def _parse_block(lines: List[String], start: Int, end: Int) raises -> List[Cue]:
while seg_start < end:
var t = -1
for j in range(seg_start, end):
if lines[j].find("-->") != -1:
if _is_timing_line(lines[j]):
t = j
break
if t == -1:
Expand All @@ -299,14 +317,24 @@ def _parse_block(lines: List[String], start: Int, end: Int) raises -> List[Cue]:
index = index * 10 + Int(b) - ord("0")
var start_ms = 0
var end_ms = 0
_parse_timing(lines[t], start_ms, end_ms)
# Isolate each segment's parse: a malformed timing line skips just
# this segment, never discarding cues already gathered from the
# block. `_is_timing_line` already validated `lines[t]`, so this
# is defense-in-depth against divergence between the two.
try:
_parse_timing(lines[t], start_ms, end_ms)
except:
seg_start = t + 1
continue
# A block can hold a second (or third...) cue glued on with no
# blank-line separator. If a later "text" line is itself a
# timing line, that's where this cue's text ends and the next
# cue begins — including its optional index line just before it.
# A bare `-->` inside cue text is not a boundary; only a line that
# parses as `timestamp --> timestamp` is.
var text_end = end
for j in range(t + 1, end):
if lines[j].find("-->") != -1:
if _is_timing_line(lines[j]):
if j > t + 1 and _is_all_digits(lines[j - 1]):
text_end = j - 1
else:
Expand Down
45 changes: 45 additions & 0 deletions test/test_captions.mojo
Original file line number Diff line number Diff line change
Expand Up @@ -334,6 +334,51 @@ def test_srt_glued_cues_without_blank_line() raises:
assert_equal(caps.cues[1].text, "Second cue text.")


def test_arrow_in_cue_text_not_a_boundary() raises:
"""A `-->` inside cue text is prose, not a timing line: the cue must
survive whole rather than being split (or discarded) as a boundary."""
var caps = parse_captions(
String("1\n00:00:01,000 --> 00:00:02,000\nUse map --> filter here.\n")
)
assert_equal(len(caps.cues), 1)
assert_equal(caps.cues[0].text, "Use map --> filter here.")
# And it must round-trip through both serializers unchanged.
var via_srt = parse_captions(to_srt(caps))
assert_equal(len(via_srt.cues), 1)
assert_equal(via_srt.cues[0].text, "Use map --> filter here.")
var via_vtt = parse_captions(to_vtt(caps))
assert_equal(len(via_vtt.cues), 1)
assert_equal(via_vtt.cues[0].text, "Use map --> filter here.")


def test_arrow_in_multiline_text_preserved() raises:
"""A `-->` on a later text line must not truncate the cue or spawn a
bogus second cue; the full multi-line text is kept."""
var caps = parse_captions(
String(
"1\n00:00:01,000 --> 00:00:02,000\n"
"first line\nx --> y transform\nthird line\n"
)
)
assert_equal(len(caps.cues), 1)
assert_equal(caps.cues[0].text, "first line\nx --> y transform\nthird line")


def test_arrow_text_still_splits_glued_real_cue() raises:
"""Even when a cue's text holds a `-->`, a genuinely glued second cue
(a real timing line, no blank separator) is still split out."""
var caps = parse_captions(
String(
"1\n00:00:01,000 --> 00:00:02,000\nUse map --> filter here.\n"
"2\n00:00:03,000 --> 00:00:04,000\nSecond cue text.\n"
)
)
assert_equal(len(caps.cues), 2)
assert_equal(caps.cues[0].text, "Use map --> filter here.")
assert_equal(caps.cues[1].index, 2)
assert_equal(caps.cues[1].text, "Second cue text.")


def test_srt_explicit_zero_index_roundtrip() raises:
"""An explicit cue number of 0 is a real index, not the "absent"
sentinel, and must survive a to_srt round trip unchanged."""
Expand Down
Loading