765 lines
34 KiB
Python
765 lines
34 KiB
Python
"""Tests for cascading progressive-reveal subtitle generation.
|
|
|
|
Covers FCPXMLModifier.generate_dynamic_subtitles(): one standalone <title>
|
|
("Essencial - Título"/Essential Title template) per word by default, cycling
|
|
round-robin across config.lane_count lanes (no compound clip, no <gap>
|
|
anchoring — both were rejected by a real Final Cut Pro import in earlier
|
|
iterations), same-lane non-overlap, minimum one-frame durations, effect
|
|
uid/param fidelity to the FCP exports the user produced by hand, absence of
|
|
any caption role, and round-trip through the parser.
|
|
"""
|
|
|
|
import re
|
|
import shutil
|
|
import tempfile
|
|
import xml.etree.ElementTree as ET
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from fcpxml.models import DynamicSubtitleConfig, TimeValue, WordStyle
|
|
from fcpxml.parser import parse_fcpxml
|
|
from fcpxml.text_layout import POINT_SCALE, REFERENCE_CANVAS_HEIGHT, ink_extent
|
|
from fcpxml.writer import FCPXMLModifier
|
|
|
|
SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml"
|
|
|
|
# The earlier rhythm: one title per WORD. The default is now the
|
|
# progressive composition (one title per LINE), covered in
|
|
# TestProgressiveComposition.
|
|
WORD_MODE = DynamicSubtitleConfig(granularity="word")
|
|
|
|
WORDS = [
|
|
{"word": "Hello", "start": 0.0, "end": 0.4},
|
|
{"word": "there", "start": 0.4, "end": 0.8},
|
|
{"word": "friend", "start": 0.8, "end": 1.3},
|
|
]
|
|
|
|
|
|
@pytest.fixture
|
|
def temp_fcpxml():
|
|
with tempfile.NamedTemporaryFile(suffix=".fcpxml", delete=False) as f:
|
|
shutil.copy(SAMPLE, f.name)
|
|
yield f.name
|
|
Path(f.name).unlink(missing_ok=True)
|
|
|
|
|
|
class TestGenerateDynamicSubtitles:
|
|
def test_one_title_per_word_no_compound_clip(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
# Default max_words_per_line=1 -> one title per word.
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
|
|
assert len(titles) == len(WORDS)
|
|
assert all(t.tag == "title" for t in titles)
|
|
# No <media>/<ref-clip> compound-clip machinery at all.
|
|
resources = modifier.root.find(".//resources")
|
|
assert resources.findall("media") == []
|
|
assert modifier.root.findall(".//ref-clip") == []
|
|
|
|
def test_title_attached_directly_to_parent_clip(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
parent = modifier._require_clip("Interview_A")
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
assert titles[0] in list(parent)
|
|
|
|
def test_effect_uid_matches_real_fcp_export(self, temp_fcpxml):
|
|
"""An incorrect/guessed effect uid makes FCP silently drop every
|
|
connected clip that references it during import — no error, the
|
|
clips just never appear in the timeline or role list. This uid comes
|
|
verbatim from an export the user made by hand in Final Cut and
|
|
exported back out ("teste.fcpxmld" and "posição.fcpxmld", effect r3
|
|
"Text"), not fabricated."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
|
|
resources = modifier.root.find(".//resources")
|
|
effect = next(
|
|
e for e in resources.findall("effect") if e.get("name") == "Text"
|
|
)
|
|
assert effect.get("uid") == (
|
|
".../Titles.localized/Basic Text.localized/"
|
|
"Text.localized/Text.moti"
|
|
)
|
|
|
|
def test_titles_carry_no_caption_role(self, temp_fcpxml):
|
|
"""Regression: dynamic subtitles are animated TITLES, not captions.
|
|
|
|
A role="subtitles.*" makes Final Cut route them to the captions lane,
|
|
which is not drawn over the video unless caption display is enabled —
|
|
so the import succeeds and nothing ever appears. No generated title
|
|
may carry a role attribute.
|
|
"""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
assert titles
|
|
assert all(t.get("role") is None for t in titles)
|
|
|
|
static = modifier.generate_dynamic_subtitles(
|
|
"Broll_City", WORDS, DynamicSubtitleConfig(animated=False)
|
|
)
|
|
assert all(t.get("role") is None for t in static)
|
|
|
|
def test_title_has_start_attribute_and_layout_param_block(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
title = modifier.generate_dynamic_subtitles("Interview_A", WORDS)[0]
|
|
assert title.get("start") == "86486400/24000s"
|
|
|
|
param_names = [p.get("name") for p in title.findall("param")]
|
|
assert param_names == [
|
|
"Position", "Layout Method", "Left Margin", "Right Margin",
|
|
"Top Margin", "Bottom Margin", "Alignment", "Line Spacing",
|
|
"Auto-Shrink", "Alignment", "Opacity", "Speed", "Custom Speed",
|
|
"Apply Speed",
|
|
]
|
|
# The "Custom Speed" param carries the template's own keyframe
|
|
# animation (opacity reveal) — absolute nominal times, constant
|
|
# across every instance, so they replay verbatim.
|
|
anims = title.findall(".//keyframeAnimation")
|
|
assert len(anims) == 1
|
|
|
|
def test_word_text_and_style(self, temp_fcpxml):
|
|
"""Font and face come from the sentence rhythm, whose first entry is
|
|
the reference export's "Toda": Helvetica Light at 170. The size scales
|
|
to the frame — the fixture is 1920x1080, a quarter the height of the
|
|
2160x3840 timeline the rhythm was calibrated on."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
title = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)[0]
|
|
run = title.find("text/text-style")
|
|
assert run.text == "Hello"
|
|
style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style")
|
|
first = WordStyle().rhythm[0]
|
|
assert style_def.get("font") == first.font
|
|
assert style_def.get("fontFace") == first.face
|
|
|
|
scale = (1080 * POINT_SCALE) / REFERENCE_CANVAS_HEIGHT
|
|
assert style_def.get("fontSize") == str(round(first.font_size * scale))
|
|
|
|
def test_font_size_scales_with_the_frame(self, temp_fcpxml):
|
|
"""A vertical 2160x3840 timeline must get the reference sizes back
|
|
unscaled — that is the format the rhythm was calibrated against."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
fmt = modifier.root.find(".//format")
|
|
fmt.set("width", "2160")
|
|
fmt.set("height", "3840")
|
|
title = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)[0]
|
|
run = title.find("text/text-style")
|
|
style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style")
|
|
assert style_def.get("fontSize") == str(WordStyle().rhythm[0].font_size)
|
|
|
|
def test_position_is_keyframed_constant_hold(self, temp_fcpxml):
|
|
"""Regression (2026-08-17): position is a STATIC param in the "Text"
|
|
template. The user's exports ("posição.fcpxmld") write the
|
|
hand-placed words' coordinates as a plain `value="x y"` with no
|
|
keyframeAnimation. Wrapping it in keyframes made FCP ignore the param
|
|
and fall back to the template default, stacking every word at the
|
|
same spot. So the position param must carry a static value — never
|
|
keyframes — and each word must get its own."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
|
|
for t in titles:
|
|
param = next(
|
|
p for p in t.findall("param")
|
|
if p.get("key") == FCPXMLModifier._TEXT_POSITION_KEY
|
|
)
|
|
# Static value, no keyframeAnimation.
|
|
assert param.get("value") is not None
|
|
assert param.find("keyframeAnimation") is None
|
|
|
|
def test_each_word_is_positioned_distinctly(self, temp_fcpxml):
|
|
"""Every word carries its own place in the block, so the template's
|
|
fixed centre never leaves them stacked on one another."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
|
|
positions = [
|
|
next(
|
|
p.get("value")
|
|
for p in t.findall("param")
|
|
if p.get("key") == FCPXMLModifier._TEXT_POSITION_KEY
|
|
)
|
|
for t in titles
|
|
]
|
|
assert len(positions) == len(WORDS)
|
|
assert len(set(positions)) == len(WORDS), "words must not share a spot"
|
|
|
|
def test_no_adjust_transform_on_text_title(self, temp_fcpxml):
|
|
"""The "Text" template positions via its Position param, not
|
|
<adjust-transform> — FCP's own export carries no adjust-transform."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
assert all(t.find("adjust-transform") is None for t in titles)
|
|
|
|
def test_each_word_of_a_block_gets_its_own_lane(self, temp_fcpxml):
|
|
"""Words of one sentence are all on screen together, so none may share
|
|
a lane — sharing one would make Final Cut reject the overlap."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
|
|
assert [t.get("lane") for t in titles] == ["1", "2", "3"]
|
|
|
|
def test_lanes_restart_with_each_block(self, temp_fcpxml):
|
|
"""The previous block has cleared by the time the next one starts, so
|
|
its lanes are free to reuse."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
words, segments = [], []
|
|
for s in range(3):
|
|
base = s * 2.0
|
|
segments.append({"start": base, "end": base + 1.5})
|
|
words += [
|
|
{
|
|
"word": f"s{s}w{i}",
|
|
"start": base + i * 0.4,
|
|
"end": base + i * 0.4 + 0.2,
|
|
}
|
|
for i in range(3)
|
|
]
|
|
titles = modifier.generate_dynamic_subtitles(
|
|
"Interview_A", words, segments=segments
|
|
)
|
|
assert [t.get("lane") for t in titles] == ["1", "2", "3"] * 3
|
|
|
|
def test_same_lane_titles_never_overlap_in_time(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
words, segments = [], []
|
|
for s in range(3):
|
|
base = s * 2.0
|
|
segments.append({"start": base, "end": base + 1.5})
|
|
words += [
|
|
{
|
|
"word": f"s{s}w{i}",
|
|
"start": base + i * 0.4,
|
|
"end": base + i * 0.4 + 0.2,
|
|
}
|
|
for i in range(3)
|
|
]
|
|
titles = modifier.generate_dynamic_subtitles(
|
|
"Interview_A", words, segments=segments
|
|
)
|
|
|
|
by_lane: dict[str, list] = {}
|
|
for t in titles:
|
|
by_lane.setdefault(t.get("lane"), []).append(t)
|
|
|
|
for lane_titles in by_lane.values():
|
|
spans = sorted(
|
|
(
|
|
modifier._parse_time(t.get("offset")),
|
|
modifier._parse_time(t.get("offset")) + modifier._parse_time(t.get("duration")),
|
|
)
|
|
for t in lane_titles
|
|
)
|
|
for (_, end_a), (start_b, _) in zip(spans, spans[1:]):
|
|
assert end_a <= start_b
|
|
|
|
def test_words_of_a_block_accumulate_on_screen(self, temp_fcpxml):
|
|
"""Each word appears later than the last but they stay up together —
|
|
that is the sentence building up in front of the viewer."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
offsets = [modifier._parse_time(t.get("offset")) for t in titles]
|
|
assert offsets == sorted(offsets)
|
|
assert offsets[0] < offsets[-1], "words must not all start at once"
|
|
first_end = offsets[0] + modifier._parse_time(titles[0].get("duration"))
|
|
assert first_end > offsets[-1], "the first word must outlast the last"
|
|
|
|
def test_every_word_of_a_block_clears_at_the_same_instant(self, temp_fcpxml):
|
|
"""The whole sentence vanishes at once, rather than word by word."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
ends = {
|
|
(
|
|
modifier._parse_time(t.get("offset"))
|
|
+ modifier._parse_time(t.get("duration"))
|
|
).to_fcpxml()
|
|
for t in titles
|
|
}
|
|
assert len(ends) == 1, f"expected one shared end, got {ends}"
|
|
|
|
def test_color_follows_the_style_rhythm(self, temp_fcpxml):
|
|
"""Colour comes from the word's place in the sentence's rhythm, which
|
|
reproduces the palette of the user's calibration export."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
|
|
|
|
def color_of(title):
|
|
run = title.find("text/text-style")
|
|
style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style")
|
|
return style_def.get("fontColor")
|
|
|
|
rhythm = WordStyle().rhythm
|
|
assert [color_of(t) for t in titles] == [
|
|
rhythm[i].color for i in range(len(WORDS))
|
|
]
|
|
|
|
def test_effect_resource_created_once_and_deduped(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
modifier.generate_dynamic_subtitles("Broll_City", WORDS)
|
|
|
|
resources = modifier.root.find(".//resources")
|
|
title_effects = [
|
|
e for e in resources.findall("effect") if e.get("name") == "Text"
|
|
]
|
|
assert len(title_effects) == 1
|
|
|
|
def test_no_duplicate_ids_across_multiple_calls_on_same_modifier(self, temp_fcpxml):
|
|
"""Regression test: the handler calls generate_dynamic_subtitles once
|
|
per spine clip on the SAME FCPXMLModifier instance. Every <title>
|
|
name and <text-style-def> id must stay unique document-wide, or FCP
|
|
rejects the whole import with 'ID ... already defined'."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
modifier.generate_dynamic_subtitles("Broll_City", WORDS)
|
|
modifier.generate_dynamic_subtitles("Broll_Studio", WORDS)
|
|
|
|
all_title_names = [t.get("name") for t in modifier.root.findall(".//title")]
|
|
all_style_def_ids = [sd.get("id") for sd in modifier.root.findall(".//text-style-def")]
|
|
|
|
assert len(all_title_names) == len(set(all_title_names))
|
|
assert len(all_style_def_ids) == len(set(all_style_def_ids))
|
|
|
|
def test_element_param_bypasses_ambiguous_duplicate_name_lookup(self, temp_fcpxml):
|
|
"""Regression test for 2026-08-17: after ripple-cut/silence-removal,
|
|
every fragment of an originally-named clip keeps the same `name`.
|
|
Passing that (now-ambiguous) name resolves via `self.clips` — keyed
|
|
by name — to whichever clip was indexed LAST, silently attaching
|
|
every clip's captions to one wrong spine element. Passing the
|
|
`ET.Element` itself (what the real handler in server.py does) must
|
|
attach to the exact clip intended, regardless of duplicate names."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
interview_el = modifier._require_clip("Interview_A")
|
|
broll_el = modifier._require_clip("Broll_City")
|
|
# Simulate the real-world collision: both clips now share one name.
|
|
broll_el.set("name", "Interview_A")
|
|
modifier._build_clip_index()
|
|
|
|
titles_a = modifier.generate_dynamic_subtitles(interview_el, WORDS)
|
|
titles_b = modifier.generate_dynamic_subtitles(
|
|
broll_el, [{"word": "elsewhere", "start": 0.5, "end": 0.9}]
|
|
)
|
|
|
|
assert titles_a and all(t in list(interview_el) for t in titles_a)
|
|
assert titles_b and all(t in list(broll_el) for t in titles_b)
|
|
# The two clips' captions must not have landed on the same element.
|
|
assert list(interview_el.findall("title")) != list(broll_el.findall("title"))
|
|
|
|
def test_no_zero_duration_titles_for_tight_word_timing(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
tight_words = [
|
|
{"word": "um", "start": 0.0, "end": 0.0},
|
|
{"word": "dois", "start": 0.0, "end": 0.0},
|
|
]
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", tight_words)
|
|
for t in titles:
|
|
assert modifier._parse_time(t.get("duration")) > TimeValue.zero()
|
|
|
|
def test_one_title_per_word_not_per_line(self, temp_fcpxml):
|
|
"""Words are never merged into a single title — each is placed
|
|
separately so it can appear on its own beat."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
long_words = WORDS + [{"word": "again", "start": 1.3, "end": 1.6}]
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", long_words, WORD_MODE)
|
|
|
|
assert len(titles) == len(long_words)
|
|
assert [t.find("text/text-style").text for t in titles] == [
|
|
w["word"] for w in long_words
|
|
]
|
|
|
|
def test_segments_split_words_into_separate_blocks(self, temp_fcpxml):
|
|
"""Two Whisper segments must produce two blocks that clear
|
|
independently, rather than one run-on block."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
words = [
|
|
{"word": "um", "start": 0.0, "end": 0.4},
|
|
{"word": "dois", "start": 0.5, "end": 0.9},
|
|
{"word": "tres", "start": 5.0, "end": 5.4},
|
|
]
|
|
segments = [{"start": 0.0, "end": 1.0}, {"start": 5.0, "end": 6.0}]
|
|
titles = modifier.generate_dynamic_subtitles(
|
|
"Interview_A", words, segments=segments
|
|
)
|
|
|
|
def end_of(t):
|
|
return (
|
|
modifier._parse_time(t.get("offset"))
|
|
+ modifier._parse_time(t.get("duration"))
|
|
)
|
|
|
|
assert end_of(titles[0]) == end_of(titles[1]), "block 1 clears together"
|
|
assert end_of(titles[2]) != end_of(titles[0]), "block 2 is independent"
|
|
|
|
def test_long_sentence_splits_into_sub_blocks(self, temp_fcpxml):
|
|
"""A sentence taller than the band is broken up rather than spilling
|
|
off screen — so no block exceeds the lanes its words need."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
words = [
|
|
{"word": f"palavra{i}", "start": float(i) * 0.2, "end": float(i) * 0.2 + 0.15}
|
|
for i in range(30)
|
|
]
|
|
segments = [{"start": 0.0, "end": 30.0}]
|
|
titles = modifier.generate_dynamic_subtitles(
|
|
"Interview_A", words, WORD_MODE, segments=segments
|
|
)
|
|
assert len(titles) == len(words)
|
|
# More than one distinct clear-time means the sentence was split.
|
|
ends = {
|
|
(
|
|
modifier._parse_time(t.get("offset"))
|
|
+ modifier._parse_time(t.get("duration"))
|
|
).to_fcpxml()
|
|
for t in titles
|
|
}
|
|
assert len(ends) > 1
|
|
|
|
def test_empty_words_returns_empty_list(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
assert modifier.generate_dynamic_subtitles("Interview_A", []) == []
|
|
|
|
def test_missing_clip_raises(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
with pytest.raises(ValueError):
|
|
modifier.generate_dynamic_subtitles("NoSuchClip", WORDS)
|
|
|
|
def test_round_trip_through_writer_and_parser(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
|
|
out_path = str(Path(temp_fcpxml).with_suffix(".out.fcpxml"))
|
|
try:
|
|
modifier.save(out_path)
|
|
tree = ET.parse(out_path)
|
|
titles = tree.getroot().findall(".//title")
|
|
assert len(titles) == len(WORDS)
|
|
|
|
parsed = parse_fcpxml(out_path)
|
|
assert parsed is not None
|
|
finally:
|
|
Path(out_path).unlink(missing_ok=True)
|
|
|
|
def test_animated_offset_uses_source_media_coordinates(self, temp_fcpxml):
|
|
"""Regression: anchored titles are positioned in SOURCE media
|
|
coordinates — FCP's own output ("exemplo de arquivos.fcpxmld") writes
|
|
offset = parent clip's `start` + timeline-relative. Writing a plain
|
|
relative offset drops the title to ~0s of the media, before the clip's
|
|
in-point, so FCP never shows it ("legendas fora"). The default
|
|
animated=True must add the clip's source `start` to every offset."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
parent = modifier._require_clip("Interview_A")
|
|
parent.set("start", "226220995/24000s")
|
|
titles = modifier.generate_dynamic_subtitles(parent, WORDS)
|
|
|
|
clip_start = modifier._parse_time("226220995/24000s")
|
|
titles = sorted(titles, key=lambda t: modifier._parse_time(t.get("offset")))
|
|
for t in titles:
|
|
offset = modifier._parse_time(t.get("offset"))
|
|
# offset must be >= clip source start (== for the first line at
|
|
# relative 0.0, > for later lines) — never a small relative value
|
|
# that FCP would read as ~0s of the media.
|
|
assert offset >= clip_start
|
|
# Relative spacing between lines is preserved (offset - clip_start).
|
|
starts = [modifier._parse_time(t.get("offset")) - clip_start for t in titles]
|
|
assert starts == sorted(starts)
|
|
assert starts[0].to_seconds() >= 0.0
|
|
|
|
def test_effect_is_basic_text_template(self, temp_fcpxml):
|
|
"""Dynamic subtitles must resolve the "Text" (Basic Text) effect uid
|
|
and produce roleless titles — the structure of FCP's own export, and
|
|
the only template we have verified renders in Final Cut."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
|
|
resources = modifier.root.find(".//resources")
|
|
effect = next(
|
|
e for e in resources.findall("effect") if e.get("name") == "Text"
|
|
)
|
|
assert effect.get("uid") == (
|
|
".../Titles.localized/Basic Text.localized/"
|
|
"Text.localized/Text.moti"
|
|
)
|
|
for t in modifier.root.findall(".//title"):
|
|
assert t.get("ref") == effect.get("id")
|
|
assert t.get("role") is None
|
|
assert t.get("start") == "86486400/24000s"
|
|
|
|
def test_all_configs_use_the_one_text_template(self, temp_fcpxml):
|
|
"""animated True/False now both resolve to the single "Text" template —
|
|
the earlier Essential Title / Título Básico templates never rendered,
|
|
so they were retired."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
modifier.generate_dynamic_subtitles(
|
|
"Interview_A", WORDS, DynamicSubtitleConfig(animated=False)
|
|
)
|
|
|
|
resources = modifier.root.find(".//resources")
|
|
effects = {e.get("name") for e in resources.findall("effect")}
|
|
assert effects == {"Text"}
|
|
|
|
|
|
class TestClipBoundaryClamping:
|
|
"""A connected title is not trimmed by its parent clip's out-point: FCP
|
|
keeps drawing it over the following clip. Words whose Whisper end (or
|
|
start) runs past the cut must therefore be clamped, or the last block of
|
|
one clip overlaps the first block of the next one on screen."""
|
|
|
|
def _parent_and_titles(self, path, words):
|
|
modifier = FCPXMLModifier(path)
|
|
parent = modifier._require_clip("Interview_A")
|
|
titles = modifier.generate_dynamic_subtitles(parent, words)
|
|
limit = modifier._parse_time(parent.get("duration")).to_seconds()
|
|
origin = modifier._parse_time(parent.get("start", "0s")).to_seconds()
|
|
spans = [
|
|
(
|
|
modifier._parse_time(t.get("offset")).to_seconds() - origin,
|
|
modifier._parse_time(t.get("duration")).to_seconds(),
|
|
)
|
|
for t in titles
|
|
]
|
|
return limit, spans
|
|
|
|
def test_last_word_end_beyond_clip_is_clamped(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
limit = modifier._parse_time(
|
|
modifier._require_clip("Interview_A").get("duration")
|
|
).to_seconds()
|
|
words = [
|
|
{"word": "fim", "start": limit - 0.5, "end": limit + 5.0},
|
|
]
|
|
clip_limit, spans = self._parent_and_titles(temp_fcpxml, words)
|
|
for offset, duration in spans:
|
|
assert offset + duration <= clip_limit + 1e-6
|
|
|
|
def test_word_starting_past_clip_end_is_dropped(self, temp_fcpxml):
|
|
"""It could only ever be seen over the NEXT clip's captions."""
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
limit = modifier._parse_time(
|
|
modifier._require_clip("Interview_A").get("duration")
|
|
).to_seconds()
|
|
words = [
|
|
{"word": "ok", "start": 0.2, "end": 0.6},
|
|
{"word": "tarde", "start": limit + 2.0, "end": limit + 3.0},
|
|
]
|
|
clip_limit, spans = self._parent_and_titles(temp_fcpxml, words)
|
|
assert len(spans) == 1
|
|
for offset, duration in spans:
|
|
assert offset < clip_limit
|
|
assert offset + duration <= clip_limit + 1e-6
|
|
|
|
|
|
class TestTextStyleIds:
|
|
"""Regression for 2026-08-17: <text-style-def id> is DTD type ID, so it
|
|
must be a valid XML Name. Deriving it from the title name (built from the
|
|
caption text) emitted ids with spaces, accents and leading digits, and
|
|
xmllint rejected the whole document — "Syntax of value for attribute id
|
|
of text-style-def is not valid" — so FCP refused the import.
|
|
"""
|
|
|
|
XML_NAME = re.compile(r"^[A-Za-z_][A-Za-z0-9_.\-]*$")
|
|
|
|
def _style_ids(self, modifier):
|
|
return [sd.get("id") for sd in modifier.root.findall(".//text-style-def")]
|
|
|
|
def test_text_title_id_is_a_valid_xml_name(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
modifier.add_text_title(
|
|
"Interview_A", "3 coisas que você precisa saber", duration="1s"
|
|
)
|
|
ids = self._style_ids(modifier)
|
|
assert ids and all(self.XML_NAME.match(i) for i in ids), ids
|
|
|
|
def test_style_id_matches_its_run_ref(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
title = modifier.add_text_title("Interview_A", "Olá, mundo!", duration="1s")
|
|
run = title.find("text/text-style")
|
|
assert run.get("ref") == title.find("text-style-def").get("id")
|
|
|
|
def test_repeated_titles_get_unique_ids(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
for _ in range(3):
|
|
modifier.add_text_title("Interview_A", "Olá", duration="1s")
|
|
ids = self._style_ids(modifier)
|
|
assert len(ids) == len(set(ids)) == 3
|
|
|
|
def test_id_survives_text_with_no_ascii_characters(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
modifier.add_text_title("Interview_A", "日本語", duration="1s")
|
|
ids = self._style_ids(modifier)
|
|
assert ids and all(self.XML_NAME.match(i) for i in ids), ids
|
|
|
|
def test_dynamic_subtitle_ids_are_valid_xml_names(self, temp_fcpxml):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
|
|
ids = self._style_ids(modifier)
|
|
assert len(ids) == len(titles)
|
|
assert all(self.XML_NAME.match(i) for i in ids), ids
|
|
|
|
|
|
class TestProgressiveComposition:
|
|
"""The default look (reference reel the user sent, 2026-08-17):
|
|
|
|
[ que vão ]
|
|
[ melhorar ]
|
|
[ sua legenda ]
|
|
|
|
One title per LINE, not per word. Supporting words are grouped and set
|
|
small in a grotesque; the sentence's key word is set large in a display
|
|
italic on a line of its own. Each block appears as its own words are
|
|
spoken and stays on screen, so the sentence assembles itself, and every
|
|
block of a composition clears at the same instant.
|
|
"""
|
|
|
|
PHRASE = "que vão melhorar sua legenda".split()
|
|
|
|
def _words(self, texts=None, step=0.5):
|
|
texts = texts or self.PHRASE
|
|
return [
|
|
{"word": w, "start": i * step, "end": i * step + step * 0.9}
|
|
for i, w in enumerate(texts)
|
|
]
|
|
|
|
def _titles(self, temp_fcpxml, words=None):
|
|
modifier = FCPXMLModifier(temp_fcpxml)
|
|
fmt = modifier.root.find(".//format")
|
|
fmt.set("width", "2160")
|
|
fmt.set("height", "3840")
|
|
words = words or self._words()
|
|
segments = [{"start": 0.0, "end": words[-1]["end"]}]
|
|
return modifier, modifier.generate_dynamic_subtitles(
|
|
"Interview_A", words, segments=segments
|
|
)
|
|
|
|
@staticmethod
|
|
def _style(title):
|
|
run = title.find("text/text-style")
|
|
return title.find(f"text-style-def[@id='{run.get('ref')}']/text-style")
|
|
|
|
@staticmethod
|
|
def _position(title):
|
|
param = title.find(f"param[@key='{FCPXMLModifier._TEXT_POSITION_KEY}']")
|
|
x, y = param.get("value").split()
|
|
return float(x), float(y)
|
|
|
|
def test_one_title_per_line_grouping_supporting_words(self, temp_fcpxml):
|
|
_, titles = self._titles(temp_fcpxml)
|
|
assert [t.find("text/text-style").text for t in titles] == [
|
|
"que vão",
|
|
"melhorar",
|
|
"sua legenda",
|
|
]
|
|
|
|
def test_key_word_is_set_large_in_the_display_italic(self, temp_fcpxml):
|
|
_, titles = self._titles(temp_fcpxml)
|
|
emphasis, body = self._style(titles[1]), self._style(titles[0])
|
|
assert emphasis.get("font") == "Playfair Display"
|
|
assert "Italic" in emphasis.get("fontFace")
|
|
assert body.get("font") == "Helvetica Neue"
|
|
assert int(emphasis.get("fontSize")) > 2 * int(body.get("fontSize"))
|
|
|
|
def test_every_line_is_white(self, temp_fcpxml):
|
|
_, titles = self._titles(temp_fcpxml)
|
|
assert {self._style(t).get("fontColor") for t in titles} == {"1 1 1 1"}
|
|
|
|
def test_lines_stack_downward_and_stagger_around_the_key_word(self, temp_fcpxml):
|
|
_, titles = self._titles(temp_fcpxml)
|
|
(x_top, y_top), (x_mid, y_mid), (x_low, y_low) = map(self._position, titles)
|
|
assert y_top > y_mid > y_low, "lines stack top to bottom"
|
|
assert x_mid == 0, "the key word stays centred"
|
|
assert x_top < 0 < x_low, "supporting lines hang off opposite edges"
|
|
|
|
def test_blocks_enter_as_spoken_and_clear_together(self, temp_fcpxml):
|
|
modifier, titles = self._titles(temp_fcpxml)
|
|
starts = [modifier._parse_time(t.get("offset")) for t in titles]
|
|
ends = [
|
|
s + modifier._parse_time(t.get("duration"))
|
|
for s, t in zip(starts, titles)
|
|
]
|
|
assert starts == sorted(starts) and starts[0] < starts[-1]
|
|
assert len({e.to_fcpxml() for e in ends}) == 1, "the composition clears as one"
|
|
|
|
def test_lines_never_overlap_on_screen(self, temp_fcpxml):
|
|
"""Measured on the REAL ink each line contains — the accents and
|
|
descenders of a display italic reach far past its cap height, and
|
|
that is the pair that touches if the stack is spaced nominally."""
|
|
_, titles = self._titles(temp_fcpxml)
|
|
boxes = []
|
|
for t in titles:
|
|
_, y = self._position(t)
|
|
style = self._style(t)
|
|
top, bottom = ink_extent(
|
|
t.find("text/text-style").text,
|
|
float(style.get("fontSize")),
|
|
font=style.get("font"),
|
|
face=style.get("fontFace"),
|
|
)
|
|
boxes.append((y + bottom, y + top))
|
|
for (_, top_above), (bottom_below, _) in zip(boxes, boxes[1:]):
|
|
assert bottom_below < top_above # ordered top to bottom
|
|
assert top_above <= boxes[0][1]
|
|
for i, (lo_a, hi_a) in enumerate(boxes):
|
|
for lo_b, hi_b in boxes[i + 1:]:
|
|
assert hi_b <= lo_a, "rendered ink must not intersect"
|
|
|
|
@pytest.mark.parametrize("phrase", [
|
|
"que vão melhorar sua legenda",
|
|
"o segredo é começar antes de estar pronto",
|
|
"ninguém ganha dinheiro gravando vídeo tremido à noite",
|
|
"ÇÃO PAGA jjj ggg ppp qqq yyy ÁÊÍÕÚ",
|
|
"a b c d e f g h i j k l",
|
|
])
|
|
def test_ink_never_overlaps_for_any_phrase(self, temp_fcpxml, phrase):
|
|
"""The guarantee has to hold for accents stacked over descenders,
|
|
for all-caps, and for a run of one-letter words — not just for the
|
|
reference sentence."""
|
|
modifier, titles = self._titles(temp_fcpxml, self._words(phrase.split()))
|
|
# Only lines that share the screen can collide. Every block of one
|
|
# composition clears at the same instant, and the next composition
|
|
# starts there, so the clear time groups what is on screen together.
|
|
onscreen = {}
|
|
for t in titles:
|
|
_, y = self._position(t)
|
|
style = self._style(t)
|
|
top, bottom = ink_extent(
|
|
t.find("text/text-style").text,
|
|
float(style.get("fontSize")),
|
|
font=style.get("font"),
|
|
face=style.get("fontFace"),
|
|
)
|
|
clear = (
|
|
modifier._parse_time(t.get("offset"))
|
|
+ modifier._parse_time(t.get("duration"))
|
|
).to_fcpxml()
|
|
onscreen.setdefault(clear, []).append((y + bottom, y + top))
|
|
|
|
assert onscreen, phrase
|
|
for boxes in onscreen.values():
|
|
for i, (lo_a, hi_a) in enumerate(boxes):
|
|
for lo_b, hi_b in boxes[i + 1:]:
|
|
assert hi_b <= lo_a or lo_b >= hi_a, phrase
|
|
|
|
def test_function_words_are_never_the_emphasis(self, temp_fcpxml):
|
|
_, titles = self._titles(temp_fcpxml, self._words("e o que importa é constância".split()))
|
|
big = max(titles, key=lambda t: int(self._style(t).get("fontSize")))
|
|
assert big.find("text/text-style").text == "constância"
|
|
|
|
def test_each_line_gets_its_own_lane(self, temp_fcpxml):
|
|
_, titles = self._titles(temp_fcpxml)
|
|
assert [t.get("lane") for t in titles] == ["1", "2", "3"]
|
|
|
|
def test_single_word_sentence_still_composes(self, temp_fcpxml):
|
|
_, titles = self._titles(temp_fcpxml, self._words(["chega"]))
|
|
assert len(titles) == 1
|
|
assert self._style(titles[0]).get("font") == "Playfair Display"
|
|
|
|
def test_long_sentence_splits_into_successive_compositions(self, temp_fcpxml):
|
|
words = self._words([f"palavra{i}" for i in range(24)], step=0.3)
|
|
modifier, titles = self._titles(temp_fcpxml, words)
|
|
ends = {
|
|
(
|
|
modifier._parse_time(t.get("offset"))
|
|
+ modifier._parse_time(t.get("duration"))
|
|
).to_fcpxml()
|
|
for t in titles
|
|
}
|
|
assert len(ends) > 1, "a long sentence must break into several compositions"
|
|
spoken = " ".join(w["word"] for w in words)
|
|
emitted = " ".join(t.find("text/text-style").text for t in titles)
|
|
assert emitted == spoken, "no spoken word may be dropped or reordered"
|