"""Tests for cascading progressive-reveal subtitle generation.
Covers FCPXMLModifier.generate_dynamic_subtitles(): one standalone
("Essencial - Título"/Essential Title template) per word by default, cycling
round-robin across config.lane_count lanes (no compound clip, no
anchoring — both were rejected by a real Final Cut Pro import in earlier
iterations), same-lane non-overlap, minimum one-frame durations, effect
uid/param fidelity to the FCP exports the user produced by hand, absence of
any caption role, and round-trip through the parser.
"""
import re
import shutil
import tempfile
import xml.etree.ElementTree as ET
from pathlib import Path
import pytest
from fcpxml.models import DynamicSubtitleConfig, TimeValue, WordLook, WordStyle
from fcpxml.parser import parse_fcpxml
from fcpxml.text_layout import (
POINT_SCALE,
REFERENCE_BLOCK_LINE_GAP,
REFERENCE_CANVAS_HEIGHT,
TEXT_TEMPLATE_FONT_SCALE,
LayoutBox,
compose_sentence,
ink_extent,
)
from fcpxml.writer import FCPXMLModifier
SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml"
def font_points(style) -> float:
"""The style's size back in CANVAS POINTS.
The emitted fontSize lives in the template's own space, which is
TEXT_TEMPLATE_FONT_SCALE times bigger than the space positions use, so any
check that mixes the two has to convert first."""
return float(style.get("fontSize")) / TEXT_TEMPLATE_FONT_SCALE
# The earlier rhythm: one title per WORD. The default is now the
# progressive composition (one title per LINE), covered in
# TestProgressiveComposition.
WORD_MODE = DynamicSubtitleConfig(granularity="word")
WORDS = [
{"word": "Hello", "start": 0.0, "end": 0.4},
{"word": "there", "start": 0.4, "end": 0.8},
{"word": "friend", "start": 0.8, "end": 1.3},
]
@pytest.fixture
def temp_fcpxml():
with tempfile.NamedTemporaryFile(suffix=".fcpxml", delete=False) as f:
shutil.copy(SAMPLE, f.name)
yield f.name
Path(f.name).unlink(missing_ok=True)
class TestGenerateDynamicSubtitles:
def test_one_title_per_word_no_compound_clip(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
# Default max_words_per_line=1 -> one title per word.
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
assert len(titles) == len(WORDS)
assert all(t.tag == "title" for t in titles)
# No / compound-clip machinery at all.
resources = modifier.root.find(".//resources")
assert resources.findall("media") == []
assert modifier.root.findall(".//ref-clip") == []
def test_title_attached_directly_to_parent_clip(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
parent = modifier._require_clip("Interview_A")
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
assert titles[0] in list(parent)
def test_effect_uid_matches_real_fcp_export(self, temp_fcpxml):
"""An incorrect/guessed effect uid makes FCP silently drop every
connected clip that references it during import — no error, the
clips just never appear in the timeline or role list. This uid comes
verbatim from an export the user made by hand in Final Cut and
exported back out ("teste.fcpxmld" and "posição.fcpxmld", effect r3
"Text"), not fabricated."""
modifier = FCPXMLModifier(temp_fcpxml)
modifier.generate_dynamic_subtitles("Interview_A", WORDS)
resources = modifier.root.find(".//resources")
effect = next(
e for e in resources.findall("effect") if e.get("name") == "Text"
)
assert effect.get("uid") == (
".../Titles.localized/Basic Text.localized/"
"Text.localized/Text.moti"
)
def test_titles_carry_no_caption_role(self, temp_fcpxml):
"""Regression: dynamic subtitles are animated TITLES, not captions.
A role="subtitles.*" makes Final Cut route them to the captions lane,
which is not drawn over the video unless caption display is enabled —
so the import succeeds and nothing ever appears. No generated title
may carry a role attribute.
"""
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
assert titles
assert all(t.get("role") is None for t in titles)
static = modifier.generate_dynamic_subtitles(
"Broll_City", WORDS, DynamicSubtitleConfig(animated=False)
)
assert all(t.get("role") is None for t in static)
def test_title_has_start_attribute_and_layout_param_block(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
title = modifier.generate_dynamic_subtitles("Interview_A", WORDS)[0]
assert title.get("start") == "86486400/24000s"
param_names = [p.get("name") for p in title.findall("param")]
assert param_names == [
"Position", "Build Out", "Layout Method", "Left Margin", "Right Margin",
"Top Margin", "Bottom Margin", "Alignment", "Line Spacing",
"Auto-Shrink", "Alignment", "Opacity", "Speed", "Custom Speed",
"Apply Speed",
]
# The "Custom Speed" param carries the template's own keyframe
# animation (opacity reveal) — absolute nominal times, constant
# across every instance, so they replay verbatim.
anims = title.findall(".//keyframeAnimation")
assert len(anims) == 1
def test_word_text_and_style(self, temp_fcpxml):
"""Font and face come from the sentence rhythm, whose first entry is
the reference export's "Toda": Helvetica Light at 170. The size scales
to the frame — the fixture is 1920x1080, a quarter the height of the
2160x3840 timeline the rhythm was calibrated on."""
modifier = FCPXMLModifier(temp_fcpxml)
title = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)[0]
run = title.find("text/text-style")
assert run.text == "Hello"
style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style")
first = WordStyle().rhythm[0]
assert style_def.get("font") == first.font
assert style_def.get("fontFace") == first.face
scale = (1080 * POINT_SCALE) / REFERENCE_CANVAS_HEIGHT
expected = round(first.font_size * scale) * TEXT_TEMPLATE_FONT_SCALE
assert float(style_def.get("fontSize")) == expected
def test_font_size_scales_with_the_frame(self, temp_fcpxml):
"""A vertical 2160x3840 timeline must get the reference sizes back
unscaled — that is the format the rhythm was calibrated against."""
modifier = FCPXMLModifier(temp_fcpxml)
fmt = modifier.root.find(".//format")
fmt.set("width", "2160")
fmt.set("height", "3840")
title = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)[0]
run = title.find("text/text-style")
style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style")
assert float(style_def.get("fontSize")) == (
WordStyle().rhythm[0].font_size * TEXT_TEMPLATE_FONT_SCALE
)
def test_position_is_keyframed_constant_hold(self, temp_fcpxml):
"""Regression (2026-08-17): position is a STATIC param in the "Text"
template. The user's exports ("posição.fcpxmld") write the
hand-placed words' coordinates as a plain `value="x y"` with no
keyframeAnimation. Wrapping it in keyframes made FCP ignore the param
and fall back to the template default, stacking every word at the
same spot. So the position param must carry a static value — never
keyframes — and each word must get its own."""
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
for t in titles:
param = next(
p for p in t.findall("param")
if p.get("key") == FCPXMLModifier._TEXT_POSITION_KEY
)
# Static value, no keyframeAnimation.
assert param.get("value") is not None
assert param.find("keyframeAnimation") is None
def test_each_word_is_positioned_distinctly(self, temp_fcpxml):
"""Every word carries its own place in the block, so the template's
fixed centre never leaves them stacked on one another."""
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
positions = [
next(
p.get("value")
for p in t.findall("param")
if p.get("key") == FCPXMLModifier._TEXT_POSITION_KEY
)
for t in titles
]
assert len(positions) == len(WORDS)
assert len(set(positions)) == len(WORDS), "words must not share a spot"
def test_no_adjust_transform_on_text_title(self, temp_fcpxml):
"""The "Text" template positions via its Position param, not
— FCP's own export carries no adjust-transform."""
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
assert all(t.find("adjust-transform") is None for t in titles)
def test_each_word_of_a_block_gets_its_own_lane(self, temp_fcpxml):
"""Words of one sentence are all on screen together, so none may share
a lane — sharing one would make Final Cut reject the overlap."""
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
assert [t.get("lane") for t in titles] == ["1", "2", "3"]
def test_lanes_restart_with_each_block(self, temp_fcpxml):
"""The previous block has cleared by the time the next one starts, so
its lanes are free to reuse."""
modifier = FCPXMLModifier(temp_fcpxml)
words, segments = [], []
for s in range(3):
base = s * 2.0
segments.append({"start": base, "end": base + 1.5})
words += [
{
"word": f"s{s}w{i}",
"start": base + i * 0.4,
"end": base + i * 0.4 + 0.2,
}
for i in range(3)
]
titles = modifier.generate_dynamic_subtitles(
"Interview_A", words, segments=segments
)
assert [t.get("lane") for t in titles] == ["1", "2", "3"] * 3
def test_same_lane_titles_never_overlap_in_time(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
words, segments = [], []
for s in range(3):
base = s * 2.0
segments.append({"start": base, "end": base + 1.5})
words += [
{
"word": f"s{s}w{i}",
"start": base + i * 0.4,
"end": base + i * 0.4 + 0.2,
}
for i in range(3)
]
titles = modifier.generate_dynamic_subtitles(
"Interview_A", words, segments=segments
)
by_lane: dict[str, list] = {}
for t in titles:
by_lane.setdefault(t.get("lane"), []).append(t)
for lane_titles in by_lane.values():
spans = sorted(
(
modifier._parse_time(t.get("offset")),
modifier._parse_time(t.get("offset")) + modifier._parse_time(t.get("duration")),
)
for t in lane_titles
)
for (_, end_a), (start_b, _) in zip(spans, spans[1:]):
assert end_a <= start_b
def test_words_of_a_block_accumulate_on_screen(self, temp_fcpxml):
"""Each word appears later than the last but they stay up together —
that is the sentence building up in front of the viewer."""
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
offsets = [modifier._parse_time(t.get("offset")) for t in titles]
assert offsets == sorted(offsets)
assert offsets[0] < offsets[-1], "words must not all start at once"
first_end = offsets[0] + modifier._parse_time(titles[0].get("duration"))
assert first_end > offsets[-1], "the first word must outlast the last"
def test_every_word_of_a_block_clears_at_the_same_instant(self, temp_fcpxml):
"""The whole sentence vanishes at once, rather than word by word."""
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
ends = {
(
modifier._parse_time(t.get("offset"))
+ modifier._parse_time(t.get("duration"))
).to_fcpxml()
for t in titles
}
assert len(ends) == 1, f"expected one shared end, got {ends}"
def test_color_follows_the_style_rhythm(self, temp_fcpxml):
"""Colour comes from the word's place in the sentence's rhythm, which
reproduces the palette of the user's calibration export."""
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
def color_of(title):
run = title.find("text/text-style")
style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style")
return style_def.get("fontColor")
rhythm = WordStyle().rhythm
assert [color_of(t) for t in titles] == [
rhythm[i].color for i in range(len(WORDS))
]
def test_effect_resource_created_once_and_deduped(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
modifier.generate_dynamic_subtitles("Interview_A", WORDS)
modifier.generate_dynamic_subtitles("Broll_City", WORDS)
resources = modifier.root.find(".//resources")
title_effects = [
e for e in resources.findall("effect") if e.get("name") == "Text"
]
assert len(title_effects) == 1
def test_no_duplicate_ids_across_multiple_calls_on_same_modifier(self, temp_fcpxml):
"""Regression test: the handler calls generate_dynamic_subtitles once
per spine clip on the SAME FCPXMLModifier instance. Every
name and id must stay unique document-wide, or FCP
rejects the whole import with 'ID ... already defined'."""
modifier = FCPXMLModifier(temp_fcpxml)
modifier.generate_dynamic_subtitles("Interview_A", WORDS)
modifier.generate_dynamic_subtitles("Broll_City", WORDS)
modifier.generate_dynamic_subtitles("Broll_Studio", WORDS)
all_title_names = [t.get("name") for t in modifier.root.findall(".//title")]
all_style_def_ids = [sd.get("id") for sd in modifier.root.findall(".//text-style-def")]
assert len(all_title_names) == len(set(all_title_names))
assert len(all_style_def_ids) == len(set(all_style_def_ids))
def test_element_param_bypasses_ambiguous_duplicate_name_lookup(self, temp_fcpxml):
"""Regression test for 2026-08-17: after ripple-cut/silence-removal,
every fragment of an originally-named clip keeps the same `name`.
Passing that (now-ambiguous) name resolves via `self.clips` — keyed
by name — to whichever clip was indexed LAST, silently attaching
every clip's captions to one wrong spine element. Passing the
`ET.Element` itself (what the real handler in server.py does) must
attach to the exact clip intended, regardless of duplicate names."""
modifier = FCPXMLModifier(temp_fcpxml)
interview_el = modifier._require_clip("Interview_A")
broll_el = modifier._require_clip("Broll_City")
# Simulate the real-world collision: both clips now share one name.
broll_el.set("name", "Interview_A")
modifier._build_clip_index()
titles_a = modifier.generate_dynamic_subtitles(interview_el, WORDS)
titles_b = modifier.generate_dynamic_subtitles(
broll_el, [{"word": "elsewhere", "start": 0.5, "end": 0.9}]
)
assert titles_a and all(t in list(interview_el) for t in titles_a)
assert titles_b and all(t in list(broll_el) for t in titles_b)
# The two clips' captions must not have landed on the same element.
assert list(interview_el.findall("title")) != list(broll_el.findall("title"))
def test_no_zero_duration_titles_for_tight_word_timing(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
tight_words = [
{"word": "um", "start": 0.0, "end": 0.0},
{"word": "dois", "start": 0.0, "end": 0.0},
]
titles = modifier.generate_dynamic_subtitles("Interview_A", tight_words)
for t in titles:
assert modifier._parse_time(t.get("duration")) > TimeValue.zero()
def test_one_title_per_word_not_per_line(self, temp_fcpxml):
"""Words are never merged into a single title — each is placed
separately so it can appear on its own beat."""
modifier = FCPXMLModifier(temp_fcpxml)
long_words = WORDS + [{"word": "again", "start": 1.3, "end": 1.6}]
titles = modifier.generate_dynamic_subtitles("Interview_A", long_words, WORD_MODE)
assert len(titles) == len(long_words)
assert [t.find("text/text-style").text for t in titles] == [
w["word"] for w in long_words
]
def test_segments_split_words_into_separate_blocks(self, temp_fcpxml):
"""Two Whisper segments must produce two blocks that clear
independently, rather than one run-on block."""
modifier = FCPXMLModifier(temp_fcpxml)
words = [
{"word": "um", "start": 0.0, "end": 0.4},
{"word": "dois", "start": 0.5, "end": 0.9},
{"word": "tres", "start": 5.0, "end": 5.4},
]
segments = [{"start": 0.0, "end": 1.0}, {"start": 5.0, "end": 6.0}]
titles = modifier.generate_dynamic_subtitles(
"Interview_A", words, segments=segments
)
def end_of(t):
return (
modifier._parse_time(t.get("offset"))
+ modifier._parse_time(t.get("duration"))
)
assert end_of(titles[0]) == end_of(titles[1]), "block 1 clears together"
assert end_of(titles[2]) != end_of(titles[0]), "block 2 is independent"
def test_long_sentence_splits_into_sub_blocks(self, temp_fcpxml):
"""A sentence taller than the band is broken up rather than spilling
off screen — so no block exceeds the lanes its words need."""
modifier = FCPXMLModifier(temp_fcpxml)
words = [
{"word": f"palavra{i}", "start": float(i) * 0.2, "end": float(i) * 0.2 + 0.15}
for i in range(30)
]
segments = [{"start": 0.0, "end": 30.0}]
titles = modifier.generate_dynamic_subtitles(
"Interview_A", words, WORD_MODE, segments=segments
)
assert len(titles) == len(words)
# More than one distinct clear-time means the sentence was split.
ends = {
(
modifier._parse_time(t.get("offset"))
+ modifier._parse_time(t.get("duration"))
).to_fcpxml()
for t in titles
}
assert len(ends) > 1
def test_empty_words_returns_empty_list(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
assert modifier.generate_dynamic_subtitles("Interview_A", []) == []
def test_missing_clip_raises(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
with pytest.raises(ValueError):
modifier.generate_dynamic_subtitles("NoSuchClip", WORDS)
def test_round_trip_through_writer_and_parser(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
out_path = str(Path(temp_fcpxml).with_suffix(".out.fcpxml"))
try:
modifier.save(out_path)
tree = ET.parse(out_path)
titles = tree.getroot().findall(".//title")
assert len(titles) == len(WORDS)
parsed = parse_fcpxml(out_path)
assert parsed is not None
finally:
Path(out_path).unlink(missing_ok=True)
def test_animated_offset_uses_source_media_coordinates(self, temp_fcpxml):
"""Regression: anchored titles are positioned in SOURCE media
coordinates — FCP's own output ("exemplo de arquivos.fcpxmld") writes
offset = parent clip's `start` + timeline-relative. Writing a plain
relative offset drops the title to ~0s of the media, before the clip's
in-point, so FCP never shows it ("legendas fora"). The default
animated=True must add the clip's source `start` to every offset."""
modifier = FCPXMLModifier(temp_fcpxml)
parent = modifier._require_clip("Interview_A")
parent.set("start", "226220995/24000s")
titles = modifier.generate_dynamic_subtitles(parent, WORDS)
clip_start = modifier._parse_time("226220995/24000s")
titles = sorted(titles, key=lambda t: modifier._parse_time(t.get("offset")))
for t in titles:
offset = modifier._parse_time(t.get("offset"))
# offset must be >= clip source start (== for the first line at
# relative 0.0, > for later lines) — never a small relative value
# that FCP would read as ~0s of the media.
assert offset >= clip_start
# Relative spacing between lines is preserved (offset - clip_start).
starts = [modifier._parse_time(t.get("offset")) - clip_start for t in titles]
assert starts == sorted(starts)
assert starts[0].to_seconds() >= 0.0
def test_effect_is_basic_text_template(self, temp_fcpxml):
"""Dynamic subtitles must resolve the "Text" (Basic Text) effect uid
and produce roleless titles — the structure of FCP's own export, and
the only template we have verified renders in Final Cut."""
modifier = FCPXMLModifier(temp_fcpxml)
modifier.generate_dynamic_subtitles("Interview_A", WORDS)
resources = modifier.root.find(".//resources")
effect = next(
e for e in resources.findall("effect") if e.get("name") == "Text"
)
assert effect.get("uid") == (
".../Titles.localized/Basic Text.localized/"
"Text.localized/Text.moti"
)
for t in modifier.root.findall(".//title"):
assert t.get("ref") == effect.get("id")
assert t.get("role") is None
assert t.get("start") == "86486400/24000s"
def test_all_configs_use_the_one_text_template(self, temp_fcpxml):
"""animated True/False now both resolve to the single "Text" template —
the earlier Essential Title / Título Básico templates never rendered,
so they were retired."""
modifier = FCPXMLModifier(temp_fcpxml)
modifier.generate_dynamic_subtitles("Interview_A", WORDS)
modifier.generate_dynamic_subtitles(
"Interview_A", WORDS, DynamicSubtitleConfig(animated=False)
)
resources = modifier.root.find(".//resources")
effects = {e.get("name") for e in resources.findall("effect")}
assert effects == {"Text"}
class TestClipBoundaryClamping:
"""A connected title is not trimmed by its parent clip's out-point: FCP
keeps drawing it over the following clip. Words whose Whisper end (or
start) runs past the cut must therefore be clamped, or the last block of
one clip overlaps the first block of the next one on screen."""
def _parent_and_titles(self, path, words):
modifier = FCPXMLModifier(path)
parent = modifier._require_clip("Interview_A")
titles = modifier.generate_dynamic_subtitles(parent, words)
limit = modifier._parse_time(parent.get("duration")).to_seconds()
origin = modifier._parse_time(parent.get("start", "0s")).to_seconds()
spans = [
(
modifier._parse_time(t.get("offset")).to_seconds() - origin,
modifier._parse_time(t.get("duration")).to_seconds(),
)
for t in titles
]
return limit, spans
def test_last_word_end_beyond_clip_is_clamped(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
limit = modifier._parse_time(
modifier._require_clip("Interview_A").get("duration")
).to_seconds()
words = [
{"word": "fim", "start": limit - 0.5, "end": limit + 5.0},
]
clip_limit, spans = self._parent_and_titles(temp_fcpxml, words)
for offset, duration in spans:
assert offset + duration <= clip_limit + 1e-6
def test_word_starting_past_clip_end_is_dropped(self, temp_fcpxml):
"""It could only ever be seen over the NEXT clip's captions."""
modifier = FCPXMLModifier(temp_fcpxml)
limit = modifier._parse_time(
modifier._require_clip("Interview_A").get("duration")
).to_seconds()
words = [
{"word": "ok", "start": 0.2, "end": 0.6},
{"word": "tarde", "start": limit + 2.0, "end": limit + 3.0},
]
clip_limit, spans = self._parent_and_titles(temp_fcpxml, words)
assert len(spans) == 1
for offset, duration in spans:
assert offset < clip_limit
assert offset + duration <= clip_limit + 1e-6
class TestTextStyleIds:
"""Regression for 2026-08-17: is DTD type ID, so it
must be a valid XML Name. Deriving it from the title name (built from the
caption text) emitted ids with spaces, accents and leading digits, and
xmllint rejected the whole document — "Syntax of value for attribute id
of text-style-def is not valid" — so FCP refused the import.
"""
XML_NAME = re.compile(r"^[A-Za-z_][A-Za-z0-9_.\-]*$")
def _style_ids(self, modifier):
return [sd.get("id") for sd in modifier.root.findall(".//text-style-def")]
def test_text_title_id_is_a_valid_xml_name(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
modifier.add_text_title(
"Interview_A", "3 coisas que você precisa saber", duration="1s"
)
ids = self._style_ids(modifier)
assert ids and all(self.XML_NAME.match(i) for i in ids), ids
def test_style_id_matches_its_run_ref(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
title = modifier.add_text_title("Interview_A", "Olá, mundo!", duration="1s")
run = title.find("text/text-style")
assert run.get("ref") == title.find("text-style-def").get("id")
def test_repeated_titles_get_unique_ids(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
for _ in range(3):
modifier.add_text_title("Interview_A", "Olá", duration="1s")
ids = self._style_ids(modifier)
assert len(ids) == len(set(ids)) == 3
def test_id_survives_text_with_no_ascii_characters(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
modifier.add_text_title("Interview_A", "日本語", duration="1s")
ids = self._style_ids(modifier)
assert ids and all(self.XML_NAME.match(i) for i in ids), ids
def test_dynamic_subtitle_ids_are_valid_xml_names(self, temp_fcpxml):
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS)
ids = self._style_ids(modifier)
assert len(ids) == len(titles)
assert all(self.XML_NAME.match(i) for i in ids), ids
class TestProgressiveComposition:
"""The default look (reference reel the user sent, 2026-08-17):
[ que vão ]
[ melhorar ]
[ sua legenda ]
One title per LINE, not per word. Supporting words are grouped and set
small in a grotesque; the sentence's key word is set large in a display
italic on a line of its own. Each block appears as its own words are
spoken and stays on screen, so the sentence assembles itself, and every
block of a composition clears at the same instant.
"""
PHRASE = "que vão melhorar sua legenda".split()
def _words(self, texts=None, step=0.5):
texts = texts or self.PHRASE
return [
{"word": w, "start": i * step, "end": i * step + step * 0.9}
for i, w in enumerate(texts)
]
def _titles(self, temp_fcpxml, words=None):
modifier = FCPXMLModifier(temp_fcpxml)
fmt = modifier.root.find(".//format")
fmt.set("width", "2160")
fmt.set("height", "3840")
words = words or self._words()
segments = [{"start": 0.0, "end": words[-1]["end"]}]
return modifier, modifier.generate_dynamic_subtitles(
"Interview_A", words, segments=segments
)
@staticmethod
def _style(title):
run = title.find("text/text-style")
return title.find(f"text-style-def[@id='{run.get('ref')}']/text-style")
@staticmethod
def _position(title):
param = title.find(f"param[@key='{FCPXMLModifier._TEXT_POSITION_KEY}']")
x, y = param.get("value").split()
return float(x), float(y)
def test_one_title_per_line_grouping_supporting_words(self, temp_fcpxml):
_, titles = self._titles(temp_fcpxml)
assert [t.find("text/text-style").text for t in titles] == [
"que vão",
"melhorar",
"sua legenda",
]
def test_key_word_is_set_large_in_the_display_italic(self, temp_fcpxml):
_, titles = self._titles(temp_fcpxml)
emphasis, body = self._style(titles[1]), self._style(titles[0])
assert emphasis.get("font") == "Playfair Display"
assert "Italic" in emphasis.get("fontFace")
assert body.get("font") == "Helvetica Neue"
assert int(emphasis.get("fontSize")) > 2 * int(body.get("fontSize"))
def test_every_line_is_white(self, temp_fcpxml):
_, titles = self._titles(temp_fcpxml)
assert {self._style(t).get("fontColor") for t in titles} == {"1 1 1 1"}
def test_lines_stack_downward_and_stagger_around_the_key_word(self, temp_fcpxml):
_, titles = self._titles(temp_fcpxml)
(x_top, y_top), (x_mid, y_mid), (x_low, y_low) = map(self._position, titles)
assert y_top > y_mid > y_low, "lines stack top to bottom"
assert x_mid == 0, "the key word stays centred"
assert x_top < 0 < x_low, "supporting lines hang off opposite edges"
def test_blocks_enter_as_spoken_and_clear_together(self, temp_fcpxml):
modifier, titles = self._titles(temp_fcpxml)
starts = [modifier._parse_time(t.get("offset")) for t in titles]
ends = [
s + modifier._parse_time(t.get("duration"))
for s, t in zip(starts, titles)
]
assert starts == sorted(starts) and starts[0] < starts[-1]
assert len({e.to_fcpxml() for e in ends}) == 1, "the composition clears as one"
def test_lines_never_overlap_on_screen(self, temp_fcpxml):
"""Measured on the REAL ink each line contains — the accents and
descenders of a display italic reach far past its cap height, and
that is the pair that touches if the stack is spaced nominally."""
_, titles = self._titles(temp_fcpxml)
boxes = []
for t in titles:
_, y = self._position(t)
style = self._style(t)
top, bottom = ink_extent(
t.find("text/text-style").text,
font_points(style),
font=style.get("font"),
face=style.get("fontFace"),
)
boxes.append((y + bottom, y + top))
for (_, top_above), (bottom_below, _) in zip(boxes, boxes[1:]):
assert bottom_below < top_above # ordered top to bottom
assert top_above <= boxes[0][1]
for i, (lo_a, hi_a) in enumerate(boxes):
for lo_b, hi_b in boxes[i + 1:]:
assert hi_b <= lo_a, "rendered ink must not intersect"
@pytest.mark.parametrize("phrase", [
"que vão melhorar sua legenda",
"o segredo é começar antes de estar pronto",
"ninguém ganha dinheiro gravando vídeo tremido à noite",
"ÇÃO PAGA jjj ggg ppp qqq yyy ÁÊÍÕÚ",
"a b c d e f g h i j k l",
])
def test_ink_never_overlaps_for_any_phrase(self, temp_fcpxml, phrase):
"""The guarantee has to hold for accents stacked over descenders,
for all-caps, and for a run of one-letter words — not just for the
reference sentence."""
modifier, titles = self._titles(temp_fcpxml, self._words(phrase.split()))
# Only lines that share the screen can collide. Every block of one
# composition clears at the same instant, and the next composition
# starts there, so the clear time groups what is on screen together.
onscreen = {}
for t in titles:
_, y = self._position(t)
style = self._style(t)
top, bottom = ink_extent(
t.find("text/text-style").text,
font_points(style),
font=style.get("font"),
face=style.get("fontFace"),
)
clear = (
modifier._parse_time(t.get("offset"))
+ modifier._parse_time(t.get("duration"))
).to_fcpxml()
onscreen.setdefault(clear, []).append((y + bottom, y + top))
assert onscreen, phrase
for boxes in onscreen.values():
for i, (lo_a, hi_a) in enumerate(boxes):
for lo_b, hi_b in boxes[i + 1:]:
assert hi_b <= lo_a or lo_b >= hi_a, phrase
def test_function_words_are_never_the_emphasis(self, temp_fcpxml):
_, titles = self._titles(temp_fcpxml, self._words("e o que importa é constância".split()))
big = max(titles, key=lambda t: int(self._style(t).get("fontSize")))
assert big.find("text/text-style").text == "constância"
def test_each_line_gets_its_own_lane(self, temp_fcpxml):
_, titles = self._titles(temp_fcpxml)
assert [t.get("lane") for t in titles] == ["1", "2", "3"]
def test_single_word_sentence_still_composes(self, temp_fcpxml):
_, titles = self._titles(temp_fcpxml, self._words(["chega"]))
assert len(titles) == 1
assert self._style(titles[0]).get("font") == "Playfair Display"
def test_long_sentence_splits_into_successive_compositions(self, temp_fcpxml):
words = self._words([f"palavra{i}" for i in range(24)], step=0.3)
modifier, titles = self._titles(temp_fcpxml, words)
ends = {
(
modifier._parse_time(t.get("offset"))
+ modifier._parse_time(t.get("duration"))
).to_fcpxml()
for t in titles
}
assert len(ends) > 1, "a long sentence must break into several compositions"
spoken = " ".join(w["word"] for w in words)
emitted = " ".join(t.find("text/text-style").text for t in titles)
assert emitted == spoken, "no spoken word may be dropped or reordered"
class TestTemplateFontScale:
"""The "Text" template sizes type in frame pixels but positions in canvas
points, so the emitted fontSize must be converted or the block renders in
the right place at half the chosen size."""
STYLE = WordStyle(
emphasis_look=WordLook(200, "1 1 1 1", font="Georgia", kerning=3.0),
body_look=WordLook(100, "1 1 1 1", font="Helvetica Neue", kerning=3.0),
)
def _styles(self, titles):
return [t.find(".//text-style-def/text-style") for t in titles]
def _generate(self, path, **kwargs):
modifier = FCPXMLModifier(path)
titles = modifier.generate_dynamic_subtitles(
"Interview_A", WORDS,
DynamicSubtitleConfig(style=self.STYLE, **kwargs),
)
return titles
def test_emitted_size_is_the_layout_size_times_the_template_scale(self, temp_fcpxml):
unscaled = self._generate(temp_fcpxml, text_scale=1.0)
scaled = self._generate(temp_fcpxml, text_scale=TEXT_TEMPLATE_FONT_SCALE)
for plain, big in zip(self._styles(unscaled), self._styles(scaled)):
assert float(big.get("fontSize")) == (
float(plain.get("fontSize")) * TEXT_TEMPLATE_FONT_SCALE
)
def test_default_config_applies_the_template_scale(self, temp_fcpxml):
assert DynamicSubtitleConfig().text_scale == TEXT_TEMPLATE_FONT_SCALE
default = self._styles(self._generate(temp_fcpxml))
unscaled = self._styles(self._generate(temp_fcpxml, text_scale=1.0))
assert [s.get("fontSize") for s in default] != [
s.get("fontSize") for s in unscaled
]
def test_kerning_scales_with_the_font_size(self, temp_fcpxml):
"""Kerning is in font units too — leaving it behind would tighten the
letter spacing to half as the type doubled."""
unscaled = self._styles(self._generate(temp_fcpxml, text_scale=1.0))
scaled = self._styles(self._generate(temp_fcpxml, text_scale=2.0))
for plain, big in zip(unscaled, scaled):
if plain.get("kerning"):
assert float(big.get("kerning")) == float(plain.get("kerning")) * 2
def _positions(self, titles):
return [
tuple(float(v) for v in t.find(
"param[@name='Position']").get("value").split())
for t in titles
]
def test_position_is_converted_with_the_type(self, temp_fcpxml):
"""The template reads fontSize and Position in the SAME space, so the
conversion has to reach both. Scaling only the type leaves the block at
the old spread with twice the type in it, and the lines collide."""
plain = self._generate(temp_fcpxml, text_scale=1.0)
big = self._generate(temp_fcpxml, text_scale=2.0)
for (x, y), (x2, y2) in zip(self._positions(plain), self._positions(big)):
assert (x2, y2) == pytest.approx((x * 2, y * 2), rel=1e-4, abs=0.01)
def test_type_and_spacing_keep_their_ratio_at_any_scale(self, temp_fcpxml):
"""The invariant that broke in the field: the distance between two
lines, measured in font sizes, must not depend on the scale."""
ratios = []
for scale in (1.0, 2.0, 3.5):
titles = self._generate(temp_fcpxml, text_scale=scale)
ys = [y for _, y in self._positions(titles)]
sizes = [float(s.get("fontSize")) for s in self._styles(titles)]
ratios.append([
(a - b) / size
for a, b, size in zip(ys, ys[1:], sizes)
])
for other in ratios[1:]:
assert other == pytest.approx(ratios[0], rel=1e-4, abs=1e-4)
class TestLineGap:
"""The air between stacked lines is a design choice, negative included."""
STYLE = WordStyle(
emphasis_look=WordLook(200, "1 1 1 1", font="Georgia", kerning=0.0),
body_look=WordLook(100, "1 1 1 1", font="Helvetica Neue", kerning=0.0),
)
PHRASE = "eu tinha muita dificuldade de encontrar roupa"
def _blocks(self, gap):
"""Compose in a band tall enough to hold every line, so the gap is the
only thing that changes — a short band would also change how many
lines fit, which is a different effect."""
words = [
{"word": w, "start": i * 0.3, "end": i * 0.3 + 0.3}
for i, w in enumerate(self.PHRASE.split())
]
box = LayoutBox(width=1080 * 0.92, height=100_000, center_y=0)
return compose_sentence(words, self.STYLE, box, line_gap=gap).blocks
def test_default_matches_the_reference_gap(self):
assert DynamicSubtitleConfig().line_gap == REFERENCE_BLOCK_LINE_GAP
def test_each_step_changes_by_exactly_the_gap(self):
"""The stack places ink boxes edge to edge, so the gap is the whole
distance between two lines beyond their own ink."""
zero = [b.y for b in self._blocks(0.0)]
loose = [b.y for b in self._blocks(50.0)]
assert len(zero) == len(loose) >= 2
for plain, spaced in zip(
[a - b for a, b in zip(zero, zero[1:])],
[a - b for a, b in zip(loose, loose[1:])],
):
assert spaced == pytest.approx(plain + 50.0)
def test_a_negative_gap_overlaps_by_exactly_that_much(self):
"""Negative is a supported look, not a failure: -40 tucks each line 40
points into the one above rather than colliding by some amount the
caller cannot predict."""
zero = [b.y for b in self._blocks(0.0)]
tucked = [b.y for b in self._blocks(-40.0)]
assert len(zero) == len(tucked) >= 2
for plain, tight in zip(
[a - b for a, b in zip(zero, zero[1:])],
[a - b for a, b in zip(tucked, tucked[1:])],
):
assert tight == pytest.approx(plain - 40.0)
def test_the_gap_reaches_the_generated_titles(self, temp_fcpxml):
"""The config field has to survive the trip to the XML."""
def spread(gap):
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles(
"Interview_A", WORDS,
DynamicSubtitleConfig(style=self.STYLE, line_gap=gap, text_scale=1.0),
)
ys = [
float(t.find("param[@name='Position']").get("value").split()[1])
for t in titles
]
return max(ys) - min(ys)
assert spread(0.0) < spread(80.0)