"""Tests for cascading progressive-reveal subtitle generation. Covers FCPXMLModifier.generate_dynamic_subtitles(): one standalone ("Essencial - Título"/Essential Title template) per word by default, cycling round-robin across config.lane_count lanes (no compound clip, no <gap> anchoring — both were rejected by a real Final Cut Pro import in earlier iterations), same-lane non-overlap, minimum one-frame durations, effect uid/param fidelity to the FCP exports the user produced by hand, absence of any caption role, and round-trip through the parser. """ import re import shutil import tempfile import xml.etree.ElementTree as ET from pathlib import Path import pytest from fcpxml.models import DynamicSubtitleConfig, TimeValue, WordStyle from fcpxml.parser import parse_fcpxml from fcpxml.text_layout import POINT_SCALE, REFERENCE_CANVAS_HEIGHT, ink_extent from fcpxml.writer import FCPXMLModifier SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml" # The earlier rhythm: one title per WORD. The default is now the # progressive composition (one title per LINE), covered in # TestProgressiveComposition. WORD_MODE = DynamicSubtitleConfig(granularity="word") WORDS = [ {"word": "Hello", "start": 0.0, "end": 0.4}, {"word": "there", "start": 0.4, "end": 0.8}, {"word": "friend", "start": 0.8, "end": 1.3}, ] @pytest.fixture def temp_fcpxml(): with tempfile.NamedTemporaryFile(suffix=".fcpxml", delete=False) as f: shutil.copy(SAMPLE, f.name) yield f.name Path(f.name).unlink(missing_ok=True) class TestGenerateDynamicSubtitles: def test_one_title_per_word_no_compound_clip(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) # Default max_words_per_line=1 -> one title per word. titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE) assert len(titles) == len(WORDS) assert all(t.tag == "title" for t in titles) # No <media>/<ref-clip> compound-clip machinery at all. resources = modifier.root.find(".//resources") assert resources.findall("media") == [] assert modifier.root.findall(".//ref-clip") == [] def test_title_attached_directly_to_parent_clip(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) parent = modifier._require_clip("Interview_A") titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS) assert titles[0] in list(parent) def test_effect_uid_matches_real_fcp_export(self, temp_fcpxml): """An incorrect/guessed effect uid makes FCP silently drop every connected clip that references it during import — no error, the clips just never appear in the timeline or role list. This uid comes verbatim from an export the user made by hand in Final Cut and exported back out ("teste.fcpxmld" and "posição.fcpxmld", effect r3 "Text"), not fabricated.""" modifier = FCPXMLModifier(temp_fcpxml) modifier.generate_dynamic_subtitles("Interview_A", WORDS) resources = modifier.root.find(".//resources") effect = next( e for e in resources.findall("effect") if e.get("name") == "Text" ) assert effect.get("uid") == ( ".../Titles.localized/Basic Text.localized/" "Text.localized/Text.moti" ) def test_titles_carry_no_caption_role(self, temp_fcpxml): """Regression: dynamic subtitles are animated TITLES, not captions. A role="subtitles.*" makes Final Cut route them to the captions lane, which is not drawn over the video unless caption display is enabled — so the import succeeds and nothing ever appears. No generated title may carry a role attribute. """ modifier = FCPXMLModifier(temp_fcpxml) titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS) assert titles assert all(t.get("role") is None for t in titles) static = modifier.generate_dynamic_subtitles( "Broll_City", WORDS, DynamicSubtitleConfig(animated=False) ) assert all(t.get("role") is None for t in static) def test_title_has_start_attribute_and_layout_param_block(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) title = modifier.generate_dynamic_subtitles("Interview_A", WORDS)[0] assert title.get("start") == "86486400/24000s" param_names = [p.get("name") for p in title.findall("param")] assert param_names == [ "Position", "Layout Method", "Left Margin", "Right Margin", "Top Margin", "Bottom Margin", "Alignment", "Line Spacing", "Auto-Shrink", "Alignment", "Opacity", "Speed", "Custom Speed", "Apply Speed", ] # The "Custom Speed" param carries the template's own keyframe # animation (opacity reveal) — absolute nominal times, constant # across every instance, so they replay verbatim. anims = title.findall(".//keyframeAnimation") assert len(anims) == 1 def test_word_text_and_style(self, temp_fcpxml): """Font and face come from the sentence rhythm, whose first entry is the reference export's "Toda": Helvetica Light at 170. The size scales to the frame — the fixture is 1920x1080, a quarter the height of the 2160x3840 timeline the rhythm was calibrated on.""" modifier = FCPXMLModifier(temp_fcpxml) title = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)[0] run = title.find("text/text-style") assert run.text == "Hello" style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style") first = WordStyle().rhythm[0] assert style_def.get("font") == first.font assert style_def.get("fontFace") == first.face scale = (1080 * POINT_SCALE) / REFERENCE_CANVAS_HEIGHT assert style_def.get("fontSize") == str(round(first.font_size * scale)) def test_font_size_scales_with_the_frame(self, temp_fcpxml): """A vertical 2160x3840 timeline must get the reference sizes back unscaled — that is the format the rhythm was calibrated against.""" modifier = FCPXMLModifier(temp_fcpxml) fmt = modifier.root.find(".//format") fmt.set("width", "2160") fmt.set("height", "3840") title = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)[0] run = title.find("text/text-style") style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style") assert style_def.get("fontSize") == str(WordStyle().rhythm[0].font_size) def test_position_is_keyframed_constant_hold(self, temp_fcpxml): """Regression (2026-08-17): position is a STATIC param in the "Text" template. The user's exports ("posição.fcpxmld") write the hand-placed words' coordinates as a plain `value="x y"` with no keyframeAnimation. Wrapping it in keyframes made FCP ignore the param and fall back to the template default, stacking every word at the same spot. So the position param must carry a static value — never keyframes — and each word must get its own.""" modifier = FCPXMLModifier(temp_fcpxml) titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS) for t in titles: param = next( p for p in t.findall("param") if p.get("key") == FCPXMLModifier._TEXT_POSITION_KEY ) # Static value, no keyframeAnimation. assert param.get("value") is not None assert param.find("keyframeAnimation") is None def test_each_word_is_positioned_distinctly(self, temp_fcpxml): """Every word carries its own place in the block, so the template's fixed centre never leaves them stacked on one another.""" modifier = FCPXMLModifier(temp_fcpxml) titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE) positions = [ next( p.get("value") for p in t.findall("param") if p.get("key") == FCPXMLModifier._TEXT_POSITION_KEY ) for t in titles ] assert len(positions) == len(WORDS) assert len(set(positions)) == len(WORDS), "words must not share a spot" def test_no_adjust_transform_on_text_title(self, temp_fcpxml): """The "Text" template positions via its Position param, not <adjust-transform> — FCP's own export carries no adjust-transform.""" modifier = FCPXMLModifier(temp_fcpxml) titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS) assert all(t.find("adjust-transform") is None for t in titles) def test_each_word_of_a_block_gets_its_own_lane(self, temp_fcpxml): """Words of one sentence are all on screen together, so none may share a lane — sharing one would make Final Cut reject the overlap.""" modifier = FCPXMLModifier(temp_fcpxml) titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE) assert [t.get("lane") for t in titles] == ["1", "2", "3"] def test_lanes_restart_with_each_block(self, temp_fcpxml): """The previous block has cleared by the time the next one starts, so its lanes are free to reuse.""" modifier = FCPXMLModifier(temp_fcpxml) words, segments = [], [] for s in range(3): base = s * 2.0 segments.append({"start": base, "end": base + 1.5}) words += [ { "word": f"s{s}w{i}", "start": base + i * 0.4, "end": base + i * 0.4 + 0.2, } for i in range(3) ] titles = modifier.generate_dynamic_subtitles( "Interview_A", words, segments=segments ) assert [t.get("lane") for t in titles] == ["1", "2", "3"] * 3 def test_same_lane_titles_never_overlap_in_time(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) words, segments = [], [] for s in range(3): base = s * 2.0 segments.append({"start": base, "end": base + 1.5}) words += [ { "word": f"s{s}w{i}", "start": base + i * 0.4, "end": base + i * 0.4 + 0.2, } for i in range(3) ] titles = modifier.generate_dynamic_subtitles( "Interview_A", words, segments=segments ) by_lane: dict[str, list] = {} for t in titles: by_lane.setdefault(t.get("lane"), []).append(t) for lane_titles in by_lane.values(): spans = sorted( ( modifier._parse_time(t.get("offset")), modifier._parse_time(t.get("offset")) + modifier._parse_time(t.get("duration")), ) for t in lane_titles ) for (_, end_a), (start_b, _) in zip(spans, spans[1:]): assert end_a <= start_b def test_words_of_a_block_accumulate_on_screen(self, temp_fcpxml): """Each word appears later than the last but they stay up together — that is the sentence building up in front of the viewer.""" modifier = FCPXMLModifier(temp_fcpxml) titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS) offsets = [modifier._parse_time(t.get("offset")) for t in titles] assert offsets == sorted(offsets) assert offsets[0] < offsets[-1], "words must not all start at once" first_end = offsets[0] + modifier._parse_time(titles[0].get("duration")) assert first_end > offsets[-1], "the first word must outlast the last" def test_every_word_of_a_block_clears_at_the_same_instant(self, temp_fcpxml): """The whole sentence vanishes at once, rather than word by word.""" modifier = FCPXMLModifier(temp_fcpxml) titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS) ends = { ( modifier._parse_time(t.get("offset")) + modifier._parse_time(t.get("duration")) ).to_fcpxml() for t in titles } assert len(ends) == 1, f"expected one shared end, got {ends}" def test_color_follows_the_style_rhythm(self, temp_fcpxml): """Colour comes from the word's place in the sentence's rhythm, which reproduces the palette of the user's calibration export.""" modifier = FCPXMLModifier(temp_fcpxml) titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE) def color_of(title): run = title.find("text/text-style") style_def = title.find(f"text-style-def[@id='{run.get('ref')}']/text-style") return style_def.get("fontColor") rhythm = WordStyle().rhythm assert [color_of(t) for t in titles] == [ rhythm[i].color for i in range(len(WORDS)) ] def test_effect_resource_created_once_and_deduped(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) modifier.generate_dynamic_subtitles("Interview_A", WORDS) modifier.generate_dynamic_subtitles("Broll_City", WORDS) resources = modifier.root.find(".//resources") title_effects = [ e for e in resources.findall("effect") if e.get("name") == "Text" ] assert len(title_effects) == 1 def test_no_duplicate_ids_across_multiple_calls_on_same_modifier(self, temp_fcpxml): """Regression test: the handler calls generate_dynamic_subtitles once per spine clip on the SAME FCPXMLModifier instance. Every <title> name and <text-style-def> id must stay unique document-wide, or FCP rejects the whole import with 'ID ... already defined'.""" modifier = FCPXMLModifier(temp_fcpxml) modifier.generate_dynamic_subtitles("Interview_A", WORDS) modifier.generate_dynamic_subtitles("Broll_City", WORDS) modifier.generate_dynamic_subtitles("Broll_Studio", WORDS) all_title_names = [t.get("name") for t in modifier.root.findall(".//title")] all_style_def_ids = [sd.get("id") for sd in modifier.root.findall(".//text-style-def")] assert len(all_title_names) == len(set(all_title_names)) assert len(all_style_def_ids) == len(set(all_style_def_ids)) def test_element_param_bypasses_ambiguous_duplicate_name_lookup(self, temp_fcpxml): """Regression test for 2026-08-17: after ripple-cut/silence-removal, every fragment of an originally-named clip keeps the same `name`. Passing that (now-ambiguous) name resolves via `self.clips` — keyed by name — to whichever clip was indexed LAST, silently attaching every clip's captions to one wrong spine element. Passing the `ET.Element` itself (what the real handler in server.py does) must attach to the exact clip intended, regardless of duplicate names.""" modifier = FCPXMLModifier(temp_fcpxml) interview_el = modifier._require_clip("Interview_A") broll_el = modifier._require_clip("Broll_City") # Simulate the real-world collision: both clips now share one name. broll_el.set("name", "Interview_A") modifier._build_clip_index() titles_a = modifier.generate_dynamic_subtitles(interview_el, WORDS) titles_b = modifier.generate_dynamic_subtitles( broll_el, [{"word": "elsewhere", "start": 0.5, "end": 0.9}] ) assert titles_a and all(t in list(interview_el) for t in titles_a) assert titles_b and all(t in list(broll_el) for t in titles_b) # The two clips' captions must not have landed on the same element. assert list(interview_el.findall("title")) != list(broll_el.findall("title")) def test_no_zero_duration_titles_for_tight_word_timing(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) tight_words = [ {"word": "um", "start": 0.0, "end": 0.0}, {"word": "dois", "start": 0.0, "end": 0.0}, ] titles = modifier.generate_dynamic_subtitles("Interview_A", tight_words) for t in titles: assert modifier._parse_time(t.get("duration")) > TimeValue.zero() def test_one_title_per_word_not_per_line(self, temp_fcpxml): """Words are never merged into a single title — each is placed separately so it can appear on its own beat.""" modifier = FCPXMLModifier(temp_fcpxml) long_words = WORDS + [{"word": "again", "start": 1.3, "end": 1.6}] titles = modifier.generate_dynamic_subtitles("Interview_A", long_words, WORD_MODE) assert len(titles) == len(long_words) assert [t.find("text/text-style").text for t in titles] == [ w["word"] for w in long_words ] def test_segments_split_words_into_separate_blocks(self, temp_fcpxml): """Two Whisper segments must produce two blocks that clear independently, rather than one run-on block.""" modifier = FCPXMLModifier(temp_fcpxml) words = [ {"word": "um", "start": 0.0, "end": 0.4}, {"word": "dois", "start": 0.5, "end": 0.9}, {"word": "tres", "start": 5.0, "end": 5.4}, ] segments = [{"start": 0.0, "end": 1.0}, {"start": 5.0, "end": 6.0}] titles = modifier.generate_dynamic_subtitles( "Interview_A", words, segments=segments ) def end_of(t): return ( modifier._parse_time(t.get("offset")) + modifier._parse_time(t.get("duration")) ) assert end_of(titles[0]) == end_of(titles[1]), "block 1 clears together" assert end_of(titles[2]) != end_of(titles[0]), "block 2 is independent" def test_long_sentence_splits_into_sub_blocks(self, temp_fcpxml): """A sentence taller than the band is broken up rather than spilling off screen — so no block exceeds the lanes its words need.""" modifier = FCPXMLModifier(temp_fcpxml) words = [ {"word": f"palavra{i}", "start": float(i) * 0.2, "end": float(i) * 0.2 + 0.15} for i in range(30) ] segments = [{"start": 0.0, "end": 30.0}] titles = modifier.generate_dynamic_subtitles( "Interview_A", words, WORD_MODE, segments=segments ) assert len(titles) == len(words) # More than one distinct clear-time means the sentence was split. ends = { ( modifier._parse_time(t.get("offset")) + modifier._parse_time(t.get("duration")) ).to_fcpxml() for t in titles } assert len(ends) > 1 def test_empty_words_returns_empty_list(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) assert modifier.generate_dynamic_subtitles("Interview_A", []) == [] def test_missing_clip_raises(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) with pytest.raises(ValueError): modifier.generate_dynamic_subtitles("NoSuchClip", WORDS) def test_round_trip_through_writer_and_parser(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE) out_path = str(Path(temp_fcpxml).with_suffix(".out.fcpxml")) try: modifier.save(out_path) tree = ET.parse(out_path) titles = tree.getroot().findall(".//title") assert len(titles) == len(WORDS) parsed = parse_fcpxml(out_path) assert parsed is not None finally: Path(out_path).unlink(missing_ok=True) def test_animated_offset_uses_source_media_coordinates(self, temp_fcpxml): """Regression: anchored titles are positioned in SOURCE media coordinates — FCP's own output ("exemplo de arquivos.fcpxmld") writes offset = parent clip's `start` + timeline-relative. Writing a plain relative offset drops the title to ~0s of the media, before the clip's in-point, so FCP never shows it ("legendas fora"). The default animated=True must add the clip's source `start` to every offset.""" modifier = FCPXMLModifier(temp_fcpxml) parent = modifier._require_clip("Interview_A") parent.set("start", "226220995/24000s") titles = modifier.generate_dynamic_subtitles(parent, WORDS) clip_start = modifier._parse_time("226220995/24000s") titles = sorted(titles, key=lambda t: modifier._parse_time(t.get("offset"))) for t in titles: offset = modifier._parse_time(t.get("offset")) # offset must be >= clip source start (== for the first line at # relative 0.0, > for later lines) — never a small relative value # that FCP would read as ~0s of the media. assert offset >= clip_start # Relative spacing between lines is preserved (offset - clip_start). starts = [modifier._parse_time(t.get("offset")) - clip_start for t in titles] assert starts == sorted(starts) assert starts[0].to_seconds() >= 0.0 def test_effect_is_basic_text_template(self, temp_fcpxml): """Dynamic subtitles must resolve the "Text" (Basic Text) effect uid and produce roleless titles — the structure of FCP's own export, and the only template we have verified renders in Final Cut.""" modifier = FCPXMLModifier(temp_fcpxml) modifier.generate_dynamic_subtitles("Interview_A", WORDS) resources = modifier.root.find(".//resources") effect = next( e for e in resources.findall("effect") if e.get("name") == "Text" ) assert effect.get("uid") == ( ".../Titles.localized/Basic Text.localized/" "Text.localized/Text.moti" ) for t in modifier.root.findall(".//title"): assert t.get("ref") == effect.get("id") assert t.get("role") is None assert t.get("start") == "86486400/24000s" def test_all_configs_use_the_one_text_template(self, temp_fcpxml): """animated True/False now both resolve to the single "Text" template — the earlier Essential Title / Título Básico templates never rendered, so they were retired.""" modifier = FCPXMLModifier(temp_fcpxml) modifier.generate_dynamic_subtitles("Interview_A", WORDS) modifier.generate_dynamic_subtitles( "Interview_A", WORDS, DynamicSubtitleConfig(animated=False) ) resources = modifier.root.find(".//resources") effects = {e.get("name") for e in resources.findall("effect")} assert effects == {"Text"} class TestClipBoundaryClamping: """A connected title is not trimmed by its parent clip's out-point: FCP keeps drawing it over the following clip. Words whose Whisper end (or start) runs past the cut must therefore be clamped, or the last block of one clip overlaps the first block of the next one on screen.""" def _parent_and_titles(self, path, words): modifier = FCPXMLModifier(path) parent = modifier._require_clip("Interview_A") titles = modifier.generate_dynamic_subtitles(parent, words) limit = modifier._parse_time(parent.get("duration")).to_seconds() origin = modifier._parse_time(parent.get("start", "0s")).to_seconds() spans = [ ( modifier._parse_time(t.get("offset")).to_seconds() - origin, modifier._parse_time(t.get("duration")).to_seconds(), ) for t in titles ] return limit, spans def test_last_word_end_beyond_clip_is_clamped(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) limit = modifier._parse_time( modifier._require_clip("Interview_A").get("duration") ).to_seconds() words = [ {"word": "fim", "start": limit - 0.5, "end": limit + 5.0}, ] clip_limit, spans = self._parent_and_titles(temp_fcpxml, words) for offset, duration in spans: assert offset + duration <= clip_limit + 1e-6 def test_word_starting_past_clip_end_is_dropped(self, temp_fcpxml): """It could only ever be seen over the NEXT clip's captions.""" modifier = FCPXMLModifier(temp_fcpxml) limit = modifier._parse_time( modifier._require_clip("Interview_A").get("duration") ).to_seconds() words = [ {"word": "ok", "start": 0.2, "end": 0.6}, {"word": "tarde", "start": limit + 2.0, "end": limit + 3.0}, ] clip_limit, spans = self._parent_and_titles(temp_fcpxml, words) assert len(spans) == 1 for offset, duration in spans: assert offset < clip_limit assert offset + duration <= clip_limit + 1e-6 class TestTextStyleIds: """Regression for 2026-08-17: <text-style-def id> is DTD type ID, so it must be a valid XML Name. Deriving it from the title name (built from the caption text) emitted ids with spaces, accents and leading digits, and xmllint rejected the whole document — "Syntax of value for attribute id of text-style-def is not valid" — so FCP refused the import. """ XML_NAME = re.compile(r"^[A-Za-z_][A-Za-z0-9_.\-]*$") def _style_ids(self, modifier): return [sd.get("id") for sd in modifier.root.findall(".//text-style-def")] def test_text_title_id_is_a_valid_xml_name(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) modifier.add_text_title( "Interview_A", "3 coisas que você precisa saber", duration="1s" ) ids = self._style_ids(modifier) assert ids and all(self.XML_NAME.match(i) for i in ids), ids def test_style_id_matches_its_run_ref(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) title = modifier.add_text_title("Interview_A", "Olá, mundo!", duration="1s") run = title.find("text/text-style") assert run.get("ref") == title.find("text-style-def").get("id") def test_repeated_titles_get_unique_ids(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) for _ in range(3): modifier.add_text_title("Interview_A", "Olá", duration="1s") ids = self._style_ids(modifier) assert len(ids) == len(set(ids)) == 3 def test_id_survives_text_with_no_ascii_characters(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) modifier.add_text_title("Interview_A", "日本語", duration="1s") ids = self._style_ids(modifier) assert ids and all(self.XML_NAME.match(i) for i in ids), ids def test_dynamic_subtitle_ids_are_valid_xml_names(self, temp_fcpxml): modifier = FCPXMLModifier(temp_fcpxml) titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS) ids = self._style_ids(modifier) assert len(ids) == len(titles) assert all(self.XML_NAME.match(i) for i in ids), ids class TestProgressiveComposition: """The default look (reference reel the user sent, 2026-08-17): [ que vão ] [ melhorar ] [ sua legenda ] One title per LINE, not per word. Supporting words are grouped and set small in a grotesque; the sentence's key word is set large in a display italic on a line of its own. Each block appears as its own words are spoken and stays on screen, so the sentence assembles itself, and every block of a composition clears at the same instant. """ PHRASE = "que vão melhorar sua legenda".split() def _words(self, texts=None, step=0.5): texts = texts or self.PHRASE return [ {"word": w, "start": i * step, "end": i * step + step * 0.9} for i, w in enumerate(texts) ] def _titles(self, temp_fcpxml, words=None): modifier = FCPXMLModifier(temp_fcpxml) fmt = modifier.root.find(".//format") fmt.set("width", "2160") fmt.set("height", "3840") words = words or self._words() segments = [{"start": 0.0, "end": words[-1]["end"]}] return modifier, modifier.generate_dynamic_subtitles( "Interview_A", words, segments=segments ) @staticmethod def _style(title): run = title.find("text/text-style") return title.find(f"text-style-def[@id='{run.get('ref')}']/text-style") @staticmethod def _position(title): param = title.find(f"param[@key='{FCPXMLModifier._TEXT_POSITION_KEY}']") x, y = param.get("value").split() return float(x), float(y) def test_one_title_per_line_grouping_supporting_words(self, temp_fcpxml): _, titles = self._titles(temp_fcpxml) assert [t.find("text/text-style").text for t in titles] == [ "que vão", "melhorar", "sua legenda", ] def test_key_word_is_set_large_in_the_display_italic(self, temp_fcpxml): _, titles = self._titles(temp_fcpxml) emphasis, body = self._style(titles[1]), self._style(titles[0]) assert emphasis.get("font") == "Playfair Display" assert "Italic" in emphasis.get("fontFace") assert body.get("font") == "Helvetica Neue" assert int(emphasis.get("fontSize")) > 2 * int(body.get("fontSize")) def test_every_line_is_white(self, temp_fcpxml): _, titles = self._titles(temp_fcpxml) assert {self._style(t).get("fontColor") for t in titles} == {"1 1 1 1"} def test_lines_stack_downward_and_stagger_around_the_key_word(self, temp_fcpxml): _, titles = self._titles(temp_fcpxml) (x_top, y_top), (x_mid, y_mid), (x_low, y_low) = map(self._position, titles) assert y_top > y_mid > y_low, "lines stack top to bottom" assert x_mid == 0, "the key word stays centred" assert x_top < 0 < x_low, "supporting lines hang off opposite edges" def test_blocks_enter_as_spoken_and_clear_together(self, temp_fcpxml): modifier, titles = self._titles(temp_fcpxml) starts = [modifier._parse_time(t.get("offset")) for t in titles] ends = [ s + modifier._parse_time(t.get("duration")) for s, t in zip(starts, titles) ] assert starts == sorted(starts) and starts[0] < starts[-1] assert len({e.to_fcpxml() for e in ends}) == 1, "the composition clears as one" def test_lines_never_overlap_on_screen(self, temp_fcpxml): """Measured on the REAL ink each line contains — the accents and descenders of a display italic reach far past its cap height, and that is the pair that touches if the stack is spaced nominally.""" _, titles = self._titles(temp_fcpxml) boxes = [] for t in titles: _, y = self._position(t) style = self._style(t) top, bottom = ink_extent( t.find("text/text-style").text, float(style.get("fontSize")), font=style.get("font"), face=style.get("fontFace"), ) boxes.append((y + bottom, y + top)) for (_, top_above), (bottom_below, _) in zip(boxes, boxes[1:]): assert bottom_below < top_above # ordered top to bottom assert top_above <= boxes[0][1] for i, (lo_a, hi_a) in enumerate(boxes): for lo_b, hi_b in boxes[i + 1:]: assert hi_b <= lo_a, "rendered ink must not intersect" @pytest.mark.parametrize("phrase", [ "que vão melhorar sua legenda", "o segredo é começar antes de estar pronto", "ninguém ganha dinheiro gravando vídeo tremido à noite", "ÇÃO PAGA jjj ggg ppp qqq yyy ÁÊÍÕÚ", "a b c d e f g h i j k l", ]) def test_ink_never_overlaps_for_any_phrase(self, temp_fcpxml, phrase): """The guarantee has to hold for accents stacked over descenders, for all-caps, and for a run of one-letter words — not just for the reference sentence.""" modifier, titles = self._titles(temp_fcpxml, self._words(phrase.split())) # Only lines that share the screen can collide. Every block of one # composition clears at the same instant, and the next composition # starts there, so the clear time groups what is on screen together. onscreen = {} for t in titles: _, y = self._position(t) style = self._style(t) top, bottom = ink_extent( t.find("text/text-style").text, float(style.get("fontSize")), font=style.get("font"), face=style.get("fontFace"), ) clear = ( modifier._parse_time(t.get("offset")) + modifier._parse_time(t.get("duration")) ).to_fcpxml() onscreen.setdefault(clear, []).append((y + bottom, y + top)) assert onscreen, phrase for boxes in onscreen.values(): for i, (lo_a, hi_a) in enumerate(boxes): for lo_b, hi_b in boxes[i + 1:]: assert hi_b <= lo_a or lo_b >= hi_a, phrase def test_function_words_are_never_the_emphasis(self, temp_fcpxml): _, titles = self._titles(temp_fcpxml, self._words("e o que importa é constância".split())) big = max(titles, key=lambda t: int(self._style(t).get("fontSize"))) assert big.find("text/text-style").text == "constância" def test_each_line_gets_its_own_lane(self, temp_fcpxml): _, titles = self._titles(temp_fcpxml) assert [t.get("lane") for t in titles] == ["1", "2", "3"] def test_single_word_sentence_still_composes(self, temp_fcpxml): _, titles = self._titles(temp_fcpxml, self._words(["chega"])) assert len(titles) == 1 assert self._style(titles[0]).get("font") == "Playfair Display" def test_long_sentence_splits_into_successive_compositions(self, temp_fcpxml): words = self._words([f"palavra{i}" for i in range(24)], step=0.3) modifier, titles = self._titles(temp_fcpxml, words) ends = { ( modifier._parse_time(t.get("offset")) + modifier._parse_time(t.get("duration")) ).to_fcpxml() for t in titles } assert len(ends) > 1, "a long sentence must break into several compositions" spoken = " ".join(w["word"] for w in words) emitted = " ".join(t.find("text/text-style").text for t in titles) assert emitted == spoken, "no spoken word may be dropped or reordered"