"""Títulos de texto e legendas dinâmicas. Extraído de writer.py — ver fcpxml/writer/__init__.py para o conjunto. """ import random import re import unicodedata import uuid import xml.etree.ElementTree as ET from typing import Any, Dict, List, Optional from ..collision import blocking, validate_titles from ..models import ( DynamicSubtitleConfig, TimeValue, ) from ..text_layout import ( TEXT_TEMPLATE_FONT_SCALE, LayoutBox, compose_sentence, layout_sentence, ) from ..transcribe import group_words_by_segment, split_into_subphrases from .helpers import _dtd_insert, _sanitize_xml_value class TitlesMixin: """Títulos de texto e legendas dinâmicas.""" _SUBTITLE_METADATA_KEY = 'com.gart.subtitle.kind' def mark_generated_subtitle(self, element: ET.Element, kind: str) -> None: metadata = element.find('metadata') if metadata is None: metadata = ET.Element('metadata') _dtd_insert(element, metadata) ET.SubElement(metadata, 'md', key=self._SUBTITLE_METADATA_KEY, value=kind) def _generated_subtitle_kind(self, element: ET.Element) -> Optional[str]: marker = element.find(f"metadata/md[@key='{self._SUBTITLE_METADATA_KEY}']") if marker is not None: return marker.get('value') # Recognize the exact signature of older G-ART exports. A role alone # is not ownership: users also assign these roles to manual titles. if element.tag == 'title': if element.get('start') != self._TEXT_TITLE_START: return None effect = self.root.find(f".//resources/effect[@id='{element.get('ref')}']") if effect is None or effect.get('uid') != self._TEXT_TITLE_UID: return None if re.fullmatch(r'caption_[0-9a-f]{8}', element.get('name', '')): return 'dynamic' text = ''.join(element.findtext('text/text-style', '')) if (element.get('role') == 'titles.convencionais' and element.get('lane') == '20' and element.get('name') == f'{text} - Text'): return 'plain' elif element.tag == 'ref-clip': media = self.root.find(f".//resources/media[@id='{element.get('ref')}']") if media is not None: titles = media.findall('.//title') if titles and all(self._generated_subtitle_kind(t) == 'dynamic' for t in titles): return 'dynamic' return None def remove_generated_subtitles(self, parent: ET.Element, kinds: tuple) -> None: """Replace only our own captions, preserving unrelated graphics.""" resources = self.root.find('.//resources') for child in list(parent): if self._generated_subtitle_kind(child) not in kinds: continue parent.remove(child) if child.tag == 'ref-clip' and resources is not None: ref = child.get('ref') if not self.root.findall(f".//ref-clip[@ref='{ref}']"): media = resources.find(f"media[@id='{ref}']") if media is not None: resources.remove(media) def suppress_plain_under_dynamic(self, parent: ET.Element) -> None: """Keep generated plain titles only on frames without dynamic text.""" import copy windows = [] for child in parent: if self._generated_subtitle_kind(child) == 'dynamic': start = self._parse_time(child.get('offset', '0s')) windows.append((start, start + self._parse_time(child.get('duration', '0s')))) for title in list(parent): if self._generated_subtitle_kind(title) != 'plain': continue start = self._parse_time(title.get('offset', '0s')) end = start + self._parse_time(title.get('duration', '0s')) remaining = [(start, end)] for lo, hi in windows: parts = [] for a, b in remaining: if a < hi and lo < b: if a < lo: parts.append((a, lo)) if hi < b: parts.append((hi, b)) else: parts.append((a, b)) remaining = parts if remaining == [(start, end)]: continue parent.remove(title) for a, b in remaining: part = copy.deepcopy(title) self._reassign_text_style_ids(part) part.set('offset', a.to_fcpxml()) part.set('duration', (b - a).to_fcpxml()) _dtd_insert(parent, part) # DYNAMIC (KARAOKE-STYLE) SUBTITLES # ======================================================================== # The "Text" (Basic Text) template — the ONLY simple title template that # Final Cut actually renders. Copied verbatim from the user's own FCP # exports ("teste.fcpxmld" and "posição.fcpxmld", FCP 1.14 in English): # a single "" run, one "", and a fixed param block # with the margins/alignment/speed the template ships with. Every prior # title template we generated ("Essencial - Título", "Título Básico") # imported cleanly but never appeared — their Motion uids did not resolve # to a real, drawable template in FCP, which discards the clip silently. # "Text" is what FCP itself writes when the user adds a title by hand, so # it is the ground truth. See Engine/docs/05_EXPERIENCIAS.md, 2026-08-17. _TEXT_TITLE_UID = ( '.../Titles.localized/Basic Text.localized/' 'Text.localized/Text.moti' ) _TEXT_TITLE_START = '86486400/24000s' # The Inspector's Position field, and the one this code overrides per # title so two titles never stack on top of each other. Verified in # "posição.fcpxmld": each hand-dragged title carries a distinct "x y" # value here while every other param stays identical. _TEXT_POSITION_KEY = '9999/10003/13260/3296672360/1/100/101' # Layout params the "Text" template ships with. These keys are the # template's own defaults and never vary between instances. # # "Build Out" is the one deliberate override: with "Apply Speed" set to # "2 (Per Object)" below, the template's whole built-in animation (build # in + build out) is always compressed to exactly fill the title's own # on-screen duration — so on a short word-length clip, build out was # eating time that build in needed to finish revealing the text before # the cut. Disabling build out hands that entire compressed window to # build in alone, which is what "sempre acelerado" turned out to mean: # no separate speed knob needed. Value captured from a real FCP export # with "Build Out" unchecked in the Inspector (see chat, 2026-08-18). _TEXT_TITLE_PARAMS = ( ('Build Out', '9999/10000/2/102', '0'), ('Layout Method', '9999/10003/13260/3296672360/2/314', '1 (Paragraph)'), ('Left Margin', '9999/10003/13260/3296672360/2/323', '-1210'), ('Right Margin', '9999/10003/13260/3296672360/2/324', '1210'), ('Top Margin', '9999/10003/13260/3296672360/2/325', '2160'), ('Bottom Margin', '9999/10003/13260/3296672360/2/326', '-2160'), ('Alignment', '9999/10003/13260/3296672360/2/354/3296667315/401', '1 (Center)'), ('Line Spacing', '9999/10003/13260/3296672360/2/354/3296667315/404', '-19'), ('Auto-Shrink', '9999/10003/13260/3296672360/2/370', '3 (To All Margins)'), ('Alignment', '9999/10003/13260/3296672360/2/373', '0 (Left) 1 (Middle)'), ('Opacity', '9999/10003/13260/3296672360/4/3296673134/1000/1044', '0'), ('Speed', '9999/10003/13260/3296672360/4/3296673134/201/208', '6 (Custom)'), ('Apply Speed', '9999/10003/13260/3296672360/4/3296673134/201/211', '2 (Per Object)'), ) # "Custom Speed" sits between "Speed" and "Apply Speed" and carries a # child rather than a plain value attribute. Its two # keyframes are the template's own absolute nominal times, constant across # every instance, so they are safe to replay verbatim. _TEXT_CUSTOM_SPEED_KEY = '9999/10003/13260/3296672360/4/3296673134/201/209' _TEXT_CUSTOM_SPEED_KEYFRAMES = ( ('-469658744/1000000000s', '0'), ('12328542033/1000000000s', '1'), ) _TEXT_SIZE_KEY = '9999/10003/13260/3296672360/5/3296672362/3' def _ensure_text_title_effect(self, resources: ET.Element) -> str: """Return the resource id of the "Text" (Basic Text) effect, creating it if absent.""" return self._ensure_effect(resources, self._TEXT_TITLE_UID, 'Text', 'r_text') def _ensure_effect( self, resources: ET.Element, uid: str, name: str, id_prefix: str, ) -> str: """Return the id of the effect resource with *uid*, creating it if absent.""" for eff in resources.findall('effect'): if eff.get('uid') == uid: return eff.get('id') effect_id = self._unique_resource_id(resources, id_prefix) eff_el = ET.SubElement(resources, 'effect') eff_el.set('id', effect_id) eff_el.set('name', name) eff_el.set('uid', uid) return effect_id # / are DTD type ID/IDREF, so the # value must be a valid XML Name: letters, digits, "_", "-", "." only, # never starting with a digit. Title names are built from the caption # text ("Olá mundo - Text"), which carries spaces, accents and often a # leading digit — xmllint rejected the whole document with "Syntax of # value for attribute id of text-style-def is not valid". _TEXT_STYLE_ID_UNSAFE = re.compile(r'[^A-Za-z0-9_.-]+') def _unique_text_style_id(self, base: str) -> str: """Return a document-unique, DTD-valid XML ID for a ````.""" folded = unicodedata.normalize('NFKD', base).encode('ascii', 'ignore').decode('ascii') slug = self._TEXT_STYLE_ID_UNSAFE.sub('_', folded).strip('_.-')[:48] stem = f"ts_{slug}" if slug else "ts" if self._text_style_ids is None: self._text_style_ids = { sd.get('id') for sd in self.root.findall('.//text-style-def') } candidate = f"{stem}_0" counter = 0 while candidate in self._text_style_ids: counter += 1 candidate = f"{stem}_{counter}" self._text_style_ids.add(candidate) return candidate def _reassign_text_style_ids(self, clip: ET.Element) -> None: """Give every ```` inside a just-deepcopy'd *clip* a fresh document-unique id, repointing any ```` in the same subtree that pointed at the old one. ``split_clip``/``cut_clip_ranges`` deepcopy the clip once per resulting segment, so a clip carrying a ```` (from a "text" voice action) keeps the exact same ``text-style-def id`` in every copy. A single cut is harmless — but the batch chain re-cuts the same clip at each step (silence removal, filler removal, dynamic subtitles), and every pass multiplies the duplicate, so the DTD validator eventually rejects the file with "ID ... already defined". Regenerating here, at the only place copies are made, fixes it for every caller instead of each one having to remember to. """ for style_def in clip.findall('.//text-style-def'): old_id = style_def.get('id') if not old_id: continue slug = old_id[3:] if old_id.startswith('ts_') else old_id slug = re.sub(r'_\d+$', '', slug) # drop a prior _<N> counter new_id = self._unique_text_style_id(slug) if new_id == old_id: continue style_def.set('id', new_id) for ref_el in clip.findall(f".//text-style[@ref='{old_id}']"): ref_el.set('ref', new_id) def _unique_tracking_shape_id(self, base: str) -> str: """Return a document-unique ``id`` for a ``<tracking-shape>``.""" stem = base or "tr" if self._tracking_shape_ids is None: self._tracking_shape_ids = { ts.get('id') for ts in self.root.findall('.//tracking-shape') } candidate = f"{stem}_0" counter = 0 while candidate in self._tracking_shape_ids: counter += 1 candidate = f"{stem}_{counter}" self._tracking_shape_ids.add(candidate) return candidate def _reassign_tracking_shape_ids(self, clip: ET.Element) -> None: """Give every ``<tracking-shape>`` inside a just-deepcopy'd *clip* a fresh document-unique id. Same mechanism as ``_reassign_text_style_ids``: ``split_clip``/ ``cut_clip_ranges`` deepcopy the clip once per resulting segment, so Cinematic object-tracking data (``<object-tracker><tracking-shape id="tr1">``, preserved from the source asset's sidecar) keeps the exact same id in every copy. A single cut is harmless — but the batch chain re-cuts the same clip at each step, multiplying the duplicate until the DTD validator rejects the file with "ID tr1 already defined". """ for shape in clip.findall('.//tracking-shape'): old_id = shape.get('id') if not old_id: continue base = re.sub(r'_\d+$', '', old_id) new_id = self._unique_tracking_shape_id(base) if new_id == old_id: continue shape.set('id', new_id) def _make_text_title_clip( self, effect_id: str, text: str, offset: 'TimeValue', duration: 'TimeValue', *, lane: int, name: str, position: Optional[str] = None, font: str = 'Helvetica Neue', font_size: int = 196, font_color: str = '1 1 1 1', bold: bool = True, face: Optional[str] = None, kerning: Optional[float] = None, font_scale: float = TEXT_TEMPLATE_FONT_SCALE, animated: bool = True, size_param: Optional[float] = None, role: Optional[str] = None, ) -> ET.Element: """Build a standalone ``<title>`` clip from the "Text" (Basic Text) template. Reproduces FCP's own output for a hand-added title exactly — the only template we have verified renders in Final Cut ("teste.fcpxmld" and "posição.fcpxmld"). *position* ("x y" canvas points) is the Inspector Position value; omit it to keep the template's centred default. Unlike the animated templates, this carries no animation switch, so the text stays put and visible for its whole duration. """ elem = ET.Element('title') elem.set('ref', effect_id) elem.set('lane', str(lane)) elem.set('offset', offset.to_fcpxml()) elem.set('name', _sanitize_xml_value(name, 256)) elem.set('start', self._TEXT_TITLE_START) elem.set('duration', duration.to_fcpxml()) if role: elem.set('role', _sanitize_xml_value(role, 256)) if position: param = ET.SubElement(elem, 'param') param.set('name', 'Position') param.set('key', self._TEXT_POSITION_KEY) param.set('value', position) def _add_param(name: str, key: str, value: str) -> None: param = ET.SubElement(elem, 'param') param.set('name', name) param.set('key', key) param.set('value', value) animation_params = {'Opacity', 'Speed', 'Apply Speed'} for param_name, param_key, param_value in self._TEXT_TITLE_PARAMS: if not animated and param_name in animation_params: continue _add_param(param_name, param_key, param_value) if animated and param_name == 'Speed': # "Custom Speed" lands between "Speed" and "Apply Speed" and # carries a <keyframeAnimation> child instead of a value. cs = ET.SubElement(elem, 'param') cs.set('name', 'Custom Speed') cs.set('key', self._TEXT_CUSTOM_SPEED_KEY) anim = ET.SubElement(cs, 'keyframeAnimation') for kf_time, kf_value in self._TEXT_CUSTOM_SPEED_KEYFRAMES: kf = ET.SubElement(anim, 'keyframe') kf.set('time', kf_time) kf.set('value', kf_value) if size_param is not None: _add_param('Size', self._TEXT_SIZE_KEY, f"{float(size_param):g}") text_el = ET.SubElement(elem, 'text') ts_id = self._unique_text_style_id(name) run = ET.SubElement(text_el, 'text-style') run.set('ref', ts_id) run.text = _sanitize_xml_value(text, 256) style_def = ET.SubElement(elem, 'text-style-def') style_def.set('id', ts_id) text_style = ET.SubElement(style_def, 'text-style') text_style.set('font', font) # Text.moti sizes type in frame pixels but positions in canvas points. # See TEXT_TEMPLATE_FONT_SCALE: layout measures in points, so only the # emitted size (and its kerning, to keep the same letter spacing) is # converted here. scale = float(font_scale) or 1.0 text_style.set('fontSize', f"{float(font_size) * scale:g}") text_style.set('fontColor', font_color) # FCP represents bold weight as the bold attribute — never as a # fontFace. Writing ``bold="0" fontFace="Bold"`` (the previous # behaviour) is contradictory and FCP refuses to render the text. # Italic, by contrast, IS a face: FCP writes both ``fontFace`` and # ``italic="1"``. See Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-19. face_lower = (face or '').strip().lower() if face_lower == 'bold': text_style.set('bold', '1') elif 'italic' in face_lower: text_style.set('fontFace', face) text_style.set('italic', '1') else: if bold: text_style.set('bold', '1') if face: text_style.set('fontFace', face) if kerning: text_style.set('kerning', f"{float(kerning) * scale:g}") text_style.set('alignment', 'center') text_style.set('lineSpacing', '-19') return elem def add_text_title( self, parent_clip: 'str | ET.Element', text: str, *, offset: str = '0s', duration: str = '1s', lane: int = 1, position: Optional[str] = None, font: str = 'Helvetica Neue', font_size: int = 196, font_color: str = '1 1 1 1', bold: bool = True, face: Optional[str] = None, animated: bool = True, font_scale: float = TEXT_TEMPLATE_FONT_SCALE, size_param: Optional[float] = None, role: Optional[str] = None, ) -> ET.Element: """Add a single static "Text" (Basic Text) title over *parent_clip*. Anchored in SOURCE media coordinates (parent's ``start`` + *offset*), matching FCP's own output, so the title lands on screen instead of at ~0s of the media (which FCP silently drops). *offset* and *duration* accept any FCPXML rational-time string; *position* is an optional "x y" canvas-point string to keep two titles from stacking. Returns: The created ``<title>`` element, already inserted into the parent in DTD order. """ parent = parent_clip if isinstance(parent_clip, ET.Element) else self._require_clip(parent_clip) resources = self.root.find('.//resources') if resources is None: raise ValueError("No <resources> element found in FCPXML") effect_id = self._ensure_text_title_effect(resources) media_origin = self._parse_time(parent.get('start', '0s')) relative = self._parse_time(offset) title = self._make_text_title_clip( effect_id, text, media_origin + relative, self._parse_time(duration), lane=lane, name=f"{text} - Text", position=position, font=font, font_size=font_size, font_color=font_color, bold=bold, face=face, animated=animated, font_scale=font_scale, size_param=size_param, role=role, ) _dtd_insert(parent, title) return title def generate_dynamic_subtitles( self, parent_clip: 'str | ET.Element', words: List[Dict[str, Any]], config: Optional['DynamicSubtitleConfig'] = None, segments: Optional[List[Dict[str, Any]]] = None, role: Optional[str] = None, configs: Optional[List['DynamicSubtitleConfig']] = None, compound_subphrases: bool = False, subphrase_min_words: int = 3, hold_between_sentences: bool = True, ) -> List[ET.Element]: """Generate progressive-reveal subtitle titles, one per word. Groups *words* into sentences (by *segments*' time windows), lays each sentence out as a compact typographic block, and emits one standalone ``<title>`` per word, positioned at its place in that block. Words appear one by one as they are spoken and accumulate on screen; every word of a block then clears at the same instant, so the sentence vanishes as a whole before the next one builds up. Each word gets its own lane, since a block's words are all on screen together. Lanes restart with each block. Size, colour, font and face cycle through ``config.style.rhythm``, reproducing the typography of the calibration export the user built in Final Cut. A sentence too tall for the band is split into successive blocks, so a long sentence never spills off screen. Args: parent_clip: The spine clip to attach titles to — either its Name/ID (resolved via ``_require_clip``, kept for backward compatibility) or the ``ET.Element`` itself. **Callers iterating multiple spine clips must pass the element, not the name**: after any ripple-cut/silence-removal operation, every fragment of an originally-named clip keeps that same ``name``, so ``self.clips`` (keyed by name) only retains the last-indexed one — a name lookup then silently resolves every call to the SAME wrong clip, stacking every line from every distinct clip's transcript onto one spine element (see Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-17). words: ``[{'word': str, 'start': float, 'end': float}, ...]`` with ``start``/``end`` in seconds *relative to the parent clip's own start* (same convention as ``add_connected_clip``'s ``offset``). config: Styling/layout options; defaults to ``DynamicSubtitleConfig()``. segments: Whisper sentence segments ``[{'start', 'end', ...}]``, on the same relative timebase as *words*. Omitted, every word falls into a single sentence, which the block layout then splits by height alone. Returns: The list of created ``<title>`` elements, in chronological order. """ # ``configs`` (a list of registered, active layouts) takes precedence # over the single ``config`` — with 2+ items, each block picks one at # random below; with 0 or 1, behaviour is identical to a single fixed # config, so old callers passing only ``config`` are unaffected. if configs: layout_configs = list(configs) elif config is not None: layout_configs = [config] else: layout_configs = [DynamicSubtitleConfig()] if not words: return [] parent = parent_clip if isinstance(parent_clip, ET.Element) else self._require_clip(parent_clip) resources = self.root.find('.//resources') if resources is None: raise ValueError("No <resources> element found in FCPXML") effect_id = self._ensure_text_title_effect(resources) # A connected title is NOT trimmed by its parent clip's out-point — # Final Cut keeps drawing it over whatever clip follows. A word that # starts after the cut would therefore only ever be seen on top of the # NEXT clip's own captions, so it is dropped rather than placed. parent_limit = self._parse_time(parent.get('duration', '0s')) has_limit = TimeValue(0, 1) < parent_limit if has_limit: limit_seconds = parent_limit.to_seconds() words = [ w for w in words if float(w.get('start', 0.0)) < limit_seconds ] if not words: return [] # Split into sentences, then lay each one out as a block. A sentence # too tall for the band comes back with overflow, which becomes the # next block — the sub-sentence split that keeps long sentences from # spilling off screen. sentences = group_words_by_segment(words, segments or []) # A comma is where the sentence breathes, so it is also where the # phrase should be packed into its own compound clip downstream. if compound_subphrases: sentences = [ sub for sentence in sentences for sub in split_into_subphrases(sentence, subphrase_min_words) ] def box_for(cfg: 'DynamicSubtitleConfig') -> LayoutBox: return LayoutBox.for_frame( self.frame_width(), self.frame_height(), band_height=cfg.band_height, center_y=cfg.block_center_y, ) def lay_out(pending: List[Dict], cfg: 'DynamicSubtitleConfig', box: LayoutBox): """Place what fits; return (units, still-unplaced words).""" # "phrase" is the progressive composition the reference reel uses: # one title per LINE ("que vão" / "melhorar" / "sua legenda"), the # key word set large in a display italic. "word" is the older # one-title-per-word rhythm, kept for callers that want every word # to land on its own. if getattr(cfg, 'granularity', 'phrase') == 'phrase': composition = compose_sentence( pending, cfg.style, box, line_gap=cfg.line_gap, ) return composition.blocks, composition.overflow layout = layout_sentence(pending, cfg.style, box) return layout.placed, layout.overflow blocks: List[List[Any]] = [] block_configs: List['DynamicSubtitleConfig'] = [] block_sentences: List[int] = [] for sentence_index, sentence in enumerate(sentences): remaining = list(sentence) while remaining: # Each block independently samples a layout from the active # set — the visual variety the user asked for. A single # active layout always resolves to itself, so this is a # no-op for the common case. active_config = layout_configs[random.randrange(len(layout_configs))] units, remaining = lay_out(remaining, active_config, box_for(active_config)) if not units: break blocks.append(units) block_configs.append(active_config) block_sentences.append(sentence_index) if not blocks: return [] # Never emit a zero-duration frame (rounds to 0 at the sequence's fps # and FCP rejects it as "unexpected value found"). min_dur_tv = self.snap_seconds_to_frame( float(self.frame_duration_fraction()) ) # Every word of a block clears at the same instant: when the next block # starts, or at the last word's end for the final block. That is what # makes a sentence build up and then vanish all at once. block_starts = [ self.snap_seconds_to_frame(min(unit.start for unit in units)) for units in blocks ] # Whisper's word end can also run past the cut, so a last block would # linger over the next clip's first block. Nothing may outlive the # clip it was written for. block_ends: List[TimeValue] = [] for i, units in enumerate(blocks): if i + 1 < len(blocks): end = block_starts[i + 1] if not hold_between_sentences and block_sentences[i] != block_sentences[i + 1]: spoken_end = self.snap_seconds_to_frame(max(unit.end for unit in units)) end = min(end, spoken_end) else: end = self.snap_seconds_to_frame( max(unit.end for unit in units) ) if end - block_starts[i] < min_dur_tv: end = block_starts[i] + min_dur_tv if has_limit and parent_limit < end: end = parent_limit block_ends.append(end) # Anchored titles are positioned in the parent clip's SOURCE media # coordinates: a title's offset is the parent clip's `start` plus its # timeline-relative position. Verified against FCP's own output in # "exemplo de arquivos.fcpxmld", where the hand-made "Essencial - # Título" sits at offset 226040815/24000s on a parent starting at # 226007782/24000s — 1.376s into a 1.835s clip. Writing a plain # relative offset instead would drop the title to ~0s of the media, # before the clip's own in-point, so it lands outside the clip and FCP # never shows it. media_origin = self._parse_time(parent.get('start', '0s')) created: List[ET.Element] = [] by_sentence: Dict[int, List[ET.Element]] = {} for units, block_end, block_config, sentence_index in zip( blocks, block_ends, block_configs, block_sentences ): # A ``titles.*`` sub-role keeps these as titles (never closed # captions) while grouping them in the role index and tinting # their lane. An explicit ``role`` argument overrides every # block; otherwise each block uses its own sampled layout's role. block_role = role or getattr(block_config, "role", None) or "titles.dinamicas" for index, unit in enumerate(units): relative_offset = self.snap_seconds_to_frame(unit.start) duration = block_end - relative_offset if duration < min_dur_tv: duration = min_dur_tv # Units of one block are all on screen together, so no two may # share a lane. Lanes restart each block, which is free — the # previous block has already cleared. lane = index + 1 offset = media_origin + relative_offset title = self._make_text_title_clip( effect_id, unit.text, offset, duration, lane=lane, name=f"caption_{uuid.uuid4().hex[:8]}", position=unit.position_param(block_config.text_scale), font=unit.font or block_config.style.font, font_size=int(round(unit.font_size)), font_color=unit.color or block_config.style.active_color, bold=block_config.style.bold, face=unit.face, kerning=unit.kerning, font_scale=block_config.text_scale, role=block_role, ) _dtd_insert(parent, title) self.mark_generated_subtitle(title, 'dynamic') created.append(title) by_sentence.setdefault(sentence_index, []).append(title) # One compound per sub-phrase: a dozen stacked title bars collapse # into a single one that can be dragged, muted or retimed as a unit. if compound_subphrases: for sentence_index in sorted(by_sentence): group = by_sentence[sentence_index] label = " ".join( str(w.get('word') or w.get('text') or '') for w in sentences[sentence_index] ).strip() compound = self.wrap_titles_in_compound( parent, group, name=label[:60] or "Legenda" ) self.mark_generated_subtitle(compound, 'dynamic') if any(getattr(cfg, 'validate', False) for cfg in layout_configs): report = self.validate_subtitle_layout() if blocking(report["severity"]): raise ValueError( "Subtitle layout validation failed: " + str(report["summary"]) ) return created def validate_subtitle_layout( self, *, safe_margin_x: float = 0.05, safe_margin_y: float = 0.05, min_font_size: Optional[float] = None, min_distance: Optional[float] = None, max_distance: Optional[float] = None, ) -> dict: """Re-measure every ``<title>`` in the document and report collisions. Reconstructs each title's on-screen box from the values the writer emitted (``fontSize``/``kerning``/``Position`` are already in template space), then checks for temporal+spatial collisions, frame/safe-area containment, and font fallbacks. This is the spec-16 validation pass the layout engine does not do on its own — it only guarantees non-overlap *by construction* while composing, and cannot see a hand-edited title. Returns the ``collision.validate_titles`` report: ``severity`` (worst bucket), ``issues`` (spec-16 occurrences) and ``summary`` (counts). """ # A compound clip carries its own time origin: a title inside one is # offset from that compound's start, not the sequence's. Measured in # one flat pass, the anchors of two different compounds both read as # "0s" and collide on paper while sitting seconds apart on the # timeline. Each compound is therefore measured as its own scope, # which is also where its titles can actually overlap — a title can # only share the screen with its own compound's siblings. scopes: List[List[ET.Element]] = [] nested: set = set() for media in self.root.findall('.//media'): group = list(media.iter('title')) if group: scopes.append(group) nested.update(id(t) for t in group) main = [t for t in self.root.iter('title') if id(t) not in nested] if main: scopes.append(main) reports = [ self._measure_title_scope( scope, safe_margin_x=safe_margin_x, safe_margin_y=safe_margin_y, min_font_size=min_font_size, min_distance=min_distance, max_distance=max_distance, ) for scope in scopes ] if len(reports) == 1: return reports[0] if not reports: return self._measure_title_scope( [], safe_margin_x=safe_margin_x, safe_margin_y=safe_margin_y, min_font_size=min_font_size, min_distance=min_distance, max_distance=max_distance, ) rank = { 'none': 0, 'render_tolerance': 1, 'warning': 2, 'probable': 3, 'severe': 4, } merged_issues = [i for r in reports for i in r['issues']] summary = dict(reports[0]['summary']) for r in reports[1:]: for key, value in r['summary'].items(): summary[key] = summary.get(key, 0) + value return { 'severity': max( (r['severity'] for r in reports), key=lambda s: rank.get(s, 0), ), 'issues': merged_issues, 'summary': summary, } def _measure_title_scope( self, elements: List[ET.Element], *, safe_margin_x: float, safe_margin_y: float, min_font_size: Optional[float], min_distance: Optional[float], max_distance: Optional[float], ) -> dict: """Measure and validate one group of titles sharing a time origin.""" titles = [] for elem in elements: # enabled="0" never renders in Final Cut (see # generate_subtitles_by_emphasis, which disables plain titles # under an emphasis phrase instead of never creating them) — a # title that is off by design must not count as a collision # against the one drawn in its place. if elem.get('enabled', '1') == '0': continue text_el = elem.find('text/text-style') text = (text_el.text or '').strip() if text_el is not None else '' style = elem.find('text-style-def/text-style') font = style.get('font') if style is not None else None face = style.get('fontFace') if style is not None else None font_size = ( float(style.get('fontSize', '0')) if style is not None else 0.0 ) kerning = ( float(style.get('kerning', '0') or 0) if style is not None else 0.0 ) x = y = 0.0 for param in elem.findall('param'): if param.get('name') == 'Position' and param.get('value'): parts = param.get('value').split() if len(parts) >= 2: x, y = float(parts[0]), float(parts[1]) start = self._parse_time(elem.get('offset', '0s')).to_seconds() duration = self._parse_time(elem.get('duration', '0s')).to_seconds() titles.append({ 'text': text, 'font': font, 'face': face, 'font_size': font_size, 'kerning': kerning, 'x': x, 'y': y, 'start': start, 'end': start + duration, 'group': start + duration, }) return validate_titles( titles, self.frame_width(), self.frame_height(), safe_margin_x=safe_margin_x, safe_margin_y=safe_margin_y, min_font_size=min_font_size, min_distance=min_distance, max_distance=max_distance, ) # ========================================================================