generate_dynamic_subtitles e a metade dinâmica de generate_subtitles_by_emphasis passam a empacotar cada sub-frase da legenda dinâmica num compound clip por padrão (compound_subphrases=True), completando o wrap_titles_in_compound e split_into_subphrases do commit anterior — que ainda não tinham chamador em produção. Também torna validate_subtitle_layout ciente de compound clips: media cada grupo (spine principal + cada <media> de compound) no seu próprio espaço de tempo, em vez de uma varredura .//title global — sem isso, âncoras de compounds diferentes liam offset "0s" e acusavam colisão espacial entre frases que nunca dividem a tela, só porque compartilham o mesmo zero de tempo local. Testado ponta a ponta na gravação real (Mastopexia): 12 compounds, 41 títulos todos empacotados, zero soltos, zero IDs duplicados, DTD válida. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
776 lines
34 KiB
Python
776 lines
34 KiB
Python
"""Títulos de texto e legendas dinâmicas.
|
|
|
|
Extraído de writer.py — ver fcpxml/writer/__init__.py para o conjunto.
|
|
"""
|
|
|
|
import random
|
|
import re
|
|
import unicodedata
|
|
import uuid
|
|
import xml.etree.ElementTree as ET
|
|
from typing import Any, Dict, List, Optional
|
|
|
|
from ..collision import blocking, validate_titles
|
|
from ..models import (
|
|
DynamicSubtitleConfig,
|
|
TimeValue,
|
|
)
|
|
from ..text_layout import (
|
|
TEXT_TEMPLATE_FONT_SCALE,
|
|
LayoutBox,
|
|
compose_sentence,
|
|
layout_sentence,
|
|
)
|
|
from ..transcribe import group_words_by_segment, split_into_subphrases
|
|
from .helpers import _dtd_insert, _sanitize_xml_value
|
|
|
|
|
|
class TitlesMixin:
|
|
"""Títulos de texto e legendas dinâmicas."""
|
|
|
|
# DYNAMIC (KARAOKE-STYLE) SUBTITLES
|
|
# ========================================================================
|
|
|
|
# The "Text" (Basic Text) template — the ONLY simple title template that
|
|
# Final Cut actually renders. Copied verbatim from the user's own FCP
|
|
# exports ("teste.fcpxmld" and "posição.fcpxmld", FCP 1.14 in English):
|
|
# a single "<text>" run, one "<text-style-def>", and a fixed param block
|
|
# with the margins/alignment/speed the template ships with. Every prior
|
|
# title template we generated ("Essencial - Título", "Título Básico")
|
|
# imported cleanly but never appeared — their Motion uids did not resolve
|
|
# to a real, drawable template in FCP, which discards the clip silently.
|
|
# "Text" is what FCP itself writes when the user adds a title by hand, so
|
|
# it is the ground truth. See Engine/docs/05_EXPERIENCIAS.md, 2026-08-17.
|
|
_TEXT_TITLE_UID = (
|
|
'.../Titles.localized/Basic Text.localized/'
|
|
'Text.localized/Text.moti'
|
|
)
|
|
_TEXT_TITLE_START = '86486400/24000s'
|
|
# The Inspector's Position field, and the one this code overrides per
|
|
# title so two titles never stack on top of each other. Verified in
|
|
# "posição.fcpxmld": each hand-dragged title carries a distinct "x y"
|
|
# value here while every other param stays identical.
|
|
_TEXT_POSITION_KEY = '9999/10003/13260/3296672360/1/100/101'
|
|
# Layout params the "Text" template ships with. These keys are the
|
|
# template's own defaults and never vary between instances.
|
|
#
|
|
# "Build Out" is the one deliberate override: with "Apply Speed" set to
|
|
# "2 (Per Object)" below, the template's whole built-in animation (build
|
|
# in + build out) is always compressed to exactly fill the title's own
|
|
# on-screen duration — so on a short word-length clip, build out was
|
|
# eating time that build in needed to finish revealing the text before
|
|
# the cut. Disabling build out hands that entire compressed window to
|
|
# build in alone, which is what "sempre acelerado" turned out to mean:
|
|
# no separate speed knob needed. Value captured from a real FCP export
|
|
# with "Build Out" unchecked in the Inspector (see chat, 2026-08-18).
|
|
_TEXT_TITLE_PARAMS = (
|
|
('Build Out', '9999/10000/2/102', '0'),
|
|
('Layout Method', '9999/10003/13260/3296672360/2/314', '1 (Paragraph)'),
|
|
('Left Margin', '9999/10003/13260/3296672360/2/323', '-1210'),
|
|
('Right Margin', '9999/10003/13260/3296672360/2/324', '1210'),
|
|
('Top Margin', '9999/10003/13260/3296672360/2/325', '2160'),
|
|
('Bottom Margin', '9999/10003/13260/3296672360/2/326', '-2160'),
|
|
('Alignment', '9999/10003/13260/3296672360/2/354/3296667315/401', '1 (Center)'),
|
|
('Line Spacing', '9999/10003/13260/3296672360/2/354/3296667315/404', '-19'),
|
|
('Auto-Shrink', '9999/10003/13260/3296672360/2/370', '3 (To All Margins)'),
|
|
('Alignment', '9999/10003/13260/3296672360/2/373', '0 (Left) 1 (Middle)'),
|
|
('Opacity', '9999/10003/13260/3296672360/4/3296673134/1000/1044', '0'),
|
|
('Speed', '9999/10003/13260/3296672360/4/3296673134/201/208', '6 (Custom)'),
|
|
('Apply Speed', '9999/10003/13260/3296672360/4/3296673134/201/211', '2 (Per Object)'),
|
|
)
|
|
# "Custom Speed" sits between "Speed" and "Apply Speed" and carries a
|
|
# <keyframeAnimation> child rather than a plain value attribute. Its two
|
|
# keyframes are the template's own absolute nominal times, constant across
|
|
# every instance, so they are safe to replay verbatim.
|
|
_TEXT_CUSTOM_SPEED_KEY = '9999/10003/13260/3296672360/4/3296673134/201/209'
|
|
_TEXT_CUSTOM_SPEED_KEYFRAMES = (
|
|
('-469658744/1000000000s', '0'),
|
|
('12328542033/1000000000s', '1'),
|
|
)
|
|
_TEXT_SIZE_KEY = '9999/10003/13260/3296672360/5/3296672362/3'
|
|
|
|
def _ensure_text_title_effect(self, resources: ET.Element) -> str:
|
|
"""Return the resource id of the "Text" (Basic Text) effect, creating it if absent."""
|
|
return self._ensure_effect(resources, self._TEXT_TITLE_UID, 'Text', 'r_text')
|
|
|
|
def _ensure_effect(
|
|
self,
|
|
resources: ET.Element,
|
|
uid: str,
|
|
name: str,
|
|
id_prefix: str,
|
|
) -> str:
|
|
"""Return the id of the effect resource with *uid*, creating it if absent."""
|
|
for eff in resources.findall('effect'):
|
|
if eff.get('uid') == uid:
|
|
return eff.get('id')
|
|
effect_id = self._unique_resource_id(resources, id_prefix)
|
|
eff_el = ET.SubElement(resources, 'effect')
|
|
eff_el.set('id', effect_id)
|
|
eff_el.set('name', name)
|
|
eff_el.set('uid', uid)
|
|
return effect_id
|
|
|
|
# <text-style-def id> / <text-style ref> are DTD type ID/IDREF, so the
|
|
# value must be a valid XML Name: letters, digits, "_", "-", "." only,
|
|
# never starting with a digit. Title names are built from the caption
|
|
# text ("Olá mundo - Text"), which carries spaces, accents and often a
|
|
# leading digit — xmllint rejected the whole document with "Syntax of
|
|
# value for attribute id of text-style-def is not valid".
|
|
_TEXT_STYLE_ID_UNSAFE = re.compile(r'[^A-Za-z0-9_.-]+')
|
|
|
|
def _unique_text_style_id(self, base: str) -> str:
|
|
"""Return a document-unique, DTD-valid XML ID for a ``<text-style-def>``."""
|
|
folded = unicodedata.normalize('NFKD', base).encode('ascii', 'ignore').decode('ascii')
|
|
slug = self._TEXT_STYLE_ID_UNSAFE.sub('_', folded).strip('_.-')[:48]
|
|
stem = f"ts_{slug}" if slug else "ts"
|
|
|
|
if self._text_style_ids is None:
|
|
self._text_style_ids = {
|
|
sd.get('id') for sd in self.root.findall('.//text-style-def')
|
|
}
|
|
candidate = f"{stem}_0"
|
|
counter = 0
|
|
while candidate in self._text_style_ids:
|
|
counter += 1
|
|
candidate = f"{stem}_{counter}"
|
|
self._text_style_ids.add(candidate)
|
|
return candidate
|
|
|
|
def _reassign_text_style_ids(self, clip: ET.Element) -> None:
|
|
"""Give every ``<text-style-def>`` inside a just-deepcopy'd *clip* a
|
|
fresh document-unique id, repointing any ``<text-style ref="...">``
|
|
in the same subtree that pointed at the old one.
|
|
|
|
``split_clip``/``cut_clip_ranges`` deepcopy the clip once per
|
|
resulting segment, so a clip carrying a ``<title>`` (from a "text"
|
|
voice action) keeps the exact same ``text-style-def id`` in every
|
|
copy. A single cut is harmless — but the batch chain re-cuts the
|
|
same clip at each step (silence removal, filler removal, dynamic
|
|
subtitles), and every pass multiplies the duplicate, so the DTD
|
|
validator eventually rejects the file with "ID ... already
|
|
defined". Regenerating here, at the only place copies are made,
|
|
fixes it for every caller instead of each one having to remember to.
|
|
"""
|
|
for style_def in clip.findall('.//text-style-def'):
|
|
old_id = style_def.get('id')
|
|
if not old_id:
|
|
continue
|
|
slug = old_id[3:] if old_id.startswith('ts_') else old_id
|
|
slug = re.sub(r'_\d+$', '', slug) # drop a prior _<N> counter
|
|
new_id = self._unique_text_style_id(slug)
|
|
if new_id == old_id:
|
|
continue
|
|
style_def.set('id', new_id)
|
|
for ref_el in clip.findall(f".//text-style[@ref='{old_id}']"):
|
|
ref_el.set('ref', new_id)
|
|
|
|
def _unique_tracking_shape_id(self, base: str) -> str:
|
|
"""Return a document-unique ``id`` for a ``<tracking-shape>``."""
|
|
stem = base or "tr"
|
|
if self._tracking_shape_ids is None:
|
|
self._tracking_shape_ids = {
|
|
ts.get('id') for ts in self.root.findall('.//tracking-shape')
|
|
}
|
|
candidate = f"{stem}_0"
|
|
counter = 0
|
|
while candidate in self._tracking_shape_ids:
|
|
counter += 1
|
|
candidate = f"{stem}_{counter}"
|
|
self._tracking_shape_ids.add(candidate)
|
|
return candidate
|
|
|
|
def _reassign_tracking_shape_ids(self, clip: ET.Element) -> None:
|
|
"""Give every ``<tracking-shape>`` inside a just-deepcopy'd *clip* a
|
|
fresh document-unique id.
|
|
|
|
Same mechanism as ``_reassign_text_style_ids``: ``split_clip``/
|
|
``cut_clip_ranges`` deepcopy the clip once per resulting segment, so
|
|
Cinematic object-tracking data (``<object-tracker><tracking-shape
|
|
id="tr1">``, preserved from the source asset's sidecar) keeps the
|
|
exact same id in every copy. A single cut is harmless — but the
|
|
batch chain re-cuts the same clip at each step, multiplying the
|
|
duplicate until the DTD validator rejects the file with "ID tr1
|
|
already defined".
|
|
"""
|
|
for shape in clip.findall('.//tracking-shape'):
|
|
old_id = shape.get('id')
|
|
if not old_id:
|
|
continue
|
|
base = re.sub(r'_\d+$', '', old_id)
|
|
new_id = self._unique_tracking_shape_id(base)
|
|
if new_id == old_id:
|
|
continue
|
|
shape.set('id', new_id)
|
|
|
|
def _make_text_title_clip(
|
|
self,
|
|
effect_id: str,
|
|
text: str,
|
|
offset: 'TimeValue',
|
|
duration: 'TimeValue',
|
|
*,
|
|
lane: int,
|
|
name: str,
|
|
position: Optional[str] = None,
|
|
font: str = 'Helvetica Neue',
|
|
font_size: int = 196,
|
|
font_color: str = '1 1 1 1',
|
|
bold: bool = True,
|
|
face: Optional[str] = None,
|
|
kerning: Optional[float] = None,
|
|
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
|
|
animated: bool = True,
|
|
size_param: Optional[float] = None,
|
|
role: Optional[str] = None,
|
|
) -> ET.Element:
|
|
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
|
|
|
|
Reproduces FCP's own output for a hand-added title exactly — the only
|
|
template we have verified renders in Final Cut ("teste.fcpxmld" and
|
|
"posição.fcpxmld"). *position* ("x y" canvas points) is the Inspector
|
|
Position value; omit it to keep the template's centred default. Unlike
|
|
the animated templates, this carries no animation switch, so the text
|
|
stays put and visible for its whole duration.
|
|
"""
|
|
elem = ET.Element('title')
|
|
elem.set('ref', effect_id)
|
|
elem.set('lane', str(lane))
|
|
elem.set('offset', offset.to_fcpxml())
|
|
elem.set('name', _sanitize_xml_value(name, 256))
|
|
elem.set('start', self._TEXT_TITLE_START)
|
|
elem.set('duration', duration.to_fcpxml())
|
|
if role:
|
|
elem.set('role', _sanitize_xml_value(role, 256))
|
|
|
|
if position:
|
|
param = ET.SubElement(elem, 'param')
|
|
param.set('name', 'Position')
|
|
param.set('key', self._TEXT_POSITION_KEY)
|
|
param.set('value', position)
|
|
|
|
def _add_param(name: str, key: str, value: str) -> None:
|
|
param = ET.SubElement(elem, 'param')
|
|
param.set('name', name)
|
|
param.set('key', key)
|
|
param.set('value', value)
|
|
|
|
animation_params = {'Opacity', 'Speed', 'Apply Speed'}
|
|
for param_name, param_key, param_value in self._TEXT_TITLE_PARAMS:
|
|
if not animated and param_name in animation_params:
|
|
continue
|
|
_add_param(param_name, param_key, param_value)
|
|
if animated and param_name == 'Speed':
|
|
# "Custom Speed" lands between "Speed" and "Apply Speed" and
|
|
# carries a <keyframeAnimation> child instead of a value.
|
|
cs = ET.SubElement(elem, 'param')
|
|
cs.set('name', 'Custom Speed')
|
|
cs.set('key', self._TEXT_CUSTOM_SPEED_KEY)
|
|
anim = ET.SubElement(cs, 'keyframeAnimation')
|
|
for kf_time, kf_value in self._TEXT_CUSTOM_SPEED_KEYFRAMES:
|
|
kf = ET.SubElement(anim, 'keyframe')
|
|
kf.set('time', kf_time)
|
|
kf.set('value', kf_value)
|
|
|
|
if size_param is not None:
|
|
_add_param('Size', self._TEXT_SIZE_KEY, f"{float(size_param):g}")
|
|
|
|
text_el = ET.SubElement(elem, 'text')
|
|
ts_id = self._unique_text_style_id(name)
|
|
run = ET.SubElement(text_el, 'text-style')
|
|
run.set('ref', ts_id)
|
|
run.text = _sanitize_xml_value(text, 256)
|
|
|
|
style_def = ET.SubElement(elem, 'text-style-def')
|
|
style_def.set('id', ts_id)
|
|
text_style = ET.SubElement(style_def, 'text-style')
|
|
text_style.set('font', font)
|
|
# Text.moti sizes type in frame pixels but positions in canvas points.
|
|
# See TEXT_TEMPLATE_FONT_SCALE: layout measures in points, so only the
|
|
# emitted size (and its kerning, to keep the same letter spacing) is
|
|
# converted here.
|
|
scale = float(font_scale) or 1.0
|
|
text_style.set('fontSize', f"{float(font_size) * scale:g}")
|
|
text_style.set('fontColor', font_color)
|
|
# FCP represents bold weight as the bold attribute — never as a
|
|
# fontFace. Writing ``bold="0" fontFace="Bold"`` (the previous
|
|
# behaviour) is contradictory and FCP refuses to render the text.
|
|
# Italic, by contrast, IS a face: FCP writes both ``fontFace`` and
|
|
# ``italic="1"``. See Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-19.
|
|
face_lower = (face or '').strip().lower()
|
|
if face_lower == 'bold':
|
|
text_style.set('bold', '1')
|
|
elif 'italic' in face_lower:
|
|
text_style.set('fontFace', face)
|
|
text_style.set('italic', '1')
|
|
else:
|
|
if bold:
|
|
text_style.set('bold', '1')
|
|
if face:
|
|
text_style.set('fontFace', face)
|
|
if kerning:
|
|
text_style.set('kerning', f"{float(kerning) * scale:g}")
|
|
text_style.set('alignment', 'center')
|
|
text_style.set('lineSpacing', '-19')
|
|
|
|
return elem
|
|
|
|
def add_text_title(
|
|
self,
|
|
parent_clip: 'str | ET.Element',
|
|
text: str,
|
|
*,
|
|
offset: str = '0s',
|
|
duration: str = '1s',
|
|
lane: int = 1,
|
|
position: Optional[str] = None,
|
|
font: str = 'Helvetica Neue',
|
|
font_size: int = 196,
|
|
font_color: str = '1 1 1 1',
|
|
bold: bool = True,
|
|
face: Optional[str] = None,
|
|
animated: bool = True,
|
|
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
|
|
size_param: Optional[float] = None,
|
|
role: Optional[str] = None,
|
|
) -> ET.Element:
|
|
"""Add a single static "Text" (Basic Text) title over *parent_clip*.
|
|
|
|
Anchored in SOURCE media coordinates (parent's ``start`` + *offset*),
|
|
matching FCP's own output, so the title lands on screen instead of at
|
|
~0s of the media (which FCP silently drops). *offset* and *duration*
|
|
accept any FCPXML rational-time string; *position* is an optional
|
|
"x y" canvas-point string to keep two titles from stacking.
|
|
|
|
Returns:
|
|
The created ``<title>`` element, already inserted into the parent
|
|
in DTD order.
|
|
"""
|
|
parent = parent_clip if isinstance(parent_clip, ET.Element) else self._require_clip(parent_clip)
|
|
resources = self.root.find('.//resources')
|
|
if resources is None:
|
|
raise ValueError("No <resources> element found in FCPXML")
|
|
effect_id = self._ensure_text_title_effect(resources)
|
|
|
|
media_origin = self._parse_time(parent.get('start', '0s'))
|
|
relative = self._parse_time(offset)
|
|
title = self._make_text_title_clip(
|
|
effect_id,
|
|
text,
|
|
media_origin + relative,
|
|
self._parse_time(duration),
|
|
lane=lane,
|
|
name=f"{text} - Text",
|
|
position=position,
|
|
font=font,
|
|
font_size=font_size,
|
|
font_color=font_color,
|
|
bold=bold,
|
|
face=face,
|
|
animated=animated,
|
|
font_scale=font_scale,
|
|
size_param=size_param,
|
|
role=role,
|
|
)
|
|
_dtd_insert(parent, title)
|
|
return title
|
|
|
|
def generate_dynamic_subtitles(
|
|
self,
|
|
parent_clip: 'str | ET.Element',
|
|
words: List[Dict[str, Any]],
|
|
config: Optional['DynamicSubtitleConfig'] = None,
|
|
segments: Optional[List[Dict[str, Any]]] = None,
|
|
role: Optional[str] = None,
|
|
configs: Optional[List['DynamicSubtitleConfig']] = None,
|
|
compound_subphrases: bool = False,
|
|
subphrase_min_words: int = 3,
|
|
) -> List[ET.Element]:
|
|
"""Generate progressive-reveal subtitle titles, one per word.
|
|
|
|
Groups *words* into sentences (by *segments*' time windows), lays each
|
|
sentence out as a compact typographic block, and emits one standalone
|
|
``<title>`` per word, positioned at its place in that block. Words
|
|
appear one by one as they are spoken and accumulate on screen; every
|
|
word of a block then clears at the same instant, so the sentence
|
|
vanishes as a whole before the next one builds up.
|
|
|
|
Each word gets its own lane, since a block's words are all on screen
|
|
together. Lanes restart with each block. Size, colour, font and face
|
|
cycle through ``config.style.rhythm``, reproducing the typography of
|
|
the calibration export the user built in Final Cut.
|
|
|
|
A sentence too tall for the band is split into successive blocks, so a
|
|
long sentence never spills off screen.
|
|
|
|
Args:
|
|
parent_clip: The spine clip to attach titles to — either its
|
|
Name/ID (resolved via ``_require_clip``, kept for backward
|
|
compatibility) or the ``ET.Element`` itself. **Callers
|
|
iterating multiple spine clips must pass the element, not
|
|
the name**: after any ripple-cut/silence-removal operation,
|
|
every fragment of an originally-named clip keeps that same
|
|
``name``, so ``self.clips`` (keyed by name) only retains the
|
|
last-indexed one — a name lookup then silently resolves
|
|
every call to the SAME wrong clip, stacking every line from
|
|
every distinct clip's transcript onto one spine element (see
|
|
Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-17).
|
|
words: ``[{'word': str, 'start': float, 'end': float}, ...]``
|
|
with ``start``/``end`` in seconds *relative to the parent
|
|
clip's own start* (same convention as ``add_connected_clip``'s
|
|
``offset``).
|
|
config: Styling/layout options; defaults to ``DynamicSubtitleConfig()``.
|
|
segments: Whisper sentence segments ``[{'start', 'end', ...}]``, on
|
|
the same relative timebase as *words*. Omitted, every word
|
|
falls into a single sentence, which the block layout then
|
|
splits by height alone.
|
|
|
|
Returns:
|
|
The list of created ``<title>`` elements, in chronological order.
|
|
"""
|
|
# ``configs`` (a list of registered, active layouts) takes precedence
|
|
# over the single ``config`` — with 2+ items, each block picks one at
|
|
# random below; with 0 or 1, behaviour is identical to a single fixed
|
|
# config, so old callers passing only ``config`` are unaffected.
|
|
if configs:
|
|
layout_configs = list(configs)
|
|
elif config is not None:
|
|
layout_configs = [config]
|
|
else:
|
|
layout_configs = [DynamicSubtitleConfig()]
|
|
if not words:
|
|
return []
|
|
|
|
parent = parent_clip if isinstance(parent_clip, ET.Element) else self._require_clip(parent_clip)
|
|
resources = self.root.find('.//resources')
|
|
if resources is None:
|
|
raise ValueError("No <resources> element found in FCPXML")
|
|
effect_id = self._ensure_text_title_effect(resources)
|
|
|
|
# A connected title is NOT trimmed by its parent clip's out-point —
|
|
# Final Cut keeps drawing it over whatever clip follows. A word that
|
|
# starts after the cut would therefore only ever be seen on top of the
|
|
# NEXT clip's own captions, so it is dropped rather than placed.
|
|
parent_limit = self._parse_time(parent.get('duration', '0s'))
|
|
has_limit = TimeValue(0, 1) < parent_limit
|
|
if has_limit:
|
|
limit_seconds = parent_limit.to_seconds()
|
|
words = [
|
|
w for w in words
|
|
if float(w.get('start', 0.0)) < limit_seconds
|
|
]
|
|
if not words:
|
|
return []
|
|
|
|
# Split into sentences, then lay each one out as a block. A sentence
|
|
# too tall for the band comes back with overflow, which becomes the
|
|
# next block — the sub-sentence split that keeps long sentences from
|
|
# spilling off screen.
|
|
sentences = group_words_by_segment(words, segments or [])
|
|
# A comma is where the sentence breathes, so it is also where the
|
|
# phrase should be packed into its own compound clip downstream.
|
|
if compound_subphrases:
|
|
sentences = [
|
|
sub
|
|
for sentence in sentences
|
|
for sub in split_into_subphrases(sentence, subphrase_min_words)
|
|
]
|
|
|
|
def box_for(cfg: 'DynamicSubtitleConfig') -> LayoutBox:
|
|
return LayoutBox.for_frame(
|
|
self.frame_width(), self.frame_height(),
|
|
band_height=cfg.band_height,
|
|
center_y=cfg.block_center_y,
|
|
)
|
|
|
|
def lay_out(pending: List[Dict], cfg: 'DynamicSubtitleConfig', box: LayoutBox):
|
|
"""Place what fits; return (units, still-unplaced words)."""
|
|
# "phrase" is the progressive composition the reference reel uses:
|
|
# one title per LINE ("que vão" / "melhorar" / "sua legenda"), the
|
|
# key word set large in a display italic. "word" is the older
|
|
# one-title-per-word rhythm, kept for callers that want every word
|
|
# to land on its own.
|
|
if getattr(cfg, 'granularity', 'phrase') == 'phrase':
|
|
composition = compose_sentence(
|
|
pending, cfg.style, box, line_gap=cfg.line_gap,
|
|
)
|
|
return composition.blocks, composition.overflow
|
|
layout = layout_sentence(pending, cfg.style, box)
|
|
return layout.placed, layout.overflow
|
|
|
|
blocks: List[List[Any]] = []
|
|
block_configs: List['DynamicSubtitleConfig'] = []
|
|
block_sentences: List[int] = []
|
|
for sentence_index, sentence in enumerate(sentences):
|
|
remaining = list(sentence)
|
|
while remaining:
|
|
# Each block independently samples a layout from the active
|
|
# set — the visual variety the user asked for. A single
|
|
# active layout always resolves to itself, so this is a
|
|
# no-op for the common case.
|
|
active_config = layout_configs[random.randrange(len(layout_configs))]
|
|
units, remaining = lay_out(remaining, active_config, box_for(active_config))
|
|
if not units:
|
|
break
|
|
blocks.append(units)
|
|
block_configs.append(active_config)
|
|
block_sentences.append(sentence_index)
|
|
if not blocks:
|
|
return []
|
|
|
|
# Never emit a zero-duration frame (rounds to 0 at the sequence's fps
|
|
# and FCP rejects it as "unexpected value found").
|
|
min_dur_tv = self.snap_seconds_to_frame(
|
|
float(self.frame_duration_fraction())
|
|
)
|
|
|
|
# Every word of a block clears at the same instant: when the next block
|
|
# starts, or at the last word's end for the final block. That is what
|
|
# makes a sentence build up and then vanish all at once.
|
|
block_starts = [
|
|
self.snap_seconds_to_frame(min(unit.start for unit in units))
|
|
for units in blocks
|
|
]
|
|
# Whisper's word end can also run past the cut, so a last block would
|
|
# linger over the next clip's first block. Nothing may outlive the
|
|
# clip it was written for.
|
|
block_ends: List[TimeValue] = []
|
|
for i, units in enumerate(blocks):
|
|
if i + 1 < len(blocks):
|
|
end = block_starts[i + 1]
|
|
else:
|
|
end = self.snap_seconds_to_frame(
|
|
max(unit.end for unit in units)
|
|
)
|
|
if end - block_starts[i] < min_dur_tv:
|
|
end = block_starts[i] + min_dur_tv
|
|
if has_limit and parent_limit < end:
|
|
end = parent_limit
|
|
block_ends.append(end)
|
|
|
|
# Anchored titles are positioned in the parent clip's SOURCE media
|
|
# coordinates: a title's offset is the parent clip's `start` plus its
|
|
# timeline-relative position. Verified against FCP's own output in
|
|
# "exemplo de arquivos.fcpxmld", where the hand-made "Essencial -
|
|
# Título" sits at offset 226040815/24000s on a parent starting at
|
|
# 226007782/24000s — 1.376s into a 1.835s clip. Writing a plain
|
|
# relative offset instead would drop the title to ~0s of the media,
|
|
# before the clip's own in-point, so it lands outside the clip and FCP
|
|
# never shows it.
|
|
media_origin = self._parse_time(parent.get('start', '0s'))
|
|
|
|
created: List[ET.Element] = []
|
|
by_sentence: Dict[int, List[ET.Element]] = {}
|
|
for units, block_end, block_config, sentence_index in zip(
|
|
blocks, block_ends, block_configs, block_sentences
|
|
):
|
|
# A ``titles.*`` sub-role keeps these as titles (never closed
|
|
# captions) while grouping them in the role index and tinting
|
|
# their lane. An explicit ``role`` argument overrides every
|
|
# block; otherwise each block uses its own sampled layout's role.
|
|
block_role = role or getattr(block_config, "role", None) or "titles.dinamicas"
|
|
for index, unit in enumerate(units):
|
|
relative_offset = self.snap_seconds_to_frame(unit.start)
|
|
duration = block_end - relative_offset
|
|
if duration < min_dur_tv:
|
|
duration = min_dur_tv
|
|
|
|
# Units of one block are all on screen together, so no two may
|
|
# share a lane. Lanes restart each block, which is free — the
|
|
# previous block has already cleared.
|
|
lane = index + 1
|
|
|
|
offset = media_origin + relative_offset
|
|
title = self._make_text_title_clip(
|
|
effect_id,
|
|
unit.text,
|
|
offset,
|
|
duration,
|
|
lane=lane,
|
|
name=f"caption_{uuid.uuid4().hex[:8]}",
|
|
position=unit.position_param(block_config.text_scale),
|
|
font=unit.font or block_config.style.font,
|
|
font_size=int(round(unit.font_size)),
|
|
font_color=unit.color or block_config.style.active_color,
|
|
bold=block_config.style.bold,
|
|
face=unit.face,
|
|
kerning=unit.kerning,
|
|
font_scale=block_config.text_scale,
|
|
role=block_role,
|
|
)
|
|
_dtd_insert(parent, title)
|
|
created.append(title)
|
|
by_sentence.setdefault(sentence_index, []).append(title)
|
|
|
|
# One compound per sub-phrase: a dozen stacked title bars collapse
|
|
# into a single one that can be dragged, muted or retimed as a unit.
|
|
if compound_subphrases:
|
|
for sentence_index in sorted(by_sentence):
|
|
group = by_sentence[sentence_index]
|
|
label = " ".join(
|
|
str(w.get('word') or w.get('text') or '')
|
|
for w in sentences[sentence_index]
|
|
).strip()
|
|
self.wrap_titles_in_compound(
|
|
parent, group, name=label[:60] or "Legenda"
|
|
)
|
|
|
|
if any(getattr(cfg, 'validate', False) for cfg in layout_configs):
|
|
report = self.validate_subtitle_layout()
|
|
if blocking(report["severity"]):
|
|
raise ValueError(
|
|
"Subtitle layout validation failed: "
|
|
+ str(report["summary"])
|
|
)
|
|
|
|
return created
|
|
|
|
def validate_subtitle_layout(
|
|
self,
|
|
*,
|
|
safe_margin_x: float = 0.05,
|
|
safe_margin_y: float = 0.05,
|
|
min_font_size: Optional[float] = None,
|
|
min_distance: Optional[float] = None,
|
|
max_distance: Optional[float] = None,
|
|
) -> dict:
|
|
"""Re-measure every ``<title>`` in the document and report collisions.
|
|
|
|
Reconstructs each title's on-screen box from the values the writer
|
|
emitted (``fontSize``/``kerning``/``Position`` are already in template
|
|
space), then checks for temporal+spatial collisions, frame/safe-area
|
|
containment, and font fallbacks. This is the spec-16 validation pass the
|
|
layout engine does not do on its own — it only guarantees non-overlap
|
|
*by construction* while composing, and cannot see a hand-edited title.
|
|
|
|
Returns the ``collision.validate_titles`` report: ``severity`` (worst
|
|
bucket), ``issues`` (spec-16 occurrences) and ``summary`` (counts).
|
|
"""
|
|
# A compound clip carries its own time origin: a title inside one is
|
|
# offset from that compound's start, not the sequence's. Measured in
|
|
# one flat pass, the anchors of two different compounds both read as
|
|
# "0s" and collide on paper while sitting seconds apart on the
|
|
# timeline. Each compound is therefore measured as its own scope,
|
|
# which is also where its titles can actually overlap — a title can
|
|
# only share the screen with its own compound's siblings.
|
|
scopes: List[List[ET.Element]] = []
|
|
nested: set = set()
|
|
for media in self.root.findall('.//media'):
|
|
group = list(media.iter('title'))
|
|
if group:
|
|
scopes.append(group)
|
|
nested.update(id(t) for t in group)
|
|
main = [t for t in self.root.iter('title') if id(t) not in nested]
|
|
if main:
|
|
scopes.append(main)
|
|
|
|
reports = [
|
|
self._measure_title_scope(
|
|
scope,
|
|
safe_margin_x=safe_margin_x,
|
|
safe_margin_y=safe_margin_y,
|
|
min_font_size=min_font_size,
|
|
min_distance=min_distance,
|
|
max_distance=max_distance,
|
|
)
|
|
for scope in scopes
|
|
]
|
|
if len(reports) == 1:
|
|
return reports[0]
|
|
if not reports:
|
|
return self._measure_title_scope(
|
|
[],
|
|
safe_margin_x=safe_margin_x,
|
|
safe_margin_y=safe_margin_y,
|
|
min_font_size=min_font_size,
|
|
min_distance=min_distance,
|
|
max_distance=max_distance,
|
|
)
|
|
|
|
rank = {
|
|
'none': 0, 'render_tolerance': 1, 'warning': 2,
|
|
'probable': 3, 'severe': 4,
|
|
}
|
|
merged_issues = [i for r in reports for i in r['issues']]
|
|
summary = dict(reports[0]['summary'])
|
|
for r in reports[1:]:
|
|
for key, value in r['summary'].items():
|
|
summary[key] = summary.get(key, 0) + value
|
|
return {
|
|
'severity': max(
|
|
(r['severity'] for r in reports),
|
|
key=lambda s: rank.get(s, 0),
|
|
),
|
|
'issues': merged_issues,
|
|
'summary': summary,
|
|
}
|
|
|
|
def _measure_title_scope(
|
|
self,
|
|
elements: List[ET.Element],
|
|
*,
|
|
safe_margin_x: float,
|
|
safe_margin_y: float,
|
|
min_font_size: Optional[float],
|
|
min_distance: Optional[float],
|
|
max_distance: Optional[float],
|
|
) -> dict:
|
|
"""Measure and validate one group of titles sharing a time origin."""
|
|
titles = []
|
|
for elem in elements:
|
|
# enabled="0" never renders in Final Cut (see
|
|
# generate_subtitles_by_emphasis, which disables plain titles
|
|
# under an emphasis phrase instead of never creating them) — a
|
|
# title that is off by design must not count as a collision
|
|
# against the one drawn in its place.
|
|
if elem.get('enabled', '1') == '0':
|
|
continue
|
|
text_el = elem.find('text/text-style')
|
|
text = (text_el.text or '').strip() if text_el is not None else ''
|
|
style = elem.find('text-style-def/text-style')
|
|
font = style.get('font') if style is not None else None
|
|
face = style.get('fontFace') if style is not None else None
|
|
font_size = (
|
|
float(style.get('fontSize', '0')) if style is not None else 0.0
|
|
)
|
|
kerning = (
|
|
float(style.get('kerning', '0') or 0)
|
|
if style is not None else 0.0
|
|
)
|
|
|
|
x = y = 0.0
|
|
for param in elem.findall('param'):
|
|
if param.get('name') == 'Position' and param.get('value'):
|
|
parts = param.get('value').split()
|
|
if len(parts) >= 2:
|
|
x, y = float(parts[0]), float(parts[1])
|
|
|
|
start = self._parse_time(elem.get('offset', '0s')).to_seconds()
|
|
duration = self._parse_time(elem.get('duration', '0s')).to_seconds()
|
|
|
|
titles.append({
|
|
'text': text,
|
|
'font': font,
|
|
'face': face,
|
|
'font_size': font_size,
|
|
'kerning': kerning,
|
|
'x': x,
|
|
'y': y,
|
|
'start': start,
|
|
'end': start + duration,
|
|
'group': start + duration,
|
|
})
|
|
|
|
return validate_titles(
|
|
titles,
|
|
self.frame_width(),
|
|
self.frame_height(),
|
|
safe_margin_x=safe_margin_x,
|
|
safe_margin_y=safe_margin_y,
|
|
min_font_size=min_font_size,
|
|
min_distance=min_distance,
|
|
max_distance=max_distance,
|
|
)
|
|
|
|
# ========================================================================
|