feat(legendas): liga compound_subphrases por padrão no pipeline

generate_dynamic_subtitles e a metade dinâmica de generate_subtitles_by_emphasis
passam a empacotar cada sub-frase da legenda dinâmica num compound clip por
padrão (compound_subphrases=True), completando o wrap_titles_in_compound
e split_into_subphrases do commit anterior — que ainda não tinham chamador
em produção.

Também torna validate_subtitle_layout ciente de compound clips: media cada
grupo (spine principal + cada <media> de compound) no seu próprio espaço de
tempo, em vez de uma varredura .//title global — sem isso, âncoras de
compounds diferentes liam offset "0s" e acusavam colisão espacial entre
frases que nunca dividem a tela, só porque compartilham o mesmo zero de
tempo local.

Testado ponta a ponta na gravação real (Mastopexia): 12 compounds, 41
títulos todos empacotados, zero soltos, zero IDs duplicados, DTD válida.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-26 17:05:20 -04:00
co-authored by Claude Sonnet 5
parent 688bdeddb6
commit c99274895c
3 changed files with 355 additions and 108 deletions
+157 -27
View File
@@ -3,6 +3,7 @@
Extraído de writer.py — ver fcpxml/writer/__init__.py para o conjunto.
"""
import random
import re
import unicodedata
import uuid
@@ -20,7 +21,7 @@ from ..text_layout import (
compose_sentence,
layout_sentence,
)
from ..transcribe import group_words_by_segment
from ..transcribe import group_words_by_segment, split_into_subphrases
from .helpers import _dtd_insert, _sanitize_xml_value
@@ -221,6 +222,7 @@ class TitlesMixin:
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
animated: bool = True,
size_param: Optional[float] = None,
role: Optional[str] = None,
) -> ET.Element:
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
@@ -238,6 +240,8 @@ class TitlesMixin:
elem.set('name', _sanitize_xml_value(name, 256))
elem.set('start', self._TEXT_TITLE_START)
elem.set('duration', duration.to_fcpxml())
if role:
elem.set('role', _sanitize_xml_value(role, 256))
if position:
param = ET.SubElement(elem, 'param')
@@ -328,6 +332,7 @@ class TitlesMixin:
animated: bool = True,
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
size_param: Optional[float] = None,
role: Optional[str] = None,
) -> ET.Element:
"""Add a single static "Text" (Basic Text) title over *parent_clip*.
@@ -365,6 +370,7 @@ class TitlesMixin:
animated=animated,
font_scale=font_scale,
size_param=size_param,
role=role,
)
_dtd_insert(parent, title)
return title
@@ -375,6 +381,10 @@ class TitlesMixin:
words: List[Dict[str, Any]],
config: Optional['DynamicSubtitleConfig'] = None,
segments: Optional[List[Dict[str, Any]]] = None,
role: Optional[str] = None,
configs: Optional[List['DynamicSubtitleConfig']] = None,
compound_subphrases: bool = False,
subphrase_min_words: int = 3,
) -> List[ET.Element]:
"""Generate progressive-reveal subtitle titles, one per word.
@@ -418,8 +428,16 @@ class TitlesMixin:
Returns:
The list of created ``<title>`` elements, in chronological order.
"""
if config is None:
config = DynamicSubtitleConfig()
# ``configs`` (a list of registered, active layouts) takes precedence
# over the single ``config`` — with 2+ items, each block picks one at
# random below; with 0 or 1, behaviour is identical to a single fixed
# config, so old callers passing only ``config`` are unaffected.
if configs:
layout_configs = list(configs)
elif config is not None:
layout_configs = [config]
else:
layout_configs = [DynamicSubtitleConfig()]
if not words:
return []
@@ -449,35 +467,54 @@ class TitlesMixin:
# next block — the sub-sentence split that keeps long sentences from
# spilling off screen.
sentences = group_words_by_segment(words, segments or [])
box = LayoutBox.for_frame(
self.frame_width(), self.frame_height(),
band_height=config.band_height,
center_y=config.block_center_y,
)
# "phrase" is the progressive composition the reference reel uses: one
# title per LINE ("que vão" / "melhorar" / "sua legenda"), the key word
# set large in a display italic. "word" is the older one-title-per-word
# rhythm, kept for callers that want every word to land on its own.
phrase_mode = getattr(config, 'granularity', 'phrase') == 'phrase'
# A comma is where the sentence breathes, so it is also where the
# phrase should be packed into its own compound clip downstream.
if compound_subphrases:
sentences = [
sub
for sentence in sentences
for sub in split_into_subphrases(sentence, subphrase_min_words)
]
def lay_out(pending: List[Dict]):
def box_for(cfg: 'DynamicSubtitleConfig') -> LayoutBox:
return LayoutBox.for_frame(
self.frame_width(), self.frame_height(),
band_height=cfg.band_height,
center_y=cfg.block_center_y,
)
def lay_out(pending: List[Dict], cfg: 'DynamicSubtitleConfig', box: LayoutBox):
"""Place what fits; return (units, still-unplaced words)."""
if phrase_mode:
# "phrase" is the progressive composition the reference reel uses:
# one title per LINE ("que vão" / "melhorar" / "sua legenda"), the
# key word set large in a display italic. "word" is the older
# one-title-per-word rhythm, kept for callers that want every word
# to land on its own.
if getattr(cfg, 'granularity', 'phrase') == 'phrase':
composition = compose_sentence(
pending, config.style, box, line_gap=config.line_gap,
pending, cfg.style, box, line_gap=cfg.line_gap,
)
return composition.blocks, composition.overflow
layout = layout_sentence(pending, config.style, box)
layout = layout_sentence(pending, cfg.style, box)
return layout.placed, layout.overflow
blocks: List[List[Any]] = []
for sentence in sentences:
block_configs: List['DynamicSubtitleConfig'] = []
block_sentences: List[int] = []
for sentence_index, sentence in enumerate(sentences):
remaining = list(sentence)
while remaining:
units, remaining = lay_out(remaining)
# Each block independently samples a layout from the active
# set — the visual variety the user asked for. A single
# active layout always resolves to itself, so this is a
# no-op for the common case.
active_config = layout_configs[random.randrange(len(layout_configs))]
units, remaining = lay_out(remaining, active_config, box_for(active_config))
if not units:
break
blocks.append(units)
block_configs.append(active_config)
block_sentences.append(sentence_index)
if not blocks:
return []
@@ -523,7 +560,15 @@ class TitlesMixin:
media_origin = self._parse_time(parent.get('start', '0s'))
created: List[ET.Element] = []
for units, block_end in zip(blocks, block_ends):
by_sentence: Dict[int, List[ET.Element]] = {}
for units, block_end, block_config, sentence_index in zip(
blocks, block_ends, block_configs, block_sentences
):
# A ``titles.*`` sub-role keeps these as titles (never closed
# captions) while grouping them in the role index and tinting
# their lane. An explicit ``role`` argument overrides every
# block; otherwise each block uses its own sampled layout's role.
block_role = role or getattr(block_config, "role", None) or "titles.dinamicas"
for index, unit in enumerate(units):
relative_offset = self.snap_seconds_to_frame(unit.start)
duration = block_end - relative_offset
@@ -543,19 +588,34 @@ class TitlesMixin:
duration,
lane=lane,
name=f"caption_{uuid.uuid4().hex[:8]}",
position=unit.position_param(config.text_scale),
font=unit.font or config.style.font,
position=unit.position_param(block_config.text_scale),
font=unit.font or block_config.style.font,
font_size=int(round(unit.font_size)),
font_color=unit.color or config.style.active_color,
bold=config.style.bold,
font_color=unit.color or block_config.style.active_color,
bold=block_config.style.bold,
face=unit.face,
kerning=unit.kerning,
font_scale=config.text_scale,
font_scale=block_config.text_scale,
role=block_role,
)
_dtd_insert(parent, title)
created.append(title)
by_sentence.setdefault(sentence_index, []).append(title)
if getattr(config, 'validate', False):
# One compound per sub-phrase: a dozen stacked title bars collapse
# into a single one that can be dragged, muted or retimed as a unit.
if compound_subphrases:
for sentence_index in sorted(by_sentence):
group = by_sentence[sentence_index]
label = " ".join(
str(w.get('word') or w.get('text') or '')
for w in sentences[sentence_index]
).strip()
self.wrap_titles_in_compound(
parent, group, name=label[:60] or "Legenda"
)
if any(getattr(cfg, 'validate', False) for cfg in layout_configs):
report = self.validate_subtitle_layout()
if blocking(report["severity"]):
raise ValueError(
@@ -586,8 +646,78 @@ class TitlesMixin:
Returns the ``collision.validate_titles`` report: ``severity`` (worst
bucket), ``issues`` (spec-16 occurrences) and ``summary`` (counts).
"""
# A compound clip carries its own time origin: a title inside one is
# offset from that compound's start, not the sequence's. Measured in
# one flat pass, the anchors of two different compounds both read as
# "0s" and collide on paper while sitting seconds apart on the
# timeline. Each compound is therefore measured as its own scope,
# which is also where its titles can actually overlap — a title can
# only share the screen with its own compound's siblings.
scopes: List[List[ET.Element]] = []
nested: set = set()
for media in self.root.findall('.//media'):
group = list(media.iter('title'))
if group:
scopes.append(group)
nested.update(id(t) for t in group)
main = [t for t in self.root.iter('title') if id(t) not in nested]
if main:
scopes.append(main)
reports = [
self._measure_title_scope(
scope,
safe_margin_x=safe_margin_x,
safe_margin_y=safe_margin_y,
min_font_size=min_font_size,
min_distance=min_distance,
max_distance=max_distance,
)
for scope in scopes
]
if len(reports) == 1:
return reports[0]
if not reports:
return self._measure_title_scope(
[],
safe_margin_x=safe_margin_x,
safe_margin_y=safe_margin_y,
min_font_size=min_font_size,
min_distance=min_distance,
max_distance=max_distance,
)
rank = {
'none': 0, 'render_tolerance': 1, 'warning': 2,
'probable': 3, 'severe': 4,
}
merged_issues = [i for r in reports for i in r['issues']]
summary = dict(reports[0]['summary'])
for r in reports[1:]:
for key, value in r['summary'].items():
summary[key] = summary.get(key, 0) + value
return {
'severity': max(
(r['severity'] for r in reports),
key=lambda s: rank.get(s, 0),
),
'issues': merged_issues,
'summary': summary,
}
def _measure_title_scope(
self,
elements: List[ET.Element],
*,
safe_margin_x: float,
safe_margin_y: float,
min_font_size: Optional[float],
min_distance: Optional[float],
max_distance: Optional[float],
) -> dict:
"""Measure and validate one group of titles sharing a time origin."""
titles = []
for elem in self.root.iter('title'):
for elem in elements:
# enabled="0" never renders in Final Cut (see
# generate_subtitles_by_emphasis, which disables plain titles
# under an emphasis phrase instead of never creating them) — a