feat(legendas): sub-frases por vírgula e empacotamento em compound clip

split_into_subphrases divide a frase na vírgula — onde a fala respira —
mas funde de volta o pedaço curto ("né?", "Então..."), que lê como parte
da frase anterior e não como bloco próprio.

wrap_titles_in_compound empacota os títulos de uma sub-frase num compound
clip, replicando a estrutura que o próprio Final Cut produz: o primeiro
título vira âncora do spine em offset 0, os demais penduram nele por lane,
e um ref-clip toma o lugar deles na lane original. Os offsets dos filhos
são rebaseados para o espaço de tempo da âncora, senão cada palavra
escorregaria pela diferença entre os dois start.

Junto: _filter_children_for_segment passa a filtrar também o <video> do
Clipe de Ajuste. Sem isso, cada corte subsequente duplicava o zoom em
todos os pedaços resultantes com o offset original intacto, e as cópias
desenhavam empilhadas na mesma posição da timeline.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-26 16:35:21 -04:00
co-authored by Claude Opus 5
parent 635d1bb553
commit 688bdeddb6
3 changed files with 157 additions and 0 deletions
+48
View File
@@ -316,6 +316,54 @@ def group_words_by_segment(
return groups return groups
def split_into_subphrases(
words: Sequence[dict],
min_words: int = 3,
) -> List[List[dict]]:
"""Split a sentence's *words* into sub-phrases at comma boundaries.
A comma is where a spoken sentence actually breathes, so it is the
natural seam for grouping subtitles — each sub-phrase becoming its own
on-screen block (and, downstream, its own compound clip).
The exception is the short tail: a fragment like "né?" or "Então..."
reads as part of the phrase before it, not as a phrase of its own, and
promoting it to its own block would flash a single word on screen. So a
piece shorter than *min_words* is merged back into its neighbour —
preferring the previous piece, falling back to the next one when the
short piece leads the sentence.
Returns one group per sub-phrase; a sentence with no comma comes back
as a single group.
"""
pieces: List[List[dict]] = []
current: List[dict] = []
for w in words:
current.append(w)
text = str(w.get('word') or w.get('text') or '')
if text.rstrip().endswith(','):
pieces.append(current)
current = []
if current:
pieces.append(current)
if len(pieces) <= 1:
return pieces
merged: List[List[dict]] = []
for piece in pieces:
if len(piece) < min_words and merged:
merged[-1].extend(piece)
else:
merged.append(piece)
# A short leading piece has no previous neighbour to fold into, so it
# folds forward instead.
if len(merged) > 1 and len(merged[0]) < min_words:
merged[1][:0] = merged[0]
merged.pop(0)
return merged
def segments_to_srt(segments: Sequence[dict]) -> str: def segments_to_srt(segments: Sequence[dict]) -> str:
"""Render transcript segments as an SRT string (for captions import).""" """Render transcript segments as an SRT string (for captions import)."""
+94
View File
@@ -12,6 +12,7 @@ from ..models import (
TimeValue, TimeValue,
) )
from .helpers import ( from .helpers import (
_dtd_insert,
_sanitize_xml_value, _sanitize_xml_value,
) )
@@ -122,6 +123,99 @@ class CompoundMixin:
return ref_clip return ref_clip
def wrap_titles_in_compound(
self,
parent_clip: ET.Element,
titles: List[ET.Element],
name: str = "Legenda",
) -> ET.Element:
"""Pack lane-nested *titles* of *parent_clip* into one compound clip.
A dynamic-subtitle sub-phrase is a dozen overlapping ``<title>``
elements stacked across as many lanes — legible on screen, unreadable
in the timeline. Collapsing each sub-phrase into a single compound
gives one bar per phrase to drag, mute or retime as a unit.
Mirrors the structure Final Cut itself produces for "New Compound
Clip" over stacked titles: the earliest title becomes the compound's
spine anchor at offset 0, the rest hang off it as lane children, and
a ``<ref-clip>`` takes their place in *parent_clip* on the anchor's
original lane.
Child offsets are rebased from *parent_clip*'s source-time space onto
the anchor's, since a lane child is anchored at its parent's
``start`` — leaving them untouched would shift every word of the
phrase by the gap between the two starts.
Args:
parent_clip: The spine clip the titles currently hang off.
titles: The ``<title>`` elements to pack; must all be direct
children of *parent_clip*.
name: Name for the resulting compound clip.
Returns:
The created ``<ref-clip>`` element, now in *parent_clip*.
"""
if not titles:
raise ValueError("No titles to wrap")
resources = self.root.find('.//resources')
if resources is None:
raise ValueError("No <resources> element found in FCPXML")
ordered = sorted(
titles, key=lambda t: self._parse_time(t.get('offset', '0s'))
)
anchor = ordered[0]
anchor_offset = self._parse_time(anchor.get('offset', '0s'))
anchor_start = self._parse_time(anchor.get('start', '0s'))
anchor_lane = anchor.get('lane')
total = TimeValue.zero()
for title in ordered:
rel = self._parse_time(title.get('offset', '0s')) - anchor_offset
end = rel + self._parse_time(title.get('duration', '0s'))
if end > total:
total = end
format_id = next(iter(self.formats), None) or 'r1'
media_id = self._unique_resource_id(resources, 'r_compound1')
media = ET.SubElement(resources, 'media')
media.set('id', media_id)
media.set('name', _sanitize_xml_value(name, 512))
media.set('uid', str(uuid.uuid4()).upper())
seq = ET.SubElement(media, 'sequence')
seq.set('format', format_id)
seq.set('duration', total.to_fcpxml())
seq.set('tcStart', '0s')
seq.set('tcFormat', 'NDF')
inner_spine = ET.SubElement(seq, 'spine')
for title in ordered:
parent_clip.remove(title)
anchor.set('offset', '0s')
if anchor_lane is not None:
del anchor.attrib['lane']
inner_spine.append(anchor)
for lane, title in enumerate(ordered[1:], start=1):
rel = self._parse_time(title.get('offset', '0s')) - anchor_offset
title.set('offset', (anchor_start + rel).to_fcpxml())
title.set('lane', str(lane))
anchor.append(title)
ref_clip = ET.Element('ref-clip')
ref_clip.set('ref', media_id)
if anchor_lane is not None:
ref_clip.set('lane', anchor_lane)
ref_clip.set('offset', anchor_offset.to_fcpxml())
ref_clip.set('name', _sanitize_xml_value(name, 512))
ref_clip.set('duration', total.to_fcpxml())
_dtd_insert(parent_clip, ref_clip)
return ref_clip
def flatten_compound_clip( def flatten_compound_clip(
self, self,
ref_clip_id: str, ref_clip_id: str,
+15
View File
@@ -41,6 +41,15 @@ class CutMixin:
cut (silence removal, filler removal) duplicates it into every cut (silence removal, filler removal) duplicates it into every
resulting piece, so the same word shows up several times across the resulting piece, so the same word shows up several times across the
edited timeline instead of once where it was placed. edited timeline instead of once where it was placed.
A lane-nested ``<video>`` zoom (the "Clipe de Ajuste" adjustment
layer ``add_zoom`` creates, ``role`` starting with ``"adjustments."``)
is the exact same phantom-duplicate case, keyed on ``offset``+
``duration`` like a keyword. Left unfiltered, every further cut
duplicates the zoom into every resulting piece with its original
offset untouched — each copy then draws at the same absolute
position, so two "Clipe de Ajuste" bars appear stacked on top of
each other in the timeline instead of the one real zoom window.
""" """
seg_end = seg_start + seg_duration seg_end = seg_start + seg_duration
to_remove = [] to_remove = []
@@ -54,6 +63,12 @@ class CutMixin:
title_offset = TimeValue.from_timecode(child.get('offset', '0s')) title_offset = TimeValue.from_timecode(child.get('offset', '0s'))
if title_offset < seg_start or title_offset >= seg_end: if title_offset < seg_start or title_offset >= seg_end:
to_remove.append(child) to_remove.append(child)
elif tag == 'video' and (child.get('role') or '').startswith('adjustments.'):
v_offset = TimeValue.from_timecode(child.get('offset', '0s'))
v_dur = TimeValue.from_timecode(child.get('duration', '0s'))
v_end = v_offset + v_dur
if v_end <= seg_start or v_offset >= seg_end:
to_remove.append(child)
elif tag == 'keyword': elif tag == 'keyword':
kw_start = TimeValue.from_timecode(child.get('start', '0s')) kw_start = TimeValue.from_timecode(child.get('start', '0s'))
kw_dur = TimeValue.from_timecode(child.get('duration', '0s')) kw_dur = TimeValue.from_timecode(child.get('duration', '0s'))