diff --git a/code/fcpxml/transcribe.py b/code/fcpxml/transcribe.py
index 182938b..54ac367 100755
--- a/code/fcpxml/transcribe.py
+++ b/code/fcpxml/transcribe.py
@@ -316,6 +316,54 @@ def group_words_by_segment(
return groups
+def split_into_subphrases(
+ words: Sequence[dict],
+ min_words: int = 3,
+) -> List[List[dict]]:
+ """Split a sentence's *words* into sub-phrases at comma boundaries.
+
+ A comma is where a spoken sentence actually breathes, so it is the
+ natural seam for grouping subtitles — each sub-phrase becoming its own
+ on-screen block (and, downstream, its own compound clip).
+
+ The exception is the short tail: a fragment like "né?" or "Então..."
+ reads as part of the phrase before it, not as a phrase of its own, and
+ promoting it to its own block would flash a single word on screen. So a
+ piece shorter than *min_words* is merged back into its neighbour —
+ preferring the previous piece, falling back to the next one when the
+ short piece leads the sentence.
+
+ Returns one group per sub-phrase; a sentence with no comma comes back
+ as a single group.
+ """
+ pieces: List[List[dict]] = []
+ current: List[dict] = []
+ for w in words:
+ current.append(w)
+ text = str(w.get('word') or w.get('text') or '')
+ if text.rstrip().endswith(','):
+ pieces.append(current)
+ current = []
+ if current:
+ pieces.append(current)
+
+ if len(pieces) <= 1:
+ return pieces
+
+ merged: List[List[dict]] = []
+ for piece in pieces:
+ if len(piece) < min_words and merged:
+ merged[-1].extend(piece)
+ else:
+ merged.append(piece)
+ # A short leading piece has no previous neighbour to fold into, so it
+ # folds forward instead.
+ if len(merged) > 1 and len(merged[0]) < min_words:
+ merged[1][:0] = merged[0]
+ merged.pop(0)
+ return merged
+
+
def segments_to_srt(segments: Sequence[dict]) -> str:
"""Render transcript segments as an SRT string (for captions import)."""
diff --git a/code/fcpxml/writer/compound.py b/code/fcpxml/writer/compound.py
index f3bae97..8a1e662 100644
--- a/code/fcpxml/writer/compound.py
+++ b/code/fcpxml/writer/compound.py
@@ -12,6 +12,7 @@ from ..models import (
TimeValue,
)
from .helpers import (
+ _dtd_insert,
_sanitize_xml_value,
)
@@ -122,6 +123,99 @@ class CompoundMixin:
return ref_clip
+ def wrap_titles_in_compound(
+ self,
+ parent_clip: ET.Element,
+ titles: List[ET.Element],
+ name: str = "Legenda",
+ ) -> ET.Element:
+ """Pack lane-nested *titles* of *parent_clip* into one compound clip.
+
+ A dynamic-subtitle sub-phrase is a dozen overlapping ``
``
+ elements stacked across as many lanes — legible on screen, unreadable
+ in the timeline. Collapsing each sub-phrase into a single compound
+ gives one bar per phrase to drag, mute or retime as a unit.
+
+ Mirrors the structure Final Cut itself produces for "New Compound
+ Clip" over stacked titles: the earliest title becomes the compound's
+ spine anchor at offset 0, the rest hang off it as lane children, and
+ a ```` takes their place in *parent_clip* on the anchor's
+ original lane.
+
+ Child offsets are rebased from *parent_clip*'s source-time space onto
+ the anchor's, since a lane child is anchored at its parent's
+ ``start`` — leaving them untouched would shift every word of the
+ phrase by the gap between the two starts.
+
+ Args:
+ parent_clip: The spine clip the titles currently hang off.
+ titles: The ```` elements to pack; must all be direct
+ children of *parent_clip*.
+ name: Name for the resulting compound clip.
+
+ Returns:
+ The created ```` element, now in *parent_clip*.
+ """
+ if not titles:
+ raise ValueError("No titles to wrap")
+ resources = self.root.find('.//resources')
+ if resources is None:
+ raise ValueError("No element found in FCPXML")
+
+ ordered = sorted(
+ titles, key=lambda t: self._parse_time(t.get('offset', '0s'))
+ )
+ anchor = ordered[0]
+ anchor_offset = self._parse_time(anchor.get('offset', '0s'))
+ anchor_start = self._parse_time(anchor.get('start', '0s'))
+ anchor_lane = anchor.get('lane')
+
+ total = TimeValue.zero()
+ for title in ordered:
+ rel = self._parse_time(title.get('offset', '0s')) - anchor_offset
+ end = rel + self._parse_time(title.get('duration', '0s'))
+ if end > total:
+ total = end
+
+ format_id = next(iter(self.formats), None) or 'r1'
+ media_id = self._unique_resource_id(resources, 'r_compound1')
+
+ media = ET.SubElement(resources, 'media')
+ media.set('id', media_id)
+ media.set('name', _sanitize_xml_value(name, 512))
+ media.set('uid', str(uuid.uuid4()).upper())
+
+ seq = ET.SubElement(media, 'sequence')
+ seq.set('format', format_id)
+ seq.set('duration', total.to_fcpxml())
+ seq.set('tcStart', '0s')
+ seq.set('tcFormat', 'NDF')
+ inner_spine = ET.SubElement(seq, 'spine')
+
+ for title in ordered:
+ parent_clip.remove(title)
+
+ anchor.set('offset', '0s')
+ if anchor_lane is not None:
+ del anchor.attrib['lane']
+ inner_spine.append(anchor)
+
+ for lane, title in enumerate(ordered[1:], start=1):
+ rel = self._parse_time(title.get('offset', '0s')) - anchor_offset
+ title.set('offset', (anchor_start + rel).to_fcpxml())
+ title.set('lane', str(lane))
+ anchor.append(title)
+
+ ref_clip = ET.Element('ref-clip')
+ ref_clip.set('ref', media_id)
+ if anchor_lane is not None:
+ ref_clip.set('lane', anchor_lane)
+ ref_clip.set('offset', anchor_offset.to_fcpxml())
+ ref_clip.set('name', _sanitize_xml_value(name, 512))
+ ref_clip.set('duration', total.to_fcpxml())
+ _dtd_insert(parent_clip, ref_clip)
+ return ref_clip
+
def flatten_compound_clip(
self,
ref_clip_id: str,
diff --git a/code/fcpxml/writer/cut.py b/code/fcpxml/writer/cut.py
index 29ff1d6..6f8268c 100644
--- a/code/fcpxml/writer/cut.py
+++ b/code/fcpxml/writer/cut.py
@@ -41,6 +41,15 @@ class CutMixin:
cut (silence removal, filler removal) duplicates it into every
resulting piece, so the same word shows up several times across the
edited timeline instead of once where it was placed.
+
+ A lane-nested ``