diff --git a/code/fcpxml/transcribe.py b/code/fcpxml/transcribe.py index 182938b..54ac367 100755 --- a/code/fcpxml/transcribe.py +++ b/code/fcpxml/transcribe.py @@ -316,6 +316,54 @@ def group_words_by_segment( return groups +def split_into_subphrases( + words: Sequence[dict], + min_words: int = 3, +) -> List[List[dict]]: + """Split a sentence's *words* into sub-phrases at comma boundaries. + + A comma is where a spoken sentence actually breathes, so it is the + natural seam for grouping subtitles — each sub-phrase becoming its own + on-screen block (and, downstream, its own compound clip). + + The exception is the short tail: a fragment like "né?" or "Então..." + reads as part of the phrase before it, not as a phrase of its own, and + promoting it to its own block would flash a single word on screen. So a + piece shorter than *min_words* is merged back into its neighbour — + preferring the previous piece, falling back to the next one when the + short piece leads the sentence. + + Returns one group per sub-phrase; a sentence with no comma comes back + as a single group. + """ + pieces: List[List[dict]] = [] + current: List[dict] = [] + for w in words: + current.append(w) + text = str(w.get('word') or w.get('text') or '') + if text.rstrip().endswith(','): + pieces.append(current) + current = [] + if current: + pieces.append(current) + + if len(pieces) <= 1: + return pieces + + merged: List[List[dict]] = [] + for piece in pieces: + if len(piece) < min_words and merged: + merged[-1].extend(piece) + else: + merged.append(piece) + # A short leading piece has no previous neighbour to fold into, so it + # folds forward instead. + if len(merged) > 1 and len(merged[0]) < min_words: + merged[1][:0] = merged[0] + merged.pop(0) + return merged + + def segments_to_srt(segments: Sequence[dict]) -> str: """Render transcript segments as an SRT string (for captions import).""" diff --git a/code/fcpxml/writer/compound.py b/code/fcpxml/writer/compound.py index f3bae97..8a1e662 100644 --- a/code/fcpxml/writer/compound.py +++ b/code/fcpxml/writer/compound.py @@ -12,6 +12,7 @@ from ..models import ( TimeValue, ) from .helpers import ( + _dtd_insert, _sanitize_xml_value, ) @@ -122,6 +123,99 @@ class CompoundMixin: return ref_clip + def wrap_titles_in_compound( + self, + parent_clip: ET.Element, + titles: List[ET.Element], + name: str = "Legenda", + ) -> ET.Element: + """Pack lane-nested *titles* of *parent_clip* into one compound clip. + + A dynamic-subtitle sub-phrase is a dozen overlapping ```` + elements stacked across as many lanes — legible on screen, unreadable + in the timeline. Collapsing each sub-phrase into a single compound + gives one bar per phrase to drag, mute or retime as a unit. + + Mirrors the structure Final Cut itself produces for "New Compound + Clip" over stacked titles: the earliest title becomes the compound's + spine anchor at offset 0, the rest hang off it as lane children, and + a ``<ref-clip>`` takes their place in *parent_clip* on the anchor's + original lane. + + Child offsets are rebased from *parent_clip*'s source-time space onto + the anchor's, since a lane child is anchored at its parent's + ``start`` — leaving them untouched would shift every word of the + phrase by the gap between the two starts. + + Args: + parent_clip: The spine clip the titles currently hang off. + titles: The ``<title>`` elements to pack; must all be direct + children of *parent_clip*. + name: Name for the resulting compound clip. + + Returns: + The created ``<ref-clip>`` element, now in *parent_clip*. + """ + if not titles: + raise ValueError("No titles to wrap") + resources = self.root.find('.//resources') + if resources is None: + raise ValueError("No <resources> element found in FCPXML") + + ordered = sorted( + titles, key=lambda t: self._parse_time(t.get('offset', '0s')) + ) + anchor = ordered[0] + anchor_offset = self._parse_time(anchor.get('offset', '0s')) + anchor_start = self._parse_time(anchor.get('start', '0s')) + anchor_lane = anchor.get('lane') + + total = TimeValue.zero() + for title in ordered: + rel = self._parse_time(title.get('offset', '0s')) - anchor_offset + end = rel + self._parse_time(title.get('duration', '0s')) + if end > total: + total = end + + format_id = next(iter(self.formats), None) or 'r1' + media_id = self._unique_resource_id(resources, 'r_compound1') + + media = ET.SubElement(resources, 'media') + media.set('id', media_id) + media.set('name', _sanitize_xml_value(name, 512)) + media.set('uid', str(uuid.uuid4()).upper()) + + seq = ET.SubElement(media, 'sequence') + seq.set('format', format_id) + seq.set('duration', total.to_fcpxml()) + seq.set('tcStart', '0s') + seq.set('tcFormat', 'NDF') + inner_spine = ET.SubElement(seq, 'spine') + + for title in ordered: + parent_clip.remove(title) + + anchor.set('offset', '0s') + if anchor_lane is not None: + del anchor.attrib['lane'] + inner_spine.append(anchor) + + for lane, title in enumerate(ordered[1:], start=1): + rel = self._parse_time(title.get('offset', '0s')) - anchor_offset + title.set('offset', (anchor_start + rel).to_fcpxml()) + title.set('lane', str(lane)) + anchor.append(title) + + ref_clip = ET.Element('ref-clip') + ref_clip.set('ref', media_id) + if anchor_lane is not None: + ref_clip.set('lane', anchor_lane) + ref_clip.set('offset', anchor_offset.to_fcpxml()) + ref_clip.set('name', _sanitize_xml_value(name, 512)) + ref_clip.set('duration', total.to_fcpxml()) + _dtd_insert(parent_clip, ref_clip) + return ref_clip + def flatten_compound_clip( self, ref_clip_id: str, diff --git a/code/fcpxml/writer/cut.py b/code/fcpxml/writer/cut.py index 29ff1d6..6f8268c 100644 --- a/code/fcpxml/writer/cut.py +++ b/code/fcpxml/writer/cut.py @@ -41,6 +41,15 @@ class CutMixin: cut (silence removal, filler removal) duplicates it into every resulting piece, so the same word shows up several times across the edited timeline instead of once where it was placed. + + A lane-nested ``<video>`` zoom (the "Clipe de Ajuste" adjustment + layer ``add_zoom`` creates, ``role`` starting with ``"adjustments."``) + is the exact same phantom-duplicate case, keyed on ``offset``+ + ``duration`` like a keyword. Left unfiltered, every further cut + duplicates the zoom into every resulting piece with its original + offset untouched — each copy then draws at the same absolute + position, so two "Clipe de Ajuste" bars appear stacked on top of + each other in the timeline instead of the one real zoom window. """ seg_end = seg_start + seg_duration to_remove = [] @@ -54,6 +63,12 @@ class CutMixin: title_offset = TimeValue.from_timecode(child.get('offset', '0s')) if title_offset < seg_start or title_offset >= seg_end: to_remove.append(child) + elif tag == 'video' and (child.get('role') or '').startswith('adjustments.'): + v_offset = TimeValue.from_timecode(child.get('offset', '0s')) + v_dur = TimeValue.from_timecode(child.get('duration', '0s')) + v_end = v_offset + v_dur + if v_end <= seg_start or v_offset >= seg_end: + to_remove.append(child) elif tag == 'keyword': kw_start = TimeValue.from_timecode(child.get('start', '0s')) kw_dur = TimeValue.from_timecode(child.get('duration', '0s'))