feat: etapa 5 do assistente — revisão de ênfases com timeline

Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-19 21:29:27 -04:00
co-authored by Claude Opus 5
parent e7748c2c58
commit 1bebee4359
31 changed files with 4622 additions and 83 deletions
+63 -2
View File
@@ -2218,13 +2218,23 @@ class FCPXMLModifier:
seg_start: 'TimeValue',
seg_duration: 'TimeValue',
) -> None:
"""Remove markers/keywords from *clip* that fall outside the segment range.
"""Remove markers/keywords/titles from *clip* that fall outside the segment range.
After ``split_clip`` deepcopy's the original clip into each segment, every
segment inherits all child elements. Markers whose ``start`` falls outside
``[seg_start, seg_start + seg_duration)`` are phantom duplicates and must be
removed. Keywords that partially overlap get their ``start``/``duration``
clamped to the segment boundaries.
A lane-nested ``<title>`` (a "text" voice action's on-screen callout,
or a caption from an earlier `generate_dynamic_subtitles` pass) is
the same kind of phantom duplicate, just keyed on ``offset`` instead
of ``start`` — its offset lives in the same source-media coordinate
space as a marker's ``start`` (see ``add_text_title``/``add_marker``,
both anchored at ``parent.start``). Left unfiltered, every further
cut (silence removal, filler removal) duplicates it into every
resulting piece, so the same word shows up several times across the
edited timeline instead of once where it was placed.
"""
seg_end = seg_start + seg_duration
to_remove = []
@@ -2234,6 +2244,10 @@ class FCPXMLModifier:
child_start = TimeValue.from_timecode(child.get('start', '0s'))
if child_start < seg_start or child_start >= seg_end:
to_remove.append(child)
elif tag == 'title':
title_offset = TimeValue.from_timecode(child.get('offset', '0s'))
if title_offset < seg_start or title_offset >= seg_end:
to_remove.append(child)
elif tag == 'keyword':
kw_start = TimeValue.from_timecode(child.get('start', '0s'))
kw_dur = TimeValue.from_timecode(child.get('duration', '0s'))
@@ -2304,6 +2318,7 @@ class FCPXMLModifier:
self._filter_children_for_segment(
new_clip, current_start, segment_duration
)
self._reassign_text_style_ids(new_clip)
spine.insert(clip_index + len(new_clips), new_clip)
new_clips.append(new_clip)
@@ -2404,6 +2419,7 @@ class FCPXMLModifier:
new_clip.set('start', seg_start.to_fcpxml())
new_clip.set('duration', seg_duration.to_fcpxml())
self._filter_children_for_segment(new_clip, seg_start, seg_duration)
self._reassign_text_style_ids(new_clip)
spine.insert(clip_index + len(new_clips), new_clip)
new_clips.append(new_clip)
current_offset = current_offset + seg_duration
@@ -2946,6 +2962,7 @@ class FCPXMLModifier:
('-469658744/1000000000s', '0'),
('12328542033/1000000000s', '1'),
)
_TEXT_SIZE_KEY = '9999/10003/13260/3296672360/5/3296672362/3'
def _ensure_text_title_effect(self, resources: ET.Element) -> str:
"""Return the resource id of the "Text" (Basic Text) effect, creating it if absent."""
@@ -2995,6 +3012,34 @@ class FCPXMLModifier:
self._text_style_ids.add(candidate)
return candidate
def _reassign_text_style_ids(self, clip: ET.Element) -> None:
"""Give every ``<text-style-def>`` inside a just-deepcopy'd *clip* a
fresh document-unique id, repointing any ``<text-style ref="...">``
in the same subtree that pointed at the old one.
``split_clip``/``cut_clip_ranges`` deepcopy the clip once per
resulting segment, so a clip carrying a ``<title>`` (from a "text"
voice action) keeps the exact same ``text-style-def id`` in every
copy. A single cut is harmless — but the batch chain re-cuts the
same clip at each step (silence removal, filler removal, dynamic
subtitles), and every pass multiplies the duplicate, so the DTD
validator eventually rejects the file with "ID ... already
defined". Regenerating here, at the only place copies are made,
fixes it for every caller instead of each one having to remember to.
"""
for style_def in clip.findall('.//text-style-def'):
old_id = style_def.get('id')
if not old_id:
continue
slug = old_id[3:] if old_id.startswith('ts_') else old_id
slug = re.sub(r'_\d+$', '', slug) # drop a prior _<N> counter
new_id = self._unique_text_style_id(slug)
if new_id == old_id:
continue
style_def.set('id', new_id)
for ref_el in clip.findall(f".//text-style[@ref='{old_id}']"):
ref_el.set('ref', new_id)
def _make_text_title_clip(
self,
effect_id: str,
@@ -3012,6 +3057,8 @@ class FCPXMLModifier:
face: Optional[str] = None,
kerning: Optional[float] = None,
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
animated: bool = True,
size_param: Optional[float] = None,
) -> ET.Element:
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
@@ -3042,9 +3089,12 @@ class FCPXMLModifier:
param.set('key', key)
param.set('value', value)
animation_params = {'Opacity', 'Speed', 'Apply Speed'}
for param_name, param_key, param_value in self._TEXT_TITLE_PARAMS:
if not animated and param_name in animation_params:
continue
_add_param(param_name, param_key, param_value)
if param_name == 'Speed':
if animated and param_name == 'Speed':
# "Custom Speed" lands between "Speed" and "Apply Speed" and
# carries a <keyframeAnimation> child instead of a value.
cs = ET.SubElement(elem, 'param')
@@ -3056,6 +3106,9 @@ class FCPXMLModifier:
kf.set('time', kf_time)
kf.set('value', kf_value)
if size_param is not None:
_add_param('Size', self._TEXT_SIZE_KEY, f"{float(size_param):g}")
text_el = ET.SubElement(elem, 'text')
ts_id = self._unique_text_style_id(name)
run = ET.SubElement(text_el, 'text-style')
@@ -3109,6 +3162,10 @@ class FCPXMLModifier:
font_size: int = 196,
font_color: str = '1 1 1 1',
bold: bool = True,
face: Optional[str] = None,
animated: bool = True,
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
size_param: Optional[float] = None,
) -> ET.Element:
"""Add a single static "Text" (Basic Text) title over *parent_clip*.
@@ -3142,6 +3199,10 @@ class FCPXMLModifier:
font_size=font_size,
font_color=font_color,
bold=bold,
face=face,
animated=animated,
font_scale=font_scale,
size_param=size_param,
)
_dtd_insert(parent, title)
return title