feat: etapa 5 do assistente — revisão de ênfases com timeline

Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-19 21:29:27 -04:00
co-authored by Claude Opus 5
parent e7748c2c58
commit 1bebee4359
31 changed files with 4622 additions and 83 deletions
+60 -7
View File
@@ -15,6 +15,7 @@ from typing import Any, Sequence
from mcp.types import TextContent
from fcpxml.media_intel import media_src_to_path
from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config
from fcpxml.models import (
DuplicateGroup,
FlashFrame,
@@ -25,6 +26,7 @@ from fcpxml.models import (
)
from fcpxml.parser import FCPXMLParser
from fcpxml.rough_cut import RoughCutGenerator
from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text
from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe
from fcpxml.writer import FCPXMLModifier
@@ -639,28 +641,79 @@ def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str:
rel_end = action.end - clip_start
if action.kind == "zoom":
config = load_voice_analysis_config()
# Only forward an explicit ease — otherwise add_zoom's own default
# (a fast ramp in, instant snap back out) is what should apply.
zoom_args = {}
if action.params.get("ease") is not None:
zoom_args["ease"] = float(action.params["ease"])
if action.params.get("ease_out") is not None:
zoom_args["ease_out"] = float(action.params["ease_out"])
zoom_args = {
"ease": float(action.params.get("ease", config["zoom_ease_in"])),
"ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])),
}
mode = str(action.params.get("mode", config["zoom_mode"]))
if mode == "in":
zoom_args["hold_at_end"] = True
zoom_args["start_at_peak"] = False
elif mode == "out":
zoom_args["hold_at_end"] = False
zoom_args["start_at_peak"] = True
elif mode == "in_out":
zoom_args["hold_at_end"] = False
zoom_args["start_at_peak"] = False
modifier.add_zoom(
clip_id=clip_el,
start=rel_start,
end=rel_end,
scale=float(action.params.get("scale", 1.3)),
scale=float(action.params.get("scale", config["zoom_scale"])),
**zoom_args,
)
return f"zoom {action.params.get('scale', 1.3):.2f}x"
return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x"
if action.kind == "text":
# Default to the "Legendas Dinâmicas" emphasis style (the font used
# to highlight a word in the captions) rather than a hardcoded
# Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of
# the video's on-screen text instead of looking like a stray default
# title. Any of these the action itself specifies still wins.
subtitle_cfg = load_dynamic_subtitle_config()
font = action.params.get("font", subtitle_cfg["emphasis_font"])
face = action.params.get("face", subtitle_cfg["emphasis_face"])
font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0)
requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"]))
requested_kerning = float(action.params.get("kerning", 0.0) or 0.0)
# Voice-action callouts are not part of the dynamic subtitle block.
# When omitted, put them above the subtitle band and shrink wide
# phrases to the title-safe width. The previous default (Position 0 0,
# full emphasis size) made long callouts like "PRÓTESES DE SILICONE"
# collide with captions and run off both sides of a vertical frame.
emitted_size = requested_size * font_scale
emitted_kerning = requested_kerning * font_scale
safe_width = modifier.frame_width() * 0.90
width = measure_text(
action.params["content"],
emitted_size,
bold=bool(action.params.get("bold", False)),
kerning=emitted_kerning,
font=font,
face=face,
)
font_size = requested_size
if width > safe_width and width > 0:
font_size = max(32, int(requested_size * safe_width / width))
position = action.params.get("position")
if not position:
position = f"0 {modifier.frame_height() * 0.23:g}"
modifier.add_text_title(
clip_el,
action.params["content"],
offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(),
position=position,
font=font,
font_size=font_size,
font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]),
face=face,
bold=action.params.get("bold", False),
)
return f"text \"{action.params['content'][:24]}\""
+176 -10
View File
@@ -6,13 +6,14 @@ Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalo
from __future__ import annotations
import json
import re
from pathlib import Path
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.media_intel import media_src_to_path
from fcpxml.model_manager import load_dynamic_subtitle_config
from fcpxml.model_manager import load_dynamic_subtitle_config, load_plain_subtitle_config
from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle
from fcpxml.writer import FCPXMLModifier
from server_tools._shared import (
@@ -69,9 +70,76 @@ TOOLS = [
"required": ["filepath"]
}
),
Tool(
name="generate_plain_subtitles",
description="Generate simple editable FCPXML text-title subtitles, synchronized to transcript words but without visual build-in/build-out effects. Words are grouped into short blocks, placed at a configurable vertical position, and written as static Text titles rather than SRT captions.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"language": {"type": "string", "description": "ISO language code hint (e.g. 'pt'); auto-detected if omitted"},
"font": {"type": "string", "description": "Text font family. Falls back to saved plain-subtitle config."},
"font_size": {"type": "integer", "description": "Font size in canvas points. Falls back to saved plain-subtitle config."},
"font_color": {"type": "string", "description": "RGBA (0-1, space-separated). Falls back to saved plain-subtitle config."},
"max_words": {"type": "integer", "description": "Maximum words per subtitle block. Falls back to saved plain-subtitle config."},
"position_y": {"type": "number", "description": "Vertical title position in canvas points; negative sits lower in frame."},
"uppercase": {"type": "boolean", "description": "Render text in uppercase."},
"keep_punctuation": {"type": "boolean", "description": "Keep punctuation such as comma and period."},
"text_scale": {"type": "number", "description": "Template font-size scale. Falls back to saved plain-subtitle config."},
"output_path": {"type": "string", "description": "Output path (default: adds _plain_subtitles suffix)"},
},
"required": ["filepath"]
}
),
]
_PUNCT_RE = re.compile(r"[^\w\sÀ-ÖØ-öø-ÿ]", re.UNICODE)
def _words_overlapping_clip(words: Sequence[dict], start: float, end: float) -> list[dict]:
"""Return transcript words that overlap a source window, rebased to it."""
clip_words: list[dict] = []
for w in words:
word_start = float(w.get("start", 0.0))
word_end = float(w.get("end", word_start))
if word_end <= start or word_start >= end:
continue
clip_words.append(
{
"word": w.get("word", ""),
"start": max(0.0, word_start - start),
"end": max(0.0, min(word_end, end) - start),
}
)
return clip_words
def _plain_word_text(word: str, *, uppercase: bool, keep_punctuation: bool) -> str:
text = str(word or "").strip()
if not keep_punctuation:
text = _PUNCT_RE.sub("", text)
text = re.sub(r"\s+", " ", text).strip()
return text.upper() if uppercase else text
def _plain_subtitle_blocks(words: Sequence[dict], max_words: int) -> list[list[dict]]:
blocks: list[list[dict]] = []
pending: list[dict] = []
for word in words:
if not str(word.get("word", "")).strip():
continue
pending.append(word)
if len(pending) >= max(1, max_words):
blocks.append(pending)
pending = []
if pending:
blocks.append(pending)
return blocks
async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]:
"""Validate title/subtitle layout for spatial collisions and safe-area
containment (collision.validate_titles over every <title> in the file)."""
@@ -210,15 +278,7 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
window_end = clip_source_start + clip_duration
clip_words = [
{
"word": w.get("word", ""),
"start": float(w.get("start", 0.0)) - clip_source_start,
"end": float(w.get("end", 0.0)) - clip_source_start,
}
for w in data.get("words", [])
if clip_source_start <= float(w.get("start", 0.0)) < window_end
]
clip_words = _words_overlapping_clip(data.get("words", []), clip_source_start, window_end)
if not clip_words:
skipped.append((name, "no words in clip's source range"))
continue
@@ -277,7 +337,113 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
return _text_result(result)
async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextContent]:
"""Generate static, editable title subtitles from word-level transcripts."""
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
clip_filter = arguments.get("clip_name")
saved = load_plain_subtitle_config()
font = arguments.get("font") or saved["font"]
font_size = int(arguments.get("font_size", saved["font_size"]))
font_color = arguments.get("font_color") or saved["font_color"]
max_words = max(1, int(arguments.get("max_words", saved["max_words"])))
position_y = float(arguments.get("position_y", saved["position_y"]))
uppercase = bool(arguments.get("uppercase", saved["uppercase"]))
keep_punctuation = bool(arguments.get("keep_punctuation", saved["keep_punctuation"]))
filepath, output_path, modifier = _setup_modifier(arguments, "_plain_subtitles")
added: list[tuple[str, int, int]] = []
skipped: list[tuple[str, str]] = []
spine_clips = [el for _, el in modifier._iter_spine_clips()]
for el in spine_clips:
name = el.get("name", "")
if clip_filter and name != clip_filter:
continue
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
media_path = media_src_to_path(src)
if not media_path or not Path(media_path).is_file():
skipped.append((name, "media file missing"))
continue
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
if data is None:
skipped.append((name, reason))
continue
clip_source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
clip_words = _words_overlapping_clip(
data.get("words", []), clip_source_start, clip_source_start + clip_duration
)
if not clip_words:
skipped.append((name, "no words in clip's source range"))
continue
blocks = _plain_subtitle_blocks(clip_words, max_words)
created = 0
for block in blocks:
parts = [
_plain_word_text(w.get("word", ""), uppercase=uppercase, keep_punctuation=keep_punctuation)
for w in block
]
text = " ".join(p for p in parts if p).strip()
if not text:
continue
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
end = max(float(w.get("end", start)) for w in block)
duration = max(end - start, modifier.frame_duration_fraction())
modifier.add_text_title(
el,
text,
offset=f"{start:.6f}s",
duration=f"{duration:.6f}s",
lane=20,
position=f"0 {position_y:g}",
font=font,
font_size=font_size,
font_color=font_color,
bold=True,
face=None,
font_scale=1.0,
size_param=font_size,
)
created += 1
if created:
added.append((name, created, len(clip_words)))
if not added:
text = "# Plain Subtitles\n\nNo subtitles generated — file unchanged (nothing saved)."
if skipped:
text += "\n\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
return _text_result(text)
modifier.save(output_path)
total_titles = sum(lines for _, lines, _ in added)
total_words = sum(words for _, _, words in added)
result = "# Plain Subtitles Generated\n\n## Summary\n"
result += (
f"- **Clips Captioned**: {len(added)}\n"
f"- **Title Clips**: {total_titles}\n"
f"- **Total Words**: {total_words}\n\n"
)
result += _markdown_table(
["Clip", "Title Clips", "Words"],
[[n, str(lines), str(words)] for n, lines, words in added],
)
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
return _text_result(result)
HANDLERS = {
"validate_subtitle_layout": handle_validate_subtitle_layout,
"generate_dynamic_subtitles": handle_generate_dynamic_subtitles,
"generate_plain_subtitles": handle_generate_plain_subtitles,
}
+2 -2
View File
@@ -67,12 +67,12 @@ TOOLS = [
),
Tool(
name="remove_filler_words",
description="Cut filler words (um, uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.",
description="Cut filler interjections (uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'um', 'uma', 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: um, uh, uhh, umm, erm, ehm, mmm, hmm, mhm)"},
"fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: uh, uhh, umm, erm, ehm, mmm, hmm, mhm; pass um/uma explicitly if desired)"},
"clip_name": {"type": "string", "description": "Only clean the clip with this name"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"padding": {"type": "number", "default": 0.02, "description": "Seconds to widen each cut on both sides (0-2, default 0.02)"},
+5 -2
View File
@@ -348,8 +348,9 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
language = arguments.get("language")
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
output_dir = arguments.get("output_dir")
transcript, reason = _load_or_transcribe(media_path, model, language)
transcript, reason = _load_or_transcribe(media_path, model, language, output_dir)
if transcript is None:
return _text_result(
f"# Voice Timeline\n\nCould not obtain a transcript "
@@ -365,9 +366,10 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
peak_percentile=config["peak_percentile"],
emphasis_floor=config["emphasis_floor"],
emotion_enabled=config["emotion_enabled"],
emotion_sensitivity=config["emotion_sensitivity"],
)
output_dir = arguments.get("output_dir")
json_path = Path(_validate_output_path(
str(voice_timeline_path(media_path, output_dir)),
anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent),
@@ -398,6 +400,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
"yes" if layers["acoustics"] else "FAILED — every acoustic value is 0",
],
["Speakers", "yes" if layers["speakers"] else "not run — single default speaker"],
["Emotion", "yes" if layers.get("emotion") else "not run"],
],
) + "\n"