feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e7748c2c58
commit
1bebee4359
@@ -15,6 +15,7 @@ from typing import Any, Sequence
|
||||
from mcp.types import TextContent
|
||||
|
||||
from fcpxml.media_intel import media_src_to_path
|
||||
from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config
|
||||
from fcpxml.models import (
|
||||
DuplicateGroup,
|
||||
FlashFrame,
|
||||
@@ -25,6 +26,7 @@ from fcpxml.models import (
|
||||
)
|
||||
from fcpxml.parser import FCPXMLParser
|
||||
from fcpxml.rough_cut import RoughCutGenerator
|
||||
from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text
|
||||
from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
@@ -639,28 +641,79 @@ def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str:
|
||||
rel_end = action.end - clip_start
|
||||
|
||||
if action.kind == "zoom":
|
||||
config = load_voice_analysis_config()
|
||||
# Only forward an explicit ease — otherwise add_zoom's own default
|
||||
# (a fast ramp in, instant snap back out) is what should apply.
|
||||
zoom_args = {}
|
||||
if action.params.get("ease") is not None:
|
||||
zoom_args["ease"] = float(action.params["ease"])
|
||||
if action.params.get("ease_out") is not None:
|
||||
zoom_args["ease_out"] = float(action.params["ease_out"])
|
||||
zoom_args = {
|
||||
"ease": float(action.params.get("ease", config["zoom_ease_in"])),
|
||||
"ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])),
|
||||
}
|
||||
mode = str(action.params.get("mode", config["zoom_mode"]))
|
||||
if mode == "in":
|
||||
zoom_args["hold_at_end"] = True
|
||||
zoom_args["start_at_peak"] = False
|
||||
elif mode == "out":
|
||||
zoom_args["hold_at_end"] = False
|
||||
zoom_args["start_at_peak"] = True
|
||||
elif mode == "in_out":
|
||||
zoom_args["hold_at_end"] = False
|
||||
zoom_args["start_at_peak"] = False
|
||||
modifier.add_zoom(
|
||||
clip_id=clip_el,
|
||||
start=rel_start,
|
||||
end=rel_end,
|
||||
scale=float(action.params.get("scale", 1.3)),
|
||||
scale=float(action.params.get("scale", config["zoom_scale"])),
|
||||
**zoom_args,
|
||||
)
|
||||
return f"zoom {action.params.get('scale', 1.3):.2f}x"
|
||||
return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x"
|
||||
|
||||
if action.kind == "text":
|
||||
# Default to the "Legendas Dinâmicas" emphasis style (the font used
|
||||
# to highlight a word in the captions) rather than a hardcoded
|
||||
# Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of
|
||||
# the video's on-screen text instead of looking like a stray default
|
||||
# title. Any of these the action itself specifies still wins.
|
||||
subtitle_cfg = load_dynamic_subtitle_config()
|
||||
font = action.params.get("font", subtitle_cfg["emphasis_font"])
|
||||
face = action.params.get("face", subtitle_cfg["emphasis_face"])
|
||||
font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0)
|
||||
requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"]))
|
||||
requested_kerning = float(action.params.get("kerning", 0.0) or 0.0)
|
||||
|
||||
# Voice-action callouts are not part of the dynamic subtitle block.
|
||||
# When omitted, put them above the subtitle band and shrink wide
|
||||
# phrases to the title-safe width. The previous default (Position 0 0,
|
||||
# full emphasis size) made long callouts like "PRÓTESES DE SILICONE"
|
||||
# collide with captions and run off both sides of a vertical frame.
|
||||
emitted_size = requested_size * font_scale
|
||||
emitted_kerning = requested_kerning * font_scale
|
||||
safe_width = modifier.frame_width() * 0.90
|
||||
width = measure_text(
|
||||
action.params["content"],
|
||||
emitted_size,
|
||||
bold=bool(action.params.get("bold", False)),
|
||||
kerning=emitted_kerning,
|
||||
font=font,
|
||||
face=face,
|
||||
)
|
||||
font_size = requested_size
|
||||
if width > safe_width and width > 0:
|
||||
font_size = max(32, int(requested_size * safe_width / width))
|
||||
position = action.params.get("position")
|
||||
if not position:
|
||||
position = f"0 {modifier.frame_height() * 0.23:g}"
|
||||
|
||||
modifier.add_text_title(
|
||||
clip_el,
|
||||
action.params["content"],
|
||||
offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
|
||||
duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(),
|
||||
position=position,
|
||||
font=font,
|
||||
font_size=font_size,
|
||||
font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]),
|
||||
face=face,
|
||||
bold=action.params.get("bold", False),
|
||||
)
|
||||
return f"text \"{action.params['content'][:24]}\""
|
||||
|
||||
|
||||
+176
-10
@@ -6,13 +6,14 @@ Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalo
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Sequence
|
||||
|
||||
from mcp.types import TextContent, Tool
|
||||
|
||||
from fcpxml.media_intel import media_src_to_path
|
||||
from fcpxml.model_manager import load_dynamic_subtitle_config
|
||||
from fcpxml.model_manager import load_dynamic_subtitle_config, load_plain_subtitle_config
|
||||
from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
from server_tools._shared import (
|
||||
@@ -69,9 +70,76 @@ TOOLS = [
|
||||
"required": ["filepath"]
|
||||
}
|
||||
),
|
||||
Tool(
|
||||
name="generate_plain_subtitles",
|
||||
description="Generate simple editable FCPXML text-title subtitles, synchronized to transcript words but without visual build-in/build-out effects. Words are grouped into short blocks, placed at a configurable vertical position, and written as static Text titles rather than SRT captions.",
|
||||
inputSchema={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
||||
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
|
||||
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
||||
"language": {"type": "string", "description": "ISO language code hint (e.g. 'pt'); auto-detected if omitted"},
|
||||
"font": {"type": "string", "description": "Text font family. Falls back to saved plain-subtitle config."},
|
||||
"font_size": {"type": "integer", "description": "Font size in canvas points. Falls back to saved plain-subtitle config."},
|
||||
"font_color": {"type": "string", "description": "RGBA (0-1, space-separated). Falls back to saved plain-subtitle config."},
|
||||
"max_words": {"type": "integer", "description": "Maximum words per subtitle block. Falls back to saved plain-subtitle config."},
|
||||
"position_y": {"type": "number", "description": "Vertical title position in canvas points; negative sits lower in frame."},
|
||||
"uppercase": {"type": "boolean", "description": "Render text in uppercase."},
|
||||
"keep_punctuation": {"type": "boolean", "description": "Keep punctuation such as comma and period."},
|
||||
"text_scale": {"type": "number", "description": "Template font-size scale. Falls back to saved plain-subtitle config."},
|
||||
"output_path": {"type": "string", "description": "Output path (default: adds _plain_subtitles suffix)"},
|
||||
},
|
||||
"required": ["filepath"]
|
||||
}
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
_PUNCT_RE = re.compile(r"[^\w\sÀ-ÖØ-öø-ÿ]", re.UNICODE)
|
||||
|
||||
|
||||
def _words_overlapping_clip(words: Sequence[dict], start: float, end: float) -> list[dict]:
|
||||
"""Return transcript words that overlap a source window, rebased to it."""
|
||||
clip_words: list[dict] = []
|
||||
for w in words:
|
||||
word_start = float(w.get("start", 0.0))
|
||||
word_end = float(w.get("end", word_start))
|
||||
if word_end <= start or word_start >= end:
|
||||
continue
|
||||
clip_words.append(
|
||||
{
|
||||
"word": w.get("word", ""),
|
||||
"start": max(0.0, word_start - start),
|
||||
"end": max(0.0, min(word_end, end) - start),
|
||||
}
|
||||
)
|
||||
return clip_words
|
||||
|
||||
|
||||
def _plain_word_text(word: str, *, uppercase: bool, keep_punctuation: bool) -> str:
|
||||
text = str(word or "").strip()
|
||||
if not keep_punctuation:
|
||||
text = _PUNCT_RE.sub("", text)
|
||||
text = re.sub(r"\s+", " ", text).strip()
|
||||
return text.upper() if uppercase else text
|
||||
|
||||
|
||||
def _plain_subtitle_blocks(words: Sequence[dict], max_words: int) -> list[list[dict]]:
|
||||
blocks: list[list[dict]] = []
|
||||
pending: list[dict] = []
|
||||
for word in words:
|
||||
if not str(word.get("word", "")).strip():
|
||||
continue
|
||||
pending.append(word)
|
||||
if len(pending) >= max(1, max_words):
|
||||
blocks.append(pending)
|
||||
pending = []
|
||||
if pending:
|
||||
blocks.append(pending)
|
||||
return blocks
|
||||
|
||||
|
||||
async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Validate title/subtitle layout for spatial collisions and safe-area
|
||||
containment (collision.validate_titles over every <title> in the file)."""
|
||||
@@ -210,15 +278,7 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
|
||||
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
||||
window_end = clip_source_start + clip_duration
|
||||
|
||||
clip_words = [
|
||||
{
|
||||
"word": w.get("word", ""),
|
||||
"start": float(w.get("start", 0.0)) - clip_source_start,
|
||||
"end": float(w.get("end", 0.0)) - clip_source_start,
|
||||
}
|
||||
for w in data.get("words", [])
|
||||
if clip_source_start <= float(w.get("start", 0.0)) < window_end
|
||||
]
|
||||
clip_words = _words_overlapping_clip(data.get("words", []), clip_source_start, window_end)
|
||||
if not clip_words:
|
||||
skipped.append((name, "no words in clip's source range"))
|
||||
continue
|
||||
@@ -277,7 +337,113 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Generate static, editable title subtitles from word-level transcripts."""
|
||||
model = arguments.get("model", "base")
|
||||
language = arguments.get("language")
|
||||
output_dir = arguments.get("output_dir")
|
||||
clip_filter = arguments.get("clip_name")
|
||||
|
||||
saved = load_plain_subtitle_config()
|
||||
font = arguments.get("font") or saved["font"]
|
||||
font_size = int(arguments.get("font_size", saved["font_size"]))
|
||||
font_color = arguments.get("font_color") or saved["font_color"]
|
||||
max_words = max(1, int(arguments.get("max_words", saved["max_words"])))
|
||||
position_y = float(arguments.get("position_y", saved["position_y"]))
|
||||
uppercase = bool(arguments.get("uppercase", saved["uppercase"]))
|
||||
keep_punctuation = bool(arguments.get("keep_punctuation", saved["keep_punctuation"]))
|
||||
|
||||
filepath, output_path, modifier = _setup_modifier(arguments, "_plain_subtitles")
|
||||
|
||||
added: list[tuple[str, int, int]] = []
|
||||
skipped: list[tuple[str, str]] = []
|
||||
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
||||
for el in spine_clips:
|
||||
name = el.get("name", "")
|
||||
if clip_filter and name != clip_filter:
|
||||
continue
|
||||
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
||||
media_path = media_src_to_path(src)
|
||||
if not media_path or not Path(media_path).is_file():
|
||||
skipped.append((name, "media file missing"))
|
||||
continue
|
||||
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
||||
if data is None:
|
||||
skipped.append((name, reason))
|
||||
continue
|
||||
|
||||
clip_source_start = modifier.source_file_start(el).to_seconds()
|
||||
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
||||
clip_words = _words_overlapping_clip(
|
||||
data.get("words", []), clip_source_start, clip_source_start + clip_duration
|
||||
)
|
||||
if not clip_words:
|
||||
skipped.append((name, "no words in clip's source range"))
|
||||
continue
|
||||
|
||||
blocks = _plain_subtitle_blocks(clip_words, max_words)
|
||||
created = 0
|
||||
for block in blocks:
|
||||
parts = [
|
||||
_plain_word_text(w.get("word", ""), uppercase=uppercase, keep_punctuation=keep_punctuation)
|
||||
for w in block
|
||||
]
|
||||
text = " ".join(p for p in parts if p).strip()
|
||||
if not text:
|
||||
continue
|
||||
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
|
||||
end = max(float(w.get("end", start)) for w in block)
|
||||
duration = max(end - start, modifier.frame_duration_fraction())
|
||||
modifier.add_text_title(
|
||||
el,
|
||||
text,
|
||||
offset=f"{start:.6f}s",
|
||||
duration=f"{duration:.6f}s",
|
||||
lane=20,
|
||||
position=f"0 {position_y:g}",
|
||||
font=font,
|
||||
font_size=font_size,
|
||||
font_color=font_color,
|
||||
bold=True,
|
||||
face=None,
|
||||
font_scale=1.0,
|
||||
size_param=font_size,
|
||||
)
|
||||
created += 1
|
||||
if created:
|
||||
added.append((name, created, len(clip_words)))
|
||||
|
||||
if not added:
|
||||
text = "# Plain Subtitles\n\nNo subtitles generated — file unchanged (nothing saved)."
|
||||
if skipped:
|
||||
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
||||
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
||||
)
|
||||
return _text_result(text)
|
||||
|
||||
modifier.save(output_path)
|
||||
total_titles = sum(lines for _, lines, _ in added)
|
||||
total_words = sum(words for _, _, words in added)
|
||||
result = "# Plain Subtitles Generated\n\n## Summary\n"
|
||||
result += (
|
||||
f"- **Clips Captioned**: {len(added)}\n"
|
||||
f"- **Title Clips**: {total_titles}\n"
|
||||
f"- **Total Words**: {total_words}\n\n"
|
||||
)
|
||||
result += _markdown_table(
|
||||
["Clip", "Title Clips", "Words"],
|
||||
[[n, str(lines), str(words)] for n, lines, words in added],
|
||||
)
|
||||
if skipped:
|
||||
result += "\n## Skipped Clips\n" + _markdown_table(
|
||||
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
||||
)
|
||||
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
HANDLERS = {
|
||||
"validate_subtitle_layout": handle_validate_subtitle_layout,
|
||||
"generate_dynamic_subtitles": handle_generate_dynamic_subtitles,
|
||||
"generate_plain_subtitles": handle_generate_plain_subtitles,
|
||||
}
|
||||
|
||||
@@ -67,12 +67,12 @@ TOOLS = [
|
||||
),
|
||||
Tool(
|
||||
name="remove_filler_words",
|
||||
description="Cut filler words (um, uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.",
|
||||
description="Cut filler interjections (uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'um', 'uma', 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.",
|
||||
inputSchema={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
||||
"fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: um, uh, uhh, umm, erm, ehm, mmm, hmm, mhm)"},
|
||||
"fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: uh, uhh, umm, erm, ehm, mmm, hmm, mhm; pass um/uma explicitly if desired)"},
|
||||
"clip_name": {"type": "string", "description": "Only clean the clip with this name"},
|
||||
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
||||
"padding": {"type": "number", "default": 0.02, "description": "Seconds to widen each cut on both sides (0-2, default 0.02)"},
|
||||
|
||||
@@ -348,8 +348,9 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
|
||||
language = arguments.get("language")
|
||||
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
|
||||
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
|
||||
output_dir = arguments.get("output_dir")
|
||||
|
||||
transcript, reason = _load_or_transcribe(media_path, model, language)
|
||||
transcript, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
||||
if transcript is None:
|
||||
return _text_result(
|
||||
f"# Voice Timeline\n\nCould not obtain a transcript "
|
||||
@@ -365,9 +366,10 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
|
||||
weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
|
||||
peak_percentile=config["peak_percentile"],
|
||||
emphasis_floor=config["emphasis_floor"],
|
||||
emotion_enabled=config["emotion_enabled"],
|
||||
emotion_sensitivity=config["emotion_sensitivity"],
|
||||
)
|
||||
|
||||
output_dir = arguments.get("output_dir")
|
||||
json_path = Path(_validate_output_path(
|
||||
str(voice_timeline_path(media_path, output_dir)),
|
||||
anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent),
|
||||
@@ -398,6 +400,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
|
||||
"yes" if layers["acoustics"] else "FAILED — every acoustic value is 0",
|
||||
],
|
||||
["Speakers", "yes" if layers["speakers"] else "not run — single default speaker"],
|
||||
["Emotion", "yes" if layers.get("emotion") else "not run"],
|
||||
],
|
||||
) + "\n"
|
||||
|
||||
|
||||
Reference in New Issue
Block a user