feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e7748c2c58
commit
1bebee4359
+176
-10
@@ -6,13 +6,14 @@ Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalo
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Sequence
|
||||
|
||||
from mcp.types import TextContent, Tool
|
||||
|
||||
from fcpxml.media_intel import media_src_to_path
|
||||
from fcpxml.model_manager import load_dynamic_subtitle_config
|
||||
from fcpxml.model_manager import load_dynamic_subtitle_config, load_plain_subtitle_config
|
||||
from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
from server_tools._shared import (
|
||||
@@ -69,9 +70,76 @@ TOOLS = [
|
||||
"required": ["filepath"]
|
||||
}
|
||||
),
|
||||
Tool(
|
||||
name="generate_plain_subtitles",
|
||||
description="Generate simple editable FCPXML text-title subtitles, synchronized to transcript words but without visual build-in/build-out effects. Words are grouped into short blocks, placed at a configurable vertical position, and written as static Text titles rather than SRT captions.",
|
||||
inputSchema={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
||||
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
|
||||
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
||||
"language": {"type": "string", "description": "ISO language code hint (e.g. 'pt'); auto-detected if omitted"},
|
||||
"font": {"type": "string", "description": "Text font family. Falls back to saved plain-subtitle config."},
|
||||
"font_size": {"type": "integer", "description": "Font size in canvas points. Falls back to saved plain-subtitle config."},
|
||||
"font_color": {"type": "string", "description": "RGBA (0-1, space-separated). Falls back to saved plain-subtitle config."},
|
||||
"max_words": {"type": "integer", "description": "Maximum words per subtitle block. Falls back to saved plain-subtitle config."},
|
||||
"position_y": {"type": "number", "description": "Vertical title position in canvas points; negative sits lower in frame."},
|
||||
"uppercase": {"type": "boolean", "description": "Render text in uppercase."},
|
||||
"keep_punctuation": {"type": "boolean", "description": "Keep punctuation such as comma and period."},
|
||||
"text_scale": {"type": "number", "description": "Template font-size scale. Falls back to saved plain-subtitle config."},
|
||||
"output_path": {"type": "string", "description": "Output path (default: adds _plain_subtitles suffix)"},
|
||||
},
|
||||
"required": ["filepath"]
|
||||
}
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
_PUNCT_RE = re.compile(r"[^\w\sÀ-ÖØ-öø-ÿ]", re.UNICODE)
|
||||
|
||||
|
||||
def _words_overlapping_clip(words: Sequence[dict], start: float, end: float) -> list[dict]:
|
||||
"""Return transcript words that overlap a source window, rebased to it."""
|
||||
clip_words: list[dict] = []
|
||||
for w in words:
|
||||
word_start = float(w.get("start", 0.0))
|
||||
word_end = float(w.get("end", word_start))
|
||||
if word_end <= start or word_start >= end:
|
||||
continue
|
||||
clip_words.append(
|
||||
{
|
||||
"word": w.get("word", ""),
|
||||
"start": max(0.0, word_start - start),
|
||||
"end": max(0.0, min(word_end, end) - start),
|
||||
}
|
||||
)
|
||||
return clip_words
|
||||
|
||||
|
||||
def _plain_word_text(word: str, *, uppercase: bool, keep_punctuation: bool) -> str:
|
||||
text = str(word or "").strip()
|
||||
if not keep_punctuation:
|
||||
text = _PUNCT_RE.sub("", text)
|
||||
text = re.sub(r"\s+", " ", text).strip()
|
||||
return text.upper() if uppercase else text
|
||||
|
||||
|
||||
def _plain_subtitle_blocks(words: Sequence[dict], max_words: int) -> list[list[dict]]:
|
||||
blocks: list[list[dict]] = []
|
||||
pending: list[dict] = []
|
||||
for word in words:
|
||||
if not str(word.get("word", "")).strip():
|
||||
continue
|
||||
pending.append(word)
|
||||
if len(pending) >= max(1, max_words):
|
||||
blocks.append(pending)
|
||||
pending = []
|
||||
if pending:
|
||||
blocks.append(pending)
|
||||
return blocks
|
||||
|
||||
|
||||
async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Validate title/subtitle layout for spatial collisions and safe-area
|
||||
containment (collision.validate_titles over every <title> in the file)."""
|
||||
@@ -210,15 +278,7 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
|
||||
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
||||
window_end = clip_source_start + clip_duration
|
||||
|
||||
clip_words = [
|
||||
{
|
||||
"word": w.get("word", ""),
|
||||
"start": float(w.get("start", 0.0)) - clip_source_start,
|
||||
"end": float(w.get("end", 0.0)) - clip_source_start,
|
||||
}
|
||||
for w in data.get("words", [])
|
||||
if clip_source_start <= float(w.get("start", 0.0)) < window_end
|
||||
]
|
||||
clip_words = _words_overlapping_clip(data.get("words", []), clip_source_start, window_end)
|
||||
if not clip_words:
|
||||
skipped.append((name, "no words in clip's source range"))
|
||||
continue
|
||||
@@ -277,7 +337,113 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Generate static, editable title subtitles from word-level transcripts."""
|
||||
model = arguments.get("model", "base")
|
||||
language = arguments.get("language")
|
||||
output_dir = arguments.get("output_dir")
|
||||
clip_filter = arguments.get("clip_name")
|
||||
|
||||
saved = load_plain_subtitle_config()
|
||||
font = arguments.get("font") or saved["font"]
|
||||
font_size = int(arguments.get("font_size", saved["font_size"]))
|
||||
font_color = arguments.get("font_color") or saved["font_color"]
|
||||
max_words = max(1, int(arguments.get("max_words", saved["max_words"])))
|
||||
position_y = float(arguments.get("position_y", saved["position_y"]))
|
||||
uppercase = bool(arguments.get("uppercase", saved["uppercase"]))
|
||||
keep_punctuation = bool(arguments.get("keep_punctuation", saved["keep_punctuation"]))
|
||||
|
||||
filepath, output_path, modifier = _setup_modifier(arguments, "_plain_subtitles")
|
||||
|
||||
added: list[tuple[str, int, int]] = []
|
||||
skipped: list[tuple[str, str]] = []
|
||||
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
||||
for el in spine_clips:
|
||||
name = el.get("name", "")
|
||||
if clip_filter and name != clip_filter:
|
||||
continue
|
||||
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
||||
media_path = media_src_to_path(src)
|
||||
if not media_path or not Path(media_path).is_file():
|
||||
skipped.append((name, "media file missing"))
|
||||
continue
|
||||
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
||||
if data is None:
|
||||
skipped.append((name, reason))
|
||||
continue
|
||||
|
||||
clip_source_start = modifier.source_file_start(el).to_seconds()
|
||||
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
||||
clip_words = _words_overlapping_clip(
|
||||
data.get("words", []), clip_source_start, clip_source_start + clip_duration
|
||||
)
|
||||
if not clip_words:
|
||||
skipped.append((name, "no words in clip's source range"))
|
||||
continue
|
||||
|
||||
blocks = _plain_subtitle_blocks(clip_words, max_words)
|
||||
created = 0
|
||||
for block in blocks:
|
||||
parts = [
|
||||
_plain_word_text(w.get("word", ""), uppercase=uppercase, keep_punctuation=keep_punctuation)
|
||||
for w in block
|
||||
]
|
||||
text = " ".join(p for p in parts if p).strip()
|
||||
if not text:
|
||||
continue
|
||||
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
|
||||
end = max(float(w.get("end", start)) for w in block)
|
||||
duration = max(end - start, modifier.frame_duration_fraction())
|
||||
modifier.add_text_title(
|
||||
el,
|
||||
text,
|
||||
offset=f"{start:.6f}s",
|
||||
duration=f"{duration:.6f}s",
|
||||
lane=20,
|
||||
position=f"0 {position_y:g}",
|
||||
font=font,
|
||||
font_size=font_size,
|
||||
font_color=font_color,
|
||||
bold=True,
|
||||
face=None,
|
||||
font_scale=1.0,
|
||||
size_param=font_size,
|
||||
)
|
||||
created += 1
|
||||
if created:
|
||||
added.append((name, created, len(clip_words)))
|
||||
|
||||
if not added:
|
||||
text = "# Plain Subtitles\n\nNo subtitles generated — file unchanged (nothing saved)."
|
||||
if skipped:
|
||||
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
||||
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
||||
)
|
||||
return _text_result(text)
|
||||
|
||||
modifier.save(output_path)
|
||||
total_titles = sum(lines for _, lines, _ in added)
|
||||
total_words = sum(words for _, _, words in added)
|
||||
result = "# Plain Subtitles Generated\n\n## Summary\n"
|
||||
result += (
|
||||
f"- **Clips Captioned**: {len(added)}\n"
|
||||
f"- **Title Clips**: {total_titles}\n"
|
||||
f"- **Total Words**: {total_words}\n\n"
|
||||
)
|
||||
result += _markdown_table(
|
||||
["Clip", "Title Clips", "Words"],
|
||||
[[n, str(lines), str(words)] for n, lines, words in added],
|
||||
)
|
||||
if skipped:
|
||||
result += "\n## Skipped Clips\n" + _markdown_table(
|
||||
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
||||
)
|
||||
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
HANDLERS = {
|
||||
"validate_subtitle_layout": handle_validate_subtitle_layout,
|
||||
"generate_dynamic_subtitles": handle_generate_dynamic_subtitles,
|
||||
"generate_plain_subtitles": handle_generate_plain_subtitles,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user