generate_dynamic_subtitles e a metade dinâmica de generate_subtitles_by_emphasis passam a empacotar cada sub-frase da legenda dinâmica num compound clip por padrão (compound_subphrases=True), completando o wrap_titles_in_compound e split_into_subphrases do commit anterior — que ainda não tinham chamador em produção. Também torna validate_subtitle_layout ciente de compound clips: media cada grupo (spine principal + cada <media> de compound) no seu próprio espaço de tempo, em vez de uma varredura .//title global — sem isso, âncoras de compounds diferentes liam offset "0s" e acusavam colisão espacial entre frases que nunca dividem a tela, só porque compartilham o mesmo zero de tempo local. Testado ponta a ponta na gravação real (Mastopexia): 12 compounds, 41 títulos todos empacotados, zero soltos, zero IDs duplicados, DTD válida. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
841 lines
44 KiB
Python
841 lines
44 KiB
Python
"""Legendas dinâmicas (geração → validação) — tool schemas and handlers.
|
|
|
|
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
from pathlib import Path
|
|
from typing import Sequence
|
|
|
|
from mcp.types import TextContent, Tool
|
|
|
|
from fcpxml.media_intel import media_src_to_path
|
|
from fcpxml.model_manager import (
|
|
get_active_dynamic_subtitle_layouts,
|
|
load_plain_subtitle_config,
|
|
)
|
|
from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle
|
|
from fcpxml.writer import FCPXMLModifier
|
|
from server_tools._shared import (
|
|
_load_or_transcribe,
|
|
_markdown_table,
|
|
_setup_modifier,
|
|
_text_result,
|
|
_validate_filepath,
|
|
)
|
|
|
|
TOOLS = [
|
|
Tool(
|
|
name="validate_subtitle_layout",
|
|
description="Re-measure every title/subtitle in an FCPXML and report spatial collisions, frame and safe-area violations, and font fallbacks. Detects overlapping boxes only for titles on screen at the same time (half-open time intervals, so a title ending exactly as the next begins is never flagged). Returns a severity (none/render_tolerance/warning/probable/severe), the list of issues with suggested corrections, and summary counts.",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"safe_margin_x": {"type": "number", "default": 0.05, "description": "Fraction of frame width to inset from each side (0.05 = 5%)"},
|
|
"safe_margin_y": {"type": "number", "default": 0.05, "description": "Fraction of frame height to inset from top/bottom"},
|
|
"min_font_size": {"type": "number", "description": "Flag titles whose emitted fontSize is below this readable minimum"},
|
|
"min_distance": {"type": "number", "description": "Flag same-block titles closer than this many pixels (insufficient_spacing)"},
|
|
"max_distance": {"type": "number", "description": "Flag same-block titles farther than this many pixels (excessive_spacing)"},
|
|
"output_format": {"type": "string", "enum": ["markdown", "json"], "default": "markdown", "description": "Report format"}
|
|
},
|
|
"required": ["filepath"]
|
|
}
|
|
),
|
|
Tool(
|
|
name="generate_dynamic_subtitles",
|
|
description="Generate progressive-composition subtitles as real, editable FCPXML title clips (the 'Text'/Basic Text template). Whisper's segments become sentences; each sentence is diagrammed as stacked blocks — supporting words grouped small in a grotesque, the sentence's key word alone and large in a display italic, body lines staggered to opposite edges. One <title> per block: each enters as its own words are spoken and stays on screen, so the sentence assembles itself, and every block clears at the same instant. Set granularity='word' for the older one-title-per-word rhythm. A sentence too tall for the band splits into successive compositions. These are TITLES, not captions: no subtitles role, so they render over the video without enabling caption display. Uses each media file's local Whisper word-level transcript (_transcript.json, auto-transcribes if missing). Style fields below (band_height through inactive_color) fall back to the style saved from the app's 'Legendas Dinâmicas' screen (~/.fcp-mcp-server/config.json via save_dynamic_subtitle_config) when omitted — pass a value here only to override that for one call. Non-destructive: writes a _dynamic_subtitles copy.",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
|
|
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
|
"language": {"type": "string", "description": "ISO language code hint (e.g. 'en'); auto-detected if omitted"},
|
|
"band_height": {"type": "number", "description": "Fraction of frame height the sentence block may fill before splitting into another block. Falls back to the saved style (default 0.22 — about three lines)"},
|
|
"block_center_y": {"type": "number", "description": "Vertical centre of the block in canvas points; negative sits below frame centre. Falls back to the saved style (default -167, just under centre)"},
|
|
"line_gap": {"type": "number", "description": "Air between stacked lines in canvas points. Lines are stacked on their real ink, so this is the whole distance beyond the glyphs themselves; negative values deliberately tuck each line into the one above. Falls back to the saved style (default 8)"},
|
|
"granularity": {"type": "string", "enum": ["phrase", "word"], "default": "phrase", "description": "'phrase': one title per LINE of the composition, key word set large (the reference look). 'word': one title per word."},
|
|
"emphasis_font": {"type": "string", "description": "Family for the key word (phrase mode). Must be installed on the editing Mac; unmeasured families fall back to estimated widths. Falls back to the saved style (default 'Playfair Display')"},
|
|
"emphasis_face": {"type": "string", "description": "Face for the key word, e.g. 'Medium Italic' or a script/calligraphic face. Falls back to the saved style (default 'Medium Italic')"},
|
|
"emphasis_size": {"type": "integer", "description": "Key-word size in canvas points, at the 2160x3840 reference frame. Falls back to the saved style (default 265)"},
|
|
"emphasis_color": {"type": "string", "description": "RGBA (0-1, space-separated) for the key word (phrase mode). Defaults to active_color, so the block reads in a single colour unless the key word is deliberately set apart"},
|
|
"text_scale": {"type": "number", "description": "Ratio between the title template's fontSize space and the canvas-point space it positions in. The \"Text\" template sizes type in frame pixels, so sizes are doubled on the way out. Falls back to the saved style (default 2.0). Lower it only if a template renders type larger than the chosen point size"},
|
|
"role": {"type": "string", "description": "Final Cut role for every generated title (a 'titles.*' sub-role, never 'subtitles.*'). Falls back to the saved style (default 'titles.dinamicas'). Groups the clips in the role index and tints their lane."},
|
|
"font": {"type": "string", "description": "Title font family (supporting lines in phrase mode). Falls back to the saved style (default 'Helvetica Neue')"},
|
|
"font_size": {"type": "integer", "description": "Supporting-line font size in canvas points, at the 2160x3840 reference frame. Falls back to the saved style (default 104)"},
|
|
"active_color": {"type": "string", "description": "RGBA (0-1, space-separated) for even-indexed lines. Falls back to the saved style (default '1 1 1 1')"},
|
|
"inactive_color": {"type": "string", "default": "0.7 0.7 0.7 1", "description": "RGBA (0-1, space-separated) for odd-indexed lines — alternates with active_color for visual variety between stacked lines"},
|
|
"output_path": {"type": "string", "description": "Output path (default: adds _dynamic_subtitles suffix)"},
|
|
},
|
|
"required": ["filepath"]
|
|
}
|
|
),
|
|
Tool(
|
|
name="generate_plain_subtitles",
|
|
description="Generate simple editable FCPXML text-title subtitles, synchronized to transcript words but without visual build-in/build-out effects. Words are grouped into short blocks, placed at a configurable vertical position, and written as static Text titles rather than SRT captions.",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
|
|
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
|
"language": {"type": "string", "description": "ISO language code hint (e.g. 'pt'); auto-detected if omitted"},
|
|
"font": {"type": "string", "description": "Text font family. Falls back to saved plain-subtitle config."},
|
|
"font_size": {"type": "integer", "description": "Font size in canvas points. Falls back to saved plain-subtitle config."},
|
|
"font_color": {"type": "string", "description": "RGBA (0-1, space-separated). Falls back to saved plain-subtitle config."},
|
|
"max_words": {"type": "integer", "description": "Maximum words per subtitle block. Falls back to saved plain-subtitle config."},
|
|
"position_y": {"type": "number", "description": "Vertical title position in canvas points; negative sits lower in frame."},
|
|
"uppercase": {"type": "boolean", "description": "Render text in uppercase."},
|
|
"keep_punctuation": {"type": "boolean", "description": "Keep punctuation such as comma and period."},
|
|
"text_scale": {"type": "number", "description": "Template font-size scale. Falls back to saved plain-subtitle config."},
|
|
"role": {"type": "string", "description": "Final Cut role for every generated title (a 'titles.*' sub-role, never 'subtitles.*'). Falls back to the saved style (default 'titles.convencionais'). Groups the clips in the role index and tints their lane."},
|
|
"output_path": {"type": "string", "description": "Output path (default: adds _plain_subtitles suffix)"},
|
|
},
|
|
"required": ["filepath"]
|
|
}
|
|
),
|
|
Tool(
|
|
name="generate_subtitles_by_emphasis",
|
|
description="Generate BOTH subtitle styles in one pass, split by word so they never coexist on the same range: dynamic progressive-composition titles (see generate_dynamic_subtitles) cover whichever whole phrases were marked as emphasis in the phrase-review step (etapa 5, zoom applied, level >= 1); plain static titles (see generate_plain_subtitles) cover every OTHER word in the clip. A plain block is simply not created where a dynamic phrase already covers — not created-then-disabled — because a disabled title still shows as its own struck-through clip in Final Cut's timeline even though it never renders, and a heavily emphasized edit ended up with dozens of dead clips cluttering the track. Trade-off: if emphasis is turned off by hand later, the plain line under it has to be regenerated, not just re-enabled. Reads emphasis spans from the media's cached '<media>_phrase_actions.json' (written by save_phrase_review after the app's etapa 5 review) — run the voice-editing wizard through that step first, or nothing is treated as emphasis and every word gets a plain title. Style knobs are the saved 'Legendas Dinâmicas'/plain-subtitle configs (~/.fcp-mcp-server/config.json); this tool does not expose per-call style overrides, only the split logic — use generate_dynamic_subtitles/generate_plain_subtitles directly if you need one-off styling.",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
|
|
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
|
"language": {"type": "string", "description": "ISO language code hint (e.g. 'pt'); auto-detected if omitted"},
|
|
"granularity": {"type": "string", "enum": ["phrase", "word"], "default": "phrase", "description": "Passed through to the dynamic half, same meaning as in generate_dynamic_subtitles"},
|
|
"max_words": {"type": "integer", "description": "Max words per block for the plain half. Falls back to saved plain-subtitle config."},
|
|
"uppercase": {"type": "boolean", "description": "Uppercase the plain half. Falls back to saved plain-subtitle config."},
|
|
"keep_punctuation": {"type": "boolean", "description": "Keep punctuation in the plain half. Falls back to saved plain-subtitle config."},
|
|
"output_path": {"type": "string", "description": "Output path (default: adds _emphasis_subtitles suffix)"},
|
|
},
|
|
"required": ["filepath"]
|
|
}
|
|
),
|
|
]
|
|
|
|
|
|
_PUNCT_RE = re.compile(r"[^\w\sÀ-ÖØ-öø-ÿ]", re.UNICODE)
|
|
_DEFAULT_DYNAMIC_ROLE = "titles.dinamicas"
|
|
_DEFAULT_PLAIN_ROLE = "titles.convencionais"
|
|
|
|
|
|
def _title_subrole(value: str | None, fallback: str) -> str:
|
|
"""Return a Final Cut title sub-role, never a closed-caption role."""
|
|
role = str(value or "").strip() or fallback
|
|
if role.startswith("subtitles."):
|
|
return "titles." + role.removeprefix("subtitles.")
|
|
if role == "subtitles":
|
|
return fallback
|
|
if not role.startswith("titles."):
|
|
return fallback
|
|
return role
|
|
|
|
|
|
def _separate_subtitle_roles(dynamic_role: str | None, plain_role: str | None) -> tuple[str, str]:
|
|
"""Keep normal and dynamic subtitles in distinct Final Cut role lanes."""
|
|
dynamic = _title_subrole(dynamic_role, _DEFAULT_DYNAMIC_ROLE)
|
|
plain = _title_subrole(plain_role, _DEFAULT_PLAIN_ROLE)
|
|
if dynamic == plain:
|
|
if dynamic != _DEFAULT_DYNAMIC_ROLE:
|
|
return dynamic, _DEFAULT_PLAIN_ROLE
|
|
return _DEFAULT_DYNAMIC_ROLE, _DEFAULT_PLAIN_ROLE
|
|
return dynamic, plain
|
|
|
|
|
|
def _words_overlapping_clip(words: Sequence[dict], start: float, end: float) -> list[dict]:
|
|
"""Return transcript words that overlap a source window, rebased to it."""
|
|
clip_words: list[dict] = []
|
|
for w in words:
|
|
word_start = float(w.get("start", 0.0))
|
|
word_end = float(w.get("end", word_start))
|
|
if word_end <= start or word_start >= end:
|
|
continue
|
|
clip_words.append(
|
|
{
|
|
"word": w.get("word", ""),
|
|
"start": max(0.0, word_start - start),
|
|
"end": max(0.0, min(word_end, end) - start),
|
|
}
|
|
)
|
|
return clip_words
|
|
|
|
|
|
def _plain_word_text(word: str, *, uppercase: bool, keep_punctuation: bool) -> str:
|
|
text = str(word or "").strip()
|
|
if not keep_punctuation:
|
|
text = _PUNCT_RE.sub("", text)
|
|
text = re.sub(r"\s+", " ", text).strip()
|
|
return text.upper() if uppercase else text
|
|
|
|
|
|
def _plain_subtitle_blocks(words: Sequence[dict], max_words: int) -> list[list[dict]]:
|
|
blocks: list[list[dict]] = []
|
|
pending: list[dict] = []
|
|
for word in words:
|
|
if not str(word.get("word", "")).strip():
|
|
continue
|
|
pending.append(word)
|
|
if len(pending) >= max(1, max_words):
|
|
blocks.append(pending)
|
|
pending = []
|
|
if pending:
|
|
blocks.append(pending)
|
|
return blocks
|
|
|
|
|
|
def _phrase_actions_path(media_path: str) -> Path:
|
|
"""Where `save_phrase_review` writes emphasis decisions for this media.
|
|
|
|
Mirrors `phrase_review.review_paths()`'s naming (stem + "_phrase_actions.json"),
|
|
without importing that module just for a path — the voice_timeline this would
|
|
normally derive from is itself named `<media stem>_voice_timeline.json`, so
|
|
stripping straight from the media stem lands on the same file.
|
|
"""
|
|
stem = Path(media_path).stem
|
|
return Path(media_path).with_name(f"{stem}_phrase_actions.json")
|
|
|
|
|
|
def _load_review_spans(media_path: str, key: str) -> list[dict]:
|
|
"""Load one span list (source-media time) saved by the etapa-5 phrase review.
|
|
|
|
``key`` is ``"emphasis_spans"`` (phrases with `subtitle_dynamic` on) or
|
|
``"plain_exclude_spans"`` (phrases with `subtitle_common` off). Returns []
|
|
if the review was never run for this media, or saved nothing under that
|
|
key — callers should treat that as "nothing marked", not as an error,
|
|
since the wizard's later steps are optional.
|
|
"""
|
|
path = _phrase_actions_path(media_path)
|
|
if not path.is_file():
|
|
return []
|
|
try:
|
|
data = json.loads(path.read_text(encoding="utf-8"))
|
|
except (OSError, json.JSONDecodeError):
|
|
return []
|
|
spans = data.get(key, [])
|
|
return [s for s in spans if isinstance(s, dict) and "start" in s and "end" in s]
|
|
|
|
|
|
def _load_emphasis_spans(media_path: str) -> list[dict]:
|
|
"""Spans (source-media time) whose phrase has `subtitle_dynamic` on."""
|
|
return _load_review_spans(media_path, "emphasis_spans")
|
|
|
|
|
|
def _load_plain_exclude_spans(media_path: str) -> list[dict]:
|
|
"""Spans (source-media time) whose phrase has `subtitle_common` off.
|
|
|
|
Independent from emphasis spans: a phrase can have `subtitle_common` off
|
|
without being emphasized, so plain must be hidden there too even though
|
|
no dynamic line is going to cover the gap.
|
|
"""
|
|
return _load_review_spans(media_path, "plain_exclude_spans")
|
|
|
|
|
|
def _word_in_spans(word_start: float, word_end: float, spans: Sequence[dict]) -> bool:
|
|
"""A word belongs to an emphasis span if its midpoint falls inside it.
|
|
|
|
Midpoint, not start, so a word straddling a span boundary (which can happen
|
|
since spans come from phrase trims, not word timestamps) lands on whichever
|
|
side it mostly belongs to instead of always defaulting to one edge.
|
|
"""
|
|
mid = (word_start + word_end) / 2.0
|
|
return any(float(s["start"]) <= mid < float(s["end"]) for s in spans)
|
|
|
|
|
|
def _words_in_spans(words: Sequence[dict], spans: Sequence[dict]) -> list[dict]:
|
|
"""The subset of source-time transcript words that fall inside a span.
|
|
|
|
Feeds only the DYNAMIC half — the plain half always gets every word, full
|
|
clip, unfiltered; this is not a partition of the word list into two
|
|
disjoint sets, it is "which words also get the dynamic treatment on top".
|
|
"""
|
|
if not spans:
|
|
return []
|
|
return [
|
|
w for w in words
|
|
if _word_in_spans(float(w.get("start", 0.0)), float(w.get("end", w.get("start", 0.0))), spans)
|
|
]
|
|
|
|
|
|
def _segments_in_spans(segments: Sequence[dict], spans: Sequence[dict]) -> list[dict]:
|
|
"""Keep only the sentences that fall inside an emphasis span (by midpoint).
|
|
|
|
Feeds the dynamic half's sentence-block builder; segments outside every span
|
|
would only produce blocks with no words left in them after the word filter.
|
|
"""
|
|
if not spans:
|
|
return []
|
|
kept = []
|
|
for seg in segments:
|
|
start = float(seg.get("start", 0.0))
|
|
end = float(seg.get("end", start))
|
|
mid = (start + end) / 2.0
|
|
if any(float(s["start"]) <= mid < float(s["end"]) for s in spans):
|
|
kept.append(seg)
|
|
return kept
|
|
|
|
|
|
def _overlaps_any_span(start: float, end: float, spans: Sequence[tuple[float, float]]) -> bool:
|
|
"""Half-open interval overlap: a plain title under this window is skipped."""
|
|
return any(start < span_end and end > span_start for span_start, span_end in spans)
|
|
|
|
|
|
async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]:
|
|
"""Validate title/subtitle layout for spatial collisions and safe-area
|
|
containment (collision.validate_titles over every <title> in the file)."""
|
|
filepath = _validate_filepath(arguments["filepath"], (".fcpxml", ".fcpxmld"))
|
|
modifier = FCPXMLModifier(filepath)
|
|
report = modifier.validate_subtitle_layout(
|
|
safe_margin_x=float(arguments.get("safe_margin_x", 0.05)),
|
|
safe_margin_y=float(arguments.get("safe_margin_y", 0.05)),
|
|
min_font_size=(
|
|
float(arguments["min_font_size"])
|
|
if arguments.get("min_font_size") is not None else None
|
|
),
|
|
min_distance=(
|
|
float(arguments["min_distance"])
|
|
if arguments.get("min_distance") is not None else None
|
|
),
|
|
max_distance=(
|
|
float(arguments["max_distance"])
|
|
if arguments.get("max_distance") is not None else None
|
|
),
|
|
)
|
|
|
|
if arguments.get("output_format") == "json":
|
|
return _text_result(json.dumps(report, indent=2))
|
|
|
|
summary = report["summary"]
|
|
lines = [
|
|
"# Subtitle Layout Validation",
|
|
"",
|
|
f"## Summary (severity: {report['severity']})",
|
|
f"- **Titles**: {summary['title_count']}",
|
|
f"- **Issues**: {summary['issue_count']}",
|
|
f"- **Collisions**: {summary['spatial_collision']}",
|
|
f"- **Outside frame**: {summary['outside_frame']}",
|
|
f"- **Outside safe area**: {summary['outside_safe_area']}",
|
|
f"- **Font fallback**: {summary['font_missing']}",
|
|
f"- **Font too small**: {summary['font_too_small']}",
|
|
"",
|
|
]
|
|
issues = report["issues"]
|
|
if issues:
|
|
lines.append(f"## Issues ({len(issues)})")
|
|
for issue in issues:
|
|
sev = issue["severity"].upper()
|
|
if issue["type"] == "spatial_collision":
|
|
corr = issue["suggested_correction"]
|
|
lines.append(
|
|
f"- [{sev}] collision: \"{issue['first_title']}\" x "
|
|
f"\"{issue['second_title']}\" "
|
|
f"(overlap {issue['overlap_width']:.0f}x"
|
|
f"{issue['overlap_height']:.0f} = "
|
|
f"{issue['overlap_area']:.0f}px, ratio "
|
|
f"{issue['overlap_ratio']:.2f}, move "
|
|
f"{corr['axis']} {corr['minimum_movement']:.0f}px)"
|
|
)
|
|
else:
|
|
detail = issue.get("title", "") or issue.get("font", "")
|
|
lines.append(f"- [{sev}] {issue['type']}: {detail}".rstrip())
|
|
else:
|
|
lines.append("_No issues found — no simultaneous titles overlap._")
|
|
|
|
return _text_result("\n".join(lines))
|
|
|
|
|
|
def _build_dynamic_subtitle_config(saved: dict, overrides: dict | None = None) -> DynamicSubtitleConfig:
|
|
"""Build a :class:`DynamicSubtitleConfig` from one registered layout dict.
|
|
|
|
``overrides`` (typically the tool call's own ``arguments``) only makes
|
|
sense to apply when there is a single active layout — callers with 2+
|
|
active layouts pass ``{}`` so every sampled block uses its layout as
|
|
registered, unambiguously.
|
|
"""
|
|
overrides = overrides or {}
|
|
body_color = overrides.get("active_color") or saved["active_color"]
|
|
return DynamicSubtitleConfig(
|
|
style=WordStyle(
|
|
font=overrides.get("font") or saved["font"],
|
|
font_size=int(overrides.get("font_size", saved["font_size"])),
|
|
active_color=body_color,
|
|
inactive_color=overrides.get("inactive_color", "0.7 0.7 0.7 1"),
|
|
emphasis_look=WordLook(
|
|
int(overrides.get("emphasis_size", saved["emphasis_size"])),
|
|
overrides.get("emphasis_color") or saved["emphasis_color"] or body_color,
|
|
font=overrides.get("emphasis_font") or saved["emphasis_font"],
|
|
face=overrides.get("emphasis_face") or saved["emphasis_face"],
|
|
kerning=0.0,
|
|
),
|
|
body_look=WordLook(
|
|
int(overrides.get("font_size", saved["font_size"])),
|
|
body_color,
|
|
font=overrides.get("font") or saved["font"],
|
|
face="Bold",
|
|
kerning=1.2,
|
|
),
|
|
),
|
|
band_height=float(overrides.get("band_height", saved["band_height"])),
|
|
block_center_y=float(overrides.get("block_center_y", saved["block_center_y"])),
|
|
granularity=overrides.get("granularity", "phrase"),
|
|
text_scale=float(overrides.get("text_scale", saved["text_scale"])),
|
|
line_gap=float(overrides.get("line_gap", saved["line_gap"])),
|
|
role=_title_subrole(saved.get("role"), _DEFAULT_DYNAMIC_ROLE),
|
|
)
|
|
|
|
|
|
async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextContent]:
|
|
"""Generate per-word subtitle titles laid out as a block per sentence.
|
|
|
|
Whisper's segments become sentences; each word becomes its own positioned
|
|
<title> connected clip, appearing as it is spoken and accumulating on
|
|
screen until the whole block clears at once. No compound clip.
|
|
|
|
Reuses the same SOURCE-media -> TIMELINE mapping as ``transcript_markers``
|
|
(``modifier.source_file_start`` per spine clip) so word timestamps land
|
|
at the correct position even across trimmed/multiple clips.
|
|
"""
|
|
model = arguments.get("model", "base")
|
|
language = arguments.get("language")
|
|
output_dir = arguments.get("output_dir")
|
|
clip_filter = arguments.get("clip_name")
|
|
|
|
# Anything the caller didn't explicitly pass falls back to the style(s)
|
|
# persisted from the "Legendas Dinâmicas" screen (~/.fcp-mcp-server/
|
|
# config.json) — one or more named, active layouts. With exactly one
|
|
# active layout, per-call overrides (arguments) still apply, same as
|
|
# before this screen supported multiple layouts. With 2+ active layouts,
|
|
# each block below randomly samples one of them, so per-call overrides
|
|
# are ambiguous (which layout would they apply to?) and are ignored —
|
|
# register/edit the layouts themselves instead.
|
|
active_layouts = get_active_dynamic_subtitle_layouts()
|
|
overrides = arguments if len(active_layouts) == 1 else {}
|
|
configs = [_build_dynamic_subtitle_config(saved, overrides) for saved in active_layouts]
|
|
single_role_override = (
|
|
_title_subrole(arguments.get("role"), configs[0].role)
|
|
if len(active_layouts) == 1 and arguments.get("role")
|
|
else None
|
|
)
|
|
|
|
filepath, output_path, modifier = _setup_modifier(arguments, "_dynamic_subtitles")
|
|
|
|
added: list[tuple[str, int, int]] = []
|
|
skipped: list[tuple[str, str]] = []
|
|
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
|
for el in spine_clips:
|
|
name = el.get("name", "")
|
|
if clip_filter and name != clip_filter:
|
|
continue
|
|
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
|
media_path = media_src_to_path(src)
|
|
if not media_path or not Path(media_path).is_file():
|
|
skipped.append((name, "media file missing"))
|
|
continue
|
|
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
|
if data is None:
|
|
skipped.append((name, reason))
|
|
continue
|
|
|
|
clip_source_start = modifier.source_file_start(el).to_seconds()
|
|
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
|
window_end = clip_source_start + clip_duration
|
|
|
|
clip_words = _words_overlapping_clip(data.get("words", []), clip_source_start, window_end)
|
|
if not clip_words:
|
|
skipped.append((name, "no words in clip's source range"))
|
|
continue
|
|
|
|
# Sentence boundaries, rebased the same way, so each sentence becomes
|
|
# its own block of titles that builds up and then clears together.
|
|
# Overlap rather than containment: a segment straddling the clip's
|
|
# in-point still governs the words that made the cut.
|
|
clip_segments = [
|
|
{
|
|
"start": float(s.get("start", 0.0)) - clip_source_start,
|
|
"end": float(s.get("end", 0.0)) - clip_source_start,
|
|
}
|
|
for s in data.get("segments", [])
|
|
if float(s.get("end", 0.0)) > clip_source_start
|
|
and float(s.get("start", 0.0)) < window_end
|
|
]
|
|
|
|
# Pass the element itself, not `name` — after ripple-cut/silence
|
|
# removal every fragment of an originally-named clip keeps the same
|
|
# `name`, so a name lookup here would resolve every clip in this
|
|
# loop to whichever one `self.clips` last indexed, stacking every
|
|
# clip's captions onto a single wrong spine element instead of each
|
|
# clip's own. See Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-17.
|
|
lines = modifier.generate_dynamic_subtitles(
|
|
el, clip_words, configs=configs, segments=clip_segments,
|
|
role=single_role_override,
|
|
compound_subphrases=True,
|
|
)
|
|
added.append((name, len(lines), len(clip_words)))
|
|
|
|
if not added:
|
|
text = "# Dynamic Subtitles\n\nNo captions generated — file unchanged (nothing saved)."
|
|
if skipped:
|
|
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
|
)
|
|
return _text_result(text)
|
|
|
|
modifier.save(output_path)
|
|
total_lines = sum(lines for _, lines, _ in added)
|
|
total_words = sum(words for _, _, words in added)
|
|
result = "# Dynamic Subtitles Generated (local Whisper)\n\n## Summary\n"
|
|
result += (
|
|
f"- **Clips Captioned**: {len(added)}\n"
|
|
f"- **Caption Lines (Title Clips)**: {total_lines}\n"
|
|
f"- **Total Words**: {total_words}\n\n"
|
|
)
|
|
result += _markdown_table(
|
|
["Clip", "Caption Lines", "Words"],
|
|
[[n, str(lines), str(words)] for n, lines, words in added],
|
|
)
|
|
if skipped:
|
|
result += "\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
|
)
|
|
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
|
|
return _text_result(result)
|
|
|
|
|
|
async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextContent]:
|
|
"""Generate static, editable title subtitles from word-level transcripts."""
|
|
model = arguments.get("model", "base")
|
|
language = arguments.get("language")
|
|
output_dir = arguments.get("output_dir")
|
|
clip_filter = arguments.get("clip_name")
|
|
|
|
saved = load_plain_subtitle_config()
|
|
saved["role"] = _title_subrole(
|
|
arguments.get("role") or saved.get("role"),
|
|
_DEFAULT_PLAIN_ROLE,
|
|
)
|
|
font = arguments.get("font") or saved["font"]
|
|
font_size = int(arguments.get("font_size", saved["font_size"]))
|
|
font_color = arguments.get("font_color") or saved["font_color"]
|
|
max_words = max(1, int(arguments.get("max_words", saved["max_words"])))
|
|
position_y = float(arguments.get("position_y", saved["position_y"]))
|
|
uppercase = bool(arguments.get("uppercase", saved["uppercase"]))
|
|
keep_punctuation = bool(arguments.get("keep_punctuation", saved["keep_punctuation"]))
|
|
|
|
filepath, output_path, modifier = _setup_modifier(arguments, "_plain_subtitles")
|
|
|
|
added: list[tuple[str, int, int]] = []
|
|
skipped: list[tuple[str, str]] = []
|
|
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
|
for el in spine_clips:
|
|
name = el.get("name", "")
|
|
if clip_filter and name != clip_filter:
|
|
continue
|
|
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
|
media_path = media_src_to_path(src)
|
|
if not media_path or not Path(media_path).is_file():
|
|
skipped.append((name, "media file missing"))
|
|
continue
|
|
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
|
if data is None:
|
|
skipped.append((name, reason))
|
|
continue
|
|
|
|
clip_source_start = modifier.source_file_start(el).to_seconds()
|
|
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
|
clip_words = _words_overlapping_clip(
|
|
data.get("words", []), clip_source_start, clip_source_start + clip_duration
|
|
)
|
|
if not clip_words:
|
|
skipped.append((name, "no words in clip's source range"))
|
|
continue
|
|
|
|
blocks = _plain_subtitle_blocks(clip_words, max_words)
|
|
created = 0
|
|
for block in blocks:
|
|
parts = [
|
|
_plain_word_text(w.get("word", ""), uppercase=uppercase, keep_punctuation=keep_punctuation)
|
|
for w in block
|
|
]
|
|
text = " ".join(p for p in parts if p).strip()
|
|
if not text:
|
|
continue
|
|
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
|
|
end = max(float(w.get("end", start)) for w in block)
|
|
duration = max(end - start, modifier.frame_duration_fraction())
|
|
modifier.add_text_title(
|
|
el,
|
|
text,
|
|
offset=f"{start:.6f}s",
|
|
duration=f"{duration:.6f}s",
|
|
lane=20,
|
|
position=f"0 {position_y:g}",
|
|
font=font,
|
|
font_size=font_size,
|
|
font_color=font_color,
|
|
bold=True,
|
|
face=None,
|
|
font_scale=1.0,
|
|
size_param=font_size,
|
|
role=saved["role"],
|
|
)
|
|
created += 1
|
|
if created:
|
|
added.append((name, created, len(clip_words)))
|
|
|
|
if not added:
|
|
text = "# Plain Subtitles\n\nNo subtitles generated — file unchanged (nothing saved)."
|
|
if skipped:
|
|
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
|
)
|
|
return _text_result(text)
|
|
|
|
modifier.save(output_path)
|
|
total_titles = sum(lines for _, lines, _ in added)
|
|
total_words = sum(words for _, _, words in added)
|
|
result = "# Plain Subtitles Generated\n\n## Summary\n"
|
|
result += (
|
|
f"- **Clips Captioned**: {len(added)}\n"
|
|
f"- **Title Clips**: {total_titles}\n"
|
|
f"- **Total Words**: {total_words}\n\n"
|
|
)
|
|
result += _markdown_table(
|
|
["Clip", "Title Clips", "Words"],
|
|
[[n, str(lines), str(words)] for n, lines, words in added],
|
|
)
|
|
if skipped:
|
|
result += "\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
|
)
|
|
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
|
|
return _text_result(result)
|
|
|
|
|
|
async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[TextContent]:
|
|
"""Generate dynamic titles for the emphasis phrases, and plain titles for
|
|
every OTHER word — a plain block is simply not created where a dynamic
|
|
phrase already covers, rather than created and disabled.
|
|
|
|
A disabled ("enabled=0") title still shows as its own struck-through clip
|
|
in Final Cut's timeline even though it never renders — a heavily
|
|
emphasized edit ended up with dozens of dead clips cluttering the track.
|
|
Not generating them there trades that clutter for a smaller gap: if the
|
|
emphasis is turned off by hand later, the plain line has to be
|
|
regenerated rather than just re-enabled.
|
|
"""
|
|
model = arguments.get("model", "base")
|
|
language = arguments.get("language")
|
|
output_dir = arguments.get("output_dir")
|
|
clip_filter = arguments.get("clip_name")
|
|
granularity = arguments.get("granularity", "phrase")
|
|
|
|
# One or more named, active layouts — with 2+ active, each emphasis block
|
|
# below randomly samples one of them (see generate_dynamic_subtitles).
|
|
active_dynamic_layouts = get_active_dynamic_subtitle_layouts()
|
|
dynamic_configs = [
|
|
_build_dynamic_subtitle_config(saved, {"granularity": granularity})
|
|
for saved in active_dynamic_layouts
|
|
]
|
|
single_dynamic_role = (
|
|
active_dynamic_layouts[0]["role"] if len(active_dynamic_layouts) == 1 else None
|
|
)
|
|
|
|
saved_plain = load_plain_subtitle_config()
|
|
dynamic_role, plain_role = _separate_subtitle_roles(
|
|
single_dynamic_role or dynamic_configs[0].role,
|
|
saved_plain.get("role"),
|
|
)
|
|
if len(active_dynamic_layouts) == 1:
|
|
single_dynamic_role = dynamic_role
|
|
else:
|
|
for cfg in dynamic_configs:
|
|
cfg.role = _title_subrole(cfg.role, _DEFAULT_DYNAMIC_ROLE)
|
|
if cfg.role == plain_role:
|
|
cfg.role = _DEFAULT_DYNAMIC_ROLE
|
|
saved_plain["role"] = plain_role
|
|
plain_font = saved_plain["font"]
|
|
plain_font_size = int(saved_plain["font_size"])
|
|
plain_font_color = saved_plain["font_color"]
|
|
max_words = max(1, int(arguments.get("max_words", saved_plain["max_words"])))
|
|
position_y = float(saved_plain["position_y"])
|
|
uppercase = bool(arguments.get("uppercase", saved_plain["uppercase"]))
|
|
keep_punctuation = bool(arguments.get("keep_punctuation", saved_plain["keep_punctuation"]))
|
|
|
|
filepath, output_path, modifier = _setup_modifier(arguments, "_emphasis_subtitles")
|
|
|
|
added: list[tuple[str, int, int, int, int]] = []
|
|
skipped: list[tuple[str, str]] = []
|
|
no_review: list[str] = []
|
|
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
|
for el in spine_clips:
|
|
name = el.get("name", "")
|
|
if clip_filter and name != clip_filter:
|
|
continue
|
|
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
|
media_path = media_src_to_path(src)
|
|
if not media_path or not Path(media_path).is_file():
|
|
skipped.append((name, "media file missing"))
|
|
continue
|
|
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
|
if data is None:
|
|
skipped.append((name, reason))
|
|
continue
|
|
|
|
spans = _load_emphasis_spans(media_path)
|
|
exclude_spans = _load_plain_exclude_spans(media_path)
|
|
if not spans and not exclude_spans:
|
|
no_review.append(name)
|
|
|
|
clip_source_start = modifier.source_file_start(el).to_seconds()
|
|
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
|
window_end = clip_source_start + clip_duration
|
|
|
|
def _clip_relative(span_list: list[dict]) -> list[tuple[float, float]]:
|
|
return [
|
|
(max(0.0, float(s["start"]) - clip_source_start), min(clip_duration, float(s["end"]) - clip_source_start))
|
|
for s in span_list
|
|
if float(s["end"]) > clip_source_start and float(s["start"]) < window_end
|
|
]
|
|
|
|
# Clip-relative windows, for deciding which plain titles to hide —
|
|
# same coordinate space add_text_title's offsets end up in. Dynamic
|
|
# spans hide plain (see the trade-off note below); explicit
|
|
# `subtitle_common: false` spans hide it too, even without a dynamic
|
|
# line covering the gap.
|
|
clip_spans = _clip_relative(spans)
|
|
clip_hide_plain_spans = clip_spans + _clip_relative(exclude_spans)
|
|
|
|
all_words = data.get("words", [])
|
|
|
|
dynamic_lines = 0
|
|
dynamic_word_count = 0
|
|
emphasis_words = _words_in_spans(all_words, spans)
|
|
clip_emphasis_words = _words_overlapping_clip(emphasis_words, clip_source_start, window_end)
|
|
if clip_emphasis_words:
|
|
all_segments = data.get("segments", [])
|
|
emphasis_segments = _segments_in_spans(all_segments, spans)
|
|
clip_segments = [
|
|
{
|
|
"start": float(s.get("start", 0.0)) - clip_source_start,
|
|
"end": float(s.get("end", 0.0)) - clip_source_start,
|
|
}
|
|
for s in emphasis_segments
|
|
if float(s.get("end", 0.0)) > clip_source_start
|
|
and float(s.get("start", 0.0)) < window_end
|
|
]
|
|
# Pass the element itself, not `name` — see the same note in
|
|
# handle_generate_dynamic_subtitles (Engine/docs/05_EXPERIENCIAS.md,
|
|
# entry 2026-08-17).
|
|
dynamic_lines = len(
|
|
modifier.generate_dynamic_subtitles(
|
|
el, clip_emphasis_words, configs=dynamic_configs, segments=clip_segments,
|
|
role=single_dynamic_role,
|
|
compound_subphrases=True,
|
|
)
|
|
)
|
|
dynamic_word_count = len(clip_emphasis_words)
|
|
|
|
# Plain covers every word OUTSIDE an emphasis span. A block landing
|
|
# under a dynamic phrase is simply not created there — generating it
|
|
# disabled was tried first, but every disabled title still shows up
|
|
# as its own clip in Final Cut's timeline (just struck through), so
|
|
# a heavily-emphasized edit ended up with dozens of dead clips
|
|
# cluttering the track for no visible benefit. The trade-off: if the
|
|
# emphasis is later turned off by hand, the plain line under it has
|
|
# to be regenerated rather than just re-enabled.
|
|
plain_created = 0
|
|
plain_hidden = 0
|
|
clip_all_words = _words_overlapping_clip(all_words, clip_source_start, window_end)
|
|
blocks = _plain_subtitle_blocks(clip_all_words, max_words)
|
|
for block in blocks:
|
|
parts = [
|
|
_plain_word_text(w.get("word", ""), uppercase=uppercase, keep_punctuation=keep_punctuation)
|
|
for w in block
|
|
]
|
|
text = " ".join(p for p in parts if p).strip()
|
|
if not text:
|
|
continue
|
|
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
|
|
end = max(float(w.get("end", start)) for w in block)
|
|
if _overlaps_any_span(start, end, clip_hide_plain_spans):
|
|
plain_hidden += 1
|
|
continue
|
|
duration = max(end - start, modifier.frame_duration_fraction())
|
|
modifier.add_text_title(
|
|
el,
|
|
text,
|
|
offset=f"{start:.6f}s",
|
|
duration=f"{duration:.6f}s",
|
|
lane=20,
|
|
position=f"0 {position_y:g}",
|
|
font=plain_font,
|
|
font_size=plain_font_size,
|
|
font_color=plain_font_color,
|
|
bold=True,
|
|
face=None,
|
|
font_scale=1.0,
|
|
size_param=plain_font_size,
|
|
role=saved_plain["role"],
|
|
)
|
|
plain_created += 1
|
|
|
|
if dynamic_lines or plain_created:
|
|
added.append(
|
|
(name, dynamic_lines, plain_created, plain_hidden, dynamic_word_count + len(clip_all_words))
|
|
)
|
|
else:
|
|
skipped.append((name, "no words in clip's source range"))
|
|
|
|
if not added:
|
|
text = "# Subtitles by Emphasis\n\nNo captions generated — file unchanged (nothing saved)."
|
|
if skipped:
|
|
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
|
)
|
|
return _text_result(text)
|
|
|
|
modifier.save(output_path)
|
|
total_dynamic = sum(d for _, d, _, _, _ in added)
|
|
total_plain = sum(p for _, _, p, _, _ in added)
|
|
total_hidden = sum(h for _, _, _, h, _ in added)
|
|
total_words = sum(w for _, _, _, _, w in added)
|
|
result = "# Subtitles by Emphasis Generated\n\n## Summary\n"
|
|
result += (
|
|
f"- **Clips Captioned**: {len(added)}\n"
|
|
f"- **Dynamic Title Lines (emphasis)**: {total_dynamic}\n"
|
|
f"- **Plain Title Blocks**: {total_plain}\n"
|
|
f"- **Plain Blocks Skipped Under Emphasis (not created there)**: {total_hidden}\n"
|
|
f"- **Total Words**: {total_words}\n\n"
|
|
)
|
|
result += _markdown_table(
|
|
["Clip", "Dynamic Lines", "Plain Blocks", "Skipped", "Words"],
|
|
[[n, str(d), str(p), str(h), str(w)] for n, d, p, h, w in added],
|
|
)
|
|
if no_review:
|
|
result += (
|
|
"\n## Sem revisão de ênfase\n"
|
|
"Nenhum `_phrase_actions.json` encontrado para: "
|
|
+ ", ".join(no_review)
|
|
+ " — todas as frases desses clipes saíram como legenda comum. "
|
|
"Rode a etapa 5 do Assistente (revisão de frases) antes, se quiser destaque dinâmico.\n"
|
|
)
|
|
if skipped:
|
|
result += "\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
|
)
|
|
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json; emphasis spans from _phrase_actions.json.*"
|
|
return _text_result(result)
|
|
|
|
|
|
HANDLERS = {
|
|
"validate_subtitle_layout": handle_validate_subtitle_layout,
|
|
"generate_dynamic_subtitles": handle_generate_dynamic_subtitles,
|
|
"generate_plain_subtitles": handle_generate_plain_subtitles,
|
|
"generate_subtitles_by_emphasis": handle_generate_subtitles_by_emphasis,
|
|
}
|