chore: atualização geral
This commit is contained in:
@@ -0,0 +1,283 @@
|
||||
"""Legendas dinâmicas (geração → validação) — tool schemas and handlers.
|
||||
|
||||
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Sequence
|
||||
|
||||
from mcp.types import TextContent, Tool
|
||||
|
||||
from fcpxml.media_intel import media_src_to_path
|
||||
from fcpxml.model_manager import load_dynamic_subtitle_config
|
||||
from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
from server_tools._shared import (
|
||||
_load_or_transcribe,
|
||||
_markdown_table,
|
||||
_setup_modifier,
|
||||
_text_result,
|
||||
_validate_filepath,
|
||||
)
|
||||
|
||||
TOOLS = [
|
||||
Tool(
|
||||
name="validate_subtitle_layout",
|
||||
description="Re-measure every title/subtitle in an FCPXML and report spatial collisions, frame and safe-area violations, and font fallbacks. Detects overlapping boxes only for titles on screen at the same time (half-open time intervals, so a title ending exactly as the next begins is never flagged). Returns a severity (none/render_tolerance/warning/probable/severe), the list of issues with suggested corrections, and summary counts.",
|
||||
inputSchema={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
||||
"safe_margin_x": {"type": "number", "default": 0.05, "description": "Fraction of frame width to inset from each side (0.05 = 5%)"},
|
||||
"safe_margin_y": {"type": "number", "default": 0.05, "description": "Fraction of frame height to inset from top/bottom"},
|
||||
"min_font_size": {"type": "number", "description": "Flag titles whose emitted fontSize is below this readable minimum"},
|
||||
"min_distance": {"type": "number", "description": "Flag same-block titles closer than this many pixels (insufficient_spacing)"},
|
||||
"max_distance": {"type": "number", "description": "Flag same-block titles farther than this many pixels (excessive_spacing)"},
|
||||
"output_format": {"type": "string", "enum": ["markdown", "json"], "default": "markdown", "description": "Report format"}
|
||||
},
|
||||
"required": ["filepath"]
|
||||
}
|
||||
),
|
||||
Tool(
|
||||
name="generate_dynamic_subtitles",
|
||||
description="Generate progressive-composition subtitles as real, editable FCPXML title clips (the 'Text'/Basic Text template). Whisper's segments become sentences; each sentence is diagrammed as stacked blocks — supporting words grouped small in a grotesque, the sentence's key word alone and large in a display italic, body lines staggered to opposite edges. One <title> per block: each enters as its own words are spoken and stays on screen, so the sentence assembles itself, and every block clears at the same instant. Set granularity='word' for the older one-title-per-word rhythm. A sentence too tall for the band splits into successive compositions. These are TITLES, not captions: no subtitles role, so they render over the video without enabling caption display. Uses each media file's local Whisper word-level transcript (_transcript.json, auto-transcribes if missing). Style fields below (band_height through inactive_color) fall back to the style saved from the app's 'Legendas Dinâmicas' screen (~/.fcp-mcp-server/config.json via save_dynamic_subtitle_config) when omitted — pass a value here only to override that for one call. Non-destructive: writes a _dynamic_subtitles copy.",
|
||||
inputSchema={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
||||
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
|
||||
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
||||
"language": {"type": "string", "description": "ISO language code hint (e.g. 'en'); auto-detected if omitted"},
|
||||
"band_height": {"type": "number", "description": "Fraction of frame height the sentence block may fill before splitting into another block. Falls back to the saved style (default 0.22 — about three lines)"},
|
||||
"block_center_y": {"type": "number", "description": "Vertical centre of the block in canvas points; negative sits below frame centre. Falls back to the saved style (default -167, just under centre)"},
|
||||
"line_gap": {"type": "number", "description": "Air between stacked lines in canvas points. Lines are stacked on their real ink, so this is the whole distance beyond the glyphs themselves; negative values deliberately tuck each line into the one above. Falls back to the saved style (default 8)"},
|
||||
"granularity": {"type": "string", "enum": ["phrase", "word"], "default": "phrase", "description": "'phrase': one title per LINE of the composition, key word set large (the reference look). 'word': one title per word."},
|
||||
"emphasis_font": {"type": "string", "description": "Family for the key word (phrase mode). Must be installed on the editing Mac; unmeasured families fall back to estimated widths. Falls back to the saved style (default 'Playfair Display')"},
|
||||
"emphasis_face": {"type": "string", "description": "Face for the key word, e.g. 'Medium Italic' or a script/calligraphic face. Falls back to the saved style (default 'Medium Italic')"},
|
||||
"emphasis_size": {"type": "integer", "description": "Key-word size in canvas points, at the 2160x3840 reference frame. Falls back to the saved style (default 265)"},
|
||||
"emphasis_color": {"type": "string", "description": "RGBA (0-1, space-separated) for the key word (phrase mode). Defaults to active_color, so the block reads in a single colour unless the key word is deliberately set apart"},
|
||||
"text_scale": {"type": "number", "description": "Ratio between the title template's fontSize space and the canvas-point space it positions in. The \"Text\" template sizes type in frame pixels, so sizes are doubled on the way out. Falls back to the saved style (default 2.0). Lower it only if a template renders type larger than the chosen point size"},
|
||||
"font": {"type": "string", "description": "Title font family (supporting lines in phrase mode). Falls back to the saved style (default 'Helvetica Neue')"},
|
||||
"font_size": {"type": "integer", "description": "Supporting-line font size in canvas points, at the 2160x3840 reference frame. Falls back to the saved style (default 104)"},
|
||||
"active_color": {"type": "string", "description": "RGBA (0-1, space-separated) for even-indexed lines. Falls back to the saved style (default '1 1 1 1')"},
|
||||
"inactive_color": {"type": "string", "default": "0.7 0.7 0.7 1", "description": "RGBA (0-1, space-separated) for odd-indexed lines — alternates with active_color for visual variety between stacked lines"},
|
||||
"output_path": {"type": "string", "description": "Output path (default: adds _dynamic_subtitles suffix)"},
|
||||
},
|
||||
"required": ["filepath"]
|
||||
}
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Validate title/subtitle layout for spatial collisions and safe-area
|
||||
containment (collision.validate_titles over every <title> in the file)."""
|
||||
filepath = _validate_filepath(arguments["filepath"], (".fcpxml", ".fcpxmld"))
|
||||
modifier = FCPXMLModifier(filepath)
|
||||
report = modifier.validate_subtitle_layout(
|
||||
safe_margin_x=float(arguments.get("safe_margin_x", 0.05)),
|
||||
safe_margin_y=float(arguments.get("safe_margin_y", 0.05)),
|
||||
min_font_size=(
|
||||
float(arguments["min_font_size"])
|
||||
if arguments.get("min_font_size") is not None else None
|
||||
),
|
||||
min_distance=(
|
||||
float(arguments["min_distance"])
|
||||
if arguments.get("min_distance") is not None else None
|
||||
),
|
||||
max_distance=(
|
||||
float(arguments["max_distance"])
|
||||
if arguments.get("max_distance") is not None else None
|
||||
),
|
||||
)
|
||||
|
||||
if arguments.get("output_format") == "json":
|
||||
return _text_result(json.dumps(report, indent=2))
|
||||
|
||||
summary = report["summary"]
|
||||
lines = [
|
||||
"# Subtitle Layout Validation",
|
||||
"",
|
||||
f"## Summary (severity: {report['severity']})",
|
||||
f"- **Titles**: {summary['title_count']}",
|
||||
f"- **Issues**: {summary['issue_count']}",
|
||||
f"- **Collisions**: {summary['spatial_collision']}",
|
||||
f"- **Outside frame**: {summary['outside_frame']}",
|
||||
f"- **Outside safe area**: {summary['outside_safe_area']}",
|
||||
f"- **Font fallback**: {summary['font_missing']}",
|
||||
f"- **Font too small**: {summary['font_too_small']}",
|
||||
"",
|
||||
]
|
||||
issues = report["issues"]
|
||||
if issues:
|
||||
lines.append(f"## Issues ({len(issues)})")
|
||||
for issue in issues:
|
||||
sev = issue["severity"].upper()
|
||||
if issue["type"] == "spatial_collision":
|
||||
corr = issue["suggested_correction"]
|
||||
lines.append(
|
||||
f"- [{sev}] collision: \"{issue['first_title']}\" x "
|
||||
f"\"{issue['second_title']}\" "
|
||||
f"(overlap {issue['overlap_width']:.0f}x"
|
||||
f"{issue['overlap_height']:.0f} = "
|
||||
f"{issue['overlap_area']:.0f}px, ratio "
|
||||
f"{issue['overlap_ratio']:.2f}, move "
|
||||
f"{corr['axis']} {corr['minimum_movement']:.0f}px)"
|
||||
)
|
||||
else:
|
||||
detail = issue.get("title", "") or issue.get("font", "")
|
||||
lines.append(f"- [{sev}] {issue['type']}: {detail}".rstrip())
|
||||
else:
|
||||
lines.append("_No issues found — no simultaneous titles overlap._")
|
||||
|
||||
return _text_result("\n".join(lines))
|
||||
|
||||
|
||||
async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Generate per-word subtitle titles laid out as a block per sentence.
|
||||
|
||||
Whisper's segments become sentences; each word becomes its own positioned
|
||||
<title> connected clip, appearing as it is spoken and accumulating on
|
||||
screen until the whole block clears at once. No compound clip.
|
||||
|
||||
Reuses the same SOURCE-media -> TIMELINE mapping as ``transcript_markers``
|
||||
(``modifier.source_file_start`` per spine clip) so word timestamps land
|
||||
at the correct position even across trimmed/multiple clips.
|
||||
"""
|
||||
model = arguments.get("model", "base")
|
||||
language = arguments.get("language")
|
||||
output_dir = arguments.get("output_dir")
|
||||
clip_filter = arguments.get("clip_name")
|
||||
|
||||
# Anything the caller didn't explicitly pass falls back to the style
|
||||
# persisted from the "Legendas Dinâmicas" screen (~/.fcp-mcp-server/
|
||||
# config.json), not a hardcoded default — so the UI is the single place
|
||||
# that configures the look, and every caller (app, MCP, this session)
|
||||
# renders the same thing without threading 11 fields through every call.
|
||||
saved = load_dynamic_subtitle_config()
|
||||
body_color = arguments.get("active_color") or saved["active_color"]
|
||||
config = DynamicSubtitleConfig(
|
||||
style=WordStyle(
|
||||
font=arguments.get("font") or saved["font"],
|
||||
font_size=int(arguments.get("font_size", saved["font_size"])),
|
||||
active_color=body_color,
|
||||
inactive_color=arguments.get("inactive_color", "0.7 0.7 0.7 1"),
|
||||
emphasis_look=WordLook(
|
||||
int(arguments.get("emphasis_size", saved["emphasis_size"])),
|
||||
arguments.get("emphasis_color") or saved["emphasis_color"] or body_color,
|
||||
font=arguments.get("emphasis_font") or saved["emphasis_font"],
|
||||
face=arguments.get("emphasis_face") or saved["emphasis_face"],
|
||||
kerning=0.0,
|
||||
),
|
||||
body_look=WordLook(
|
||||
int(arguments.get("font_size", saved["font_size"])),
|
||||
body_color,
|
||||
font=arguments.get("font") or saved["font"],
|
||||
face="Bold",
|
||||
kerning=1.2,
|
||||
),
|
||||
),
|
||||
band_height=float(arguments.get("band_height", saved["band_height"])),
|
||||
block_center_y=float(arguments.get("block_center_y", saved["block_center_y"])),
|
||||
granularity=arguments.get("granularity", "phrase"),
|
||||
text_scale=float(arguments.get("text_scale", saved["text_scale"])),
|
||||
line_gap=float(arguments.get("line_gap", saved["line_gap"])),
|
||||
)
|
||||
|
||||
filepath, output_path, modifier = _setup_modifier(arguments, "_dynamic_subtitles")
|
||||
|
||||
added: list[tuple[str, int, int]] = []
|
||||
skipped: list[tuple[str, str]] = []
|
||||
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
||||
for el in spine_clips:
|
||||
name = el.get("name", "")
|
||||
if clip_filter and name != clip_filter:
|
||||
continue
|
||||
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
||||
media_path = media_src_to_path(src)
|
||||
if not media_path or not Path(media_path).is_file():
|
||||
skipped.append((name, "media file missing"))
|
||||
continue
|
||||
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
||||
if data is None:
|
||||
skipped.append((name, reason))
|
||||
continue
|
||||
|
||||
clip_source_start = modifier.source_file_start(el).to_seconds()
|
||||
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
||||
window_end = clip_source_start + clip_duration
|
||||
|
||||
clip_words = [
|
||||
{
|
||||
"word": w.get("word", ""),
|
||||
"start": float(w.get("start", 0.0)) - clip_source_start,
|
||||
"end": float(w.get("end", 0.0)) - clip_source_start,
|
||||
}
|
||||
for w in data.get("words", [])
|
||||
if clip_source_start <= float(w.get("start", 0.0)) < window_end
|
||||
]
|
||||
if not clip_words:
|
||||
skipped.append((name, "no words in clip's source range"))
|
||||
continue
|
||||
|
||||
# Sentence boundaries, rebased the same way, so each sentence becomes
|
||||
# its own block of titles that builds up and then clears together.
|
||||
# Overlap rather than containment: a segment straddling the clip's
|
||||
# in-point still governs the words that made the cut.
|
||||
clip_segments = [
|
||||
{
|
||||
"start": float(s.get("start", 0.0)) - clip_source_start,
|
||||
"end": float(s.get("end", 0.0)) - clip_source_start,
|
||||
}
|
||||
for s in data.get("segments", [])
|
||||
if float(s.get("end", 0.0)) > clip_source_start
|
||||
and float(s.get("start", 0.0)) < window_end
|
||||
]
|
||||
|
||||
# Pass the element itself, not `name` — after ripple-cut/silence
|
||||
# removal every fragment of an originally-named clip keeps the same
|
||||
# `name`, so a name lookup here would resolve every clip in this
|
||||
# loop to whichever one `self.clips` last indexed, stacking every
|
||||
# clip's captions onto a single wrong spine element instead of each
|
||||
# clip's own. See Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-17.
|
||||
lines = modifier.generate_dynamic_subtitles(
|
||||
el, clip_words, config, segments=clip_segments
|
||||
)
|
||||
added.append((name, len(lines), len(clip_words)))
|
||||
|
||||
if not added:
|
||||
text = "# Dynamic Subtitles\n\nNo captions generated — file unchanged (nothing saved)."
|
||||
if skipped:
|
||||
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
||||
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
||||
)
|
||||
return _text_result(text)
|
||||
|
||||
modifier.save(output_path)
|
||||
total_lines = sum(lines for _, lines, _ in added)
|
||||
total_words = sum(words for _, _, words in added)
|
||||
result = "# Dynamic Subtitles Generated (local Whisper)\n\n## Summary\n"
|
||||
result += (
|
||||
f"- **Clips Captioned**: {len(added)}\n"
|
||||
f"- **Caption Lines (Title Clips)**: {total_lines}\n"
|
||||
f"- **Total Words**: {total_words}\n\n"
|
||||
)
|
||||
result += _markdown_table(
|
||||
["Clip", "Caption Lines", "Words"],
|
||||
[[n, str(lines), str(words)] for n, lines, words in added],
|
||||
)
|
||||
if skipped:
|
||||
result += "\n## Skipped Clips\n" + _markdown_table(
|
||||
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
||||
)
|
||||
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
HANDLERS = {
|
||||
"validate_subtitle_layout": handle_validate_subtitle_layout,
|
||||
"generate_dynamic_subtitles": handle_generate_dynamic_subtitles,
|
||||
}
|
||||
Reference in New Issue
Block a user