chore: atualização geral

This commit is contained in:
João Henrique
2026-08-19 16:35:29 -04:00
parent 8fca456ceb
commit e7748c2c58
66 changed files with 13037 additions and 4237 deletions
+8
View File
@@ -0,0 +1,8 @@
"""Tool handlers and schemas for the FCPXML MCP server, split by category.
server.py is the composition root: it imports each module's TOOLS/HANDLERS
and concatenates them for list_tools()/TOOL_HANDLERS. Each module here owns
one category (see Engine/docs/03_SERVER_TOOLS.md) — its Tool() schemas and
handle_<name> functions live together, so a tool's contract and its
implementation are never in different files.
"""
+829
View File
@@ -0,0 +1,829 @@
"""Shared internal helpers used by tool handlers across categories.
Extracted from server.py — validation, formatting, and small parsing utilities
that more than one server_tools/*.py module needs.
"""
from __future__ import annotations
import json
import os
import re
from pathlib import Path
from typing import Any, Sequence
from mcp.types import TextContent
from fcpxml.media_intel import media_src_to_path
from fcpxml.models import (
DuplicateGroup,
FlashFrame,
FlashFrameSeverity,
GapInfo,
Timecode,
TimeValue,
)
from fcpxml.parser import FCPXMLParser
from fcpxml.rough_cut import RoughCutGenerator
from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe
from fcpxml.writer import FCPXMLModifier
PROJECTS_DIR = os.environ.get("FCP_PROJECTS_DIR", os.path.expanduser("~/Movies"))
_SANDBOX_ENABLED = "FCP_PROJECTS_DIR" in os.environ
MAX_FILE_SIZE = 100 * 1024 * 1024
MAX_MEDIA_FILE_SIZE = 32 * 1024 * 1024 * 1024
_MAX_JSON_DEPTH = 50
def _check_json_depth(obj: object, _depth: int = 0) -> None:
"""Reject JSON structures nested beyond _MAX_JSON_DEPTH.
Prevents denial-of-service via deeply nested objects that exhaust the
call stack or memory during downstream processing. Called after
json.load() since Python's json module has no built-in depth limit.
"""
if _depth > _MAX_JSON_DEPTH:
raise ValueError(
f"JSON nesting depth exceeds {_MAX_JSON_DEPTH} — "
"file may be malformed or adversarial"
)
if isinstance(obj, dict):
for v in obj.values():
_check_json_depth(v, _depth + 1)
elif isinstance(obj, list):
for item in obj:
_check_json_depth(item, _depth + 1)
def _validate_filepath(
filepath: str,
allowed_extensions: tuple[str, ...] | None = None,
max_size: int = MAX_FILE_SIZE,
) -> str:
"""Validate a user-provided file path against traversal and size attacks.
Resolves symlinks, blocks null bytes, enforces extension whitelist, and
checks file size before any parsing takes place.
``max_size`` defaults to the document limit; callers handling source
media pass ``MAX_MEDIA_FILE_SIZE``, since media is streamed rather than
parsed into memory (see the constant for why).
Raises:
ValueError: For invalid paths (null bytes, bad extensions, oversized).
FileNotFoundError: When the resolved path does not exist.
"""
if '\x00' in filepath:
raise ValueError("Invalid file path: null byte detected")
resolved = Path(filepath).resolve()
if not resolved.exists():
raise FileNotFoundError(f"File not found: {filepath}")
# .fcpxmld bundles are directories (a package wrapping Info.fcpxml plus
# sidecar data files for object tracking / Cinematic mode). The size
# check applies to the inner Info.fcpxml, which is what gets parsed.
if resolved.is_dir():
if resolved.suffix.lower() != '.fcpxmld':
raise ValueError(f"Not a regular file: {filepath}")
inner = resolved / 'Info.fcpxml'
if not inner.is_file():
raise ValueError(f"Invalid bundle (no Info.fcpxml): {filepath}")
size_target = inner
elif not resolved.is_file():
raise ValueError(f"Not a regular file: {filepath}")
else:
size_target = resolved
if allowed_extensions and resolved.suffix.lower() not in allowed_extensions:
raise ValueError(
f"Invalid file type '{resolved.suffix}'. "
f"Allowed: {', '.join(allowed_extensions)}"
)
if size_target.stat().st_size > max_size:
size_mb = size_target.stat().st_size / (1024 * 1024)
raise ValueError(f"File too large ({size_mb:.1f} MB). Maximum: {max_size // (1024 * 1024)} MB")
return str(resolved)
def _validate_output_path(output_path: str, *, anchor_dir: str | None = None) -> str:
"""Validate an output path with optional sandbox enforcement.
Resolves traversal, blocks null bytes, ensures parent exists, and — when
*anchor_dir* is provided — verifies the resolved output lives under that
directory. This prevents LLM-generated tool calls from writing to
arbitrary filesystem locations (e.g. ``/etc/cron.d/backdoor``).
Args:
output_path: The raw output path to validate.
anchor_dir: If set, the resolved output must be a child of this
directory. Typically the parent directory of the input file so
outputs stay co-located with their sources.
Raises:
ValueError: For null bytes, missing parent, or sandbox escape.
"""
if '\x00' in output_path:
raise ValueError("Invalid output path: null byte detected")
resolved = Path(output_path).resolve()
if not resolved.parent.exists():
raise ValueError(f"Output directory does not exist: {resolved.parent}")
if anchor_dir is not None:
anchor = Path(anchor_dir).resolve()
try:
resolved.relative_to(anchor)
except ValueError:
raise ValueError(
f"Output path escapes allowed directory: "
f"{resolved} is not under {anchor}"
)
return str(resolved)
def _validate_directory(directory: str, *, allowed_root: str | None = None) -> str:
"""Validate a user-provided directory path against traversal and injection.
Resolves symlinks, blocks null bytes, and verifies the path is a real
directory. When *allowed_root* is given, the resolved path must be a
descendant of (or equal to) that root — preventing filesystem enumeration
beyond the project workspace.
Raises:
ValueError: For invalid paths (null bytes, not a directory, sandbox escape).
"""
if '\x00' in directory:
raise ValueError("Invalid directory path: null byte detected")
resolved = Path(directory).resolve()
if not resolved.is_dir():
raise ValueError(f"Not a valid directory: {directory}")
if allowed_root is not None:
root = Path(allowed_root).resolve()
try:
resolved.relative_to(root)
except ValueError:
raise ValueError(
f"Directory escapes allowed root: "
f"{resolved} is not under {root}"
)
return str(resolved)
def find_fcpxml_files(directory: str) -> list[str]:
"""Find all FCPXML files in a directory."""
path = Path(directory)
files = list(str(f) for f in path.rglob("*.fcpxml"))
files.extend(str(f) for f in path.rglob("*.fcpxmld"))
return sorted(files)
def format_timecode(tc) -> str:
"""Format a Timecode object to SMPTE string."""
return tc.to_smpte() if tc else "00:00:00:00"
def format_duration(seconds: float) -> str:
"""Format seconds into human-readable duration."""
if seconds < 1:
return f"{seconds*1000:.0f}ms"
elif seconds < 60:
return f"{seconds:.2f}s"
return f"{int(seconds // 60)}m {seconds % 60:.1f}s"
def _format_clip_table(clips: list, header: str) -> str:
"""Render a list of clips as a markdown table with timecodes and durations.
Shared by handlers that filter clips by duration threshold
(find_short_cuts, find_long_clips).
"""
result = f"{header}\n\n| Name | TC | Duration |\n|------|----|---------|\n"
result += "\n".join(
f"| {c.name} | {format_timecode(c.start)} | {format_duration(c.duration_seconds)} |"
for c in clips
)
return result
def _markdown_table(headers: list[str], rows: list[list[str]]) -> str:
"""Build a markdown table from headers and rows.
Returns header row, separator row, and data rows as a single string.
Callers avoid repeating the ``| H1 | H2 |\\n|---|---|`` boilerplate
that appears in 15+ handlers.
"""
header_line = "| " + " | ".join(headers) + " |"
sep_line = "|" + "|".join("------" for _ in headers) + "|"
data_lines = "\n".join(
"| " + " | ".join(str(c) for c in row) + " |" for row in rows
)
return f"{header_line}\n{sep_line}\n{data_lines}"
def _format_batch_result(
title: str,
summary: dict[str, str],
headers: list[str],
rows: list[list[str]],
output_path: str,
) -> str:
"""Build a standard batch-operation result with summary, table, and save footer.
Used by batch fix handlers (flash frames, rapid trim, fill gaps) that all
share the same markdown structure: ``# Title → ## Summary → ## Details table
→ Saved to`` footer.
"""
summary_lines = "\n".join(f"- **{k}**: {v}" for k, v in summary.items())
table = _markdown_table(headers, rows)
return (
f"# {title}\n\n"
f"## Summary\n{summary_lines}\n\n"
f"## Details\n{table}\n\n"
f"Saved to: `{output_path}`"
)
def _fmt_suggestions(suggestions: list[str]) -> str:
"""Format pacing suggestions as markdown list (Python 3.10 compatible)."""
if not suggestions:
return "- Pacing looks good!"
nl = "\n"
return nl.join(f"- {s}" for s in suggestions)
def generate_output_path(input_path: str, suffix: str = "_modified") -> str:
"""Generate output path from input path.
The suffix is sanitized to prevent path-component injection — only
alphanumeric, hyphen, underscore, and dot characters survive.
"""
# Strip anything that could inject path separators or traversal sequences
clean_suffix = re.sub(r'[^a-zA-Z0-9._-]', '', suffix)
if not clean_suffix:
clean_suffix = "_modified"
p = Path(input_path)
return str(p.parent / f"{p.stem}{clean_suffix}{p.suffix}")
def _parse_project(filepath: str):
"""Parse an FCPXML file and return the project with its primary timeline."""
filepath = _validate_filepath(filepath, ('.fcpxml', '.fcpxmld'))
project = FCPXMLParser().parse_file(filepath)
if not project.timelines:
return None, None
return project, project.primary_timeline
def _text_result(text: str) -> list[TextContent]:
"""Wrap a string in the MCP TextContent list that every tool handler returns."""
return [TextContent(type="text", text=text)]
def _no_timeline():
"""Standard response when no timelines are found."""
return _text_result("No timelines found")
def _require_timeline(filepath: str):
"""Parse FCPXML and return (project, timeline), raising if no timeline exists.
Centralises the repeated _parse_project + _no_timeline guard that
appears in every read-only timeline handler. Returns a tuple so
callers can destructure directly::
project, tl = _require_timeline(arguments["filepath"])
"""
project, tl = _parse_project(filepath)
if not tl:
raise _NoTimelineError()
return project, tl
class _NoTimelineError(Exception):
"""Sentinel raised by _require_timeline when no timelines exist."""
def _resolve_io_paths(
arguments: dict,
suffix: str = "_modified",
) -> tuple[str, str]:
"""Validate input filepath and resolve the output path.
Shared foundation for every handler that reads an FCPXML and writes
a derived file. Validates the input, falls back to a suffixed
output name when ``output_path`` is not supplied, and sandbox-checks
the result.
Args:
arguments: Tool arguments dict (must contain ``filepath``; may
contain ``output_path``).
suffix: Default output filename suffix when ``output_path`` is
not provided (e.g. ``"_modified"``, ``"_beats"``).
Returns:
``(filepath, output_path)`` tuple with both paths validated.
"""
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
# Anchor write operations to the input file's directory so LLM-generated
# tool calls cannot write to arbitrary filesystem locations (e.g.
# /etc/cron.d/backdoor). When the explicit sandbox is off, the anchor
# still prevents writes outside the source directory tree.
# `output_dir` is where the caller wants the file written, not merely a
# sandbox boundary: the app's "Pasta do projeto" promises that everything
# generated lands there. Deriving the name from the input but keeping the
# input's directory made every cross-directory call fail its own anchor
# check ("output path escapes allowed directory"), so the setting silently
# only worked when it pointed at the directory the file was already going
# to. An explicit `output_path` still wins, and still has to sit inside
# the anchor.
output_dir = arguments.get("output_dir")
if output_dir:
anchor = _validate_directory(str(output_dir))
default_output = str(Path(anchor) / Path(generate_output_path(filepath, suffix)).name)
else:
anchor = str(Path(filepath).resolve().parent)
default_output = generate_output_path(filepath, suffix)
output_path = _validate_output_path(
arguments.get("output_path") or default_output,
anchor_dir=anchor,
)
return filepath, output_path
def _setup_modifier(
arguments: dict,
suffix: str = "_modified",
) -> tuple[str, str, "FCPXMLModifier"]:
"""Common setup for write handlers: validate paths and create modifier.
Consolidates the repeated validate-filepath → resolve-output-path →
create-modifier boilerplate shared by 18+ write handlers.
Args:
arguments: Tool arguments dict (must contain ``filepath``; may
contain ``output_path``).
suffix: Default output filename suffix when ``output_path`` is
not provided (e.g. ``"_modified"``, ``"_flash_fixed"``).
Returns:
``(filepath, output_path, modifier)`` tuple ready for the
handler's domain-specific operation.
"""
filepath, output_path = _resolve_io_paths(arguments, suffix)
modifier = FCPXMLModifier(filepath)
return filepath, output_path, modifier
def _setup_generator(
arguments: dict,
suffix: str = "_roughcut",
) -> tuple[str, str, "RoughCutGenerator"]:
"""Common setup for generation handlers: validate paths and create generator.
Args:
arguments: Tool arguments dict (must contain ``filepath`` and
``output_path``).
suffix: Default output filename suffix.
Returns:
``(filepath, output_path, generator)`` tuple.
"""
filepath, output_path = _resolve_io_paths(arguments, suffix)
generator = RoughCutGenerator(filepath)
return filepath, output_path, generator
def _parse_timestamp_parts(
parts: list[str], *, frame_rate: float = 24.0
) -> float | None:
"""Convert colon-separated timestamp parts to total seconds.
Handles 2-part (M:SS), 3-part (H:MM:SS / HH:MM:SS.ms), and
4-part (HH:MM:SS:FF SMPTE) formats. Returns ``None`` when the
part count is unrecognised so callers can skip.
Args:
parts: Colon-split timestamp components.
frame_rate: FPS used to convert the frame component of SMPTE
timecodes into fractional seconds (default 24.0).
"""
if len(parts) == 2:
return int(parts[0]) * 60 + float(parts[1])
elif len(parts) == 3:
return int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2])
elif len(parts) == 4:
# SMPTE: HH:MM:SS:FF — convert frames to fractional seconds
base = int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2])
frames = int(parts[3])
return base + (frames / frame_rate) if frame_rate > 0 else base
return None
def _raw_markers_to_batch(
raw_markers: list[dict],
marker_type: str = "chapter",
max_label: int | None = None,
) -> list[dict]:
"""Convert raw {seconds, text} marker dicts to batch_add_markers format.
Shared by import_srt_markers and import_transcript_markers.
"""
batch = []
for m in raw_markers:
label = m["text"]
if max_label and len(label) > max_label:
label = label[:max_label]
batch.append({
"timecode": f"{m['seconds']}s",
"name": label,
"marker_type": marker_type.upper(),
})
return batch
def _extract_subtitle_blocks(text: str, *, strip_vtt_tags: bool = False) -> list[dict]:
"""Extract timestamp/text pairs from subtitle cue blocks (SRT or VTT).
Both SRT and VTT use the same ``start --> end`` cue syntax with
text lines underneath; only header stripping and tag cleaning differ.
"""
markers = []
blocks = re.split(r'\n\s*\n', text.strip())
for block in blocks:
lines = block.strip().split('\n')
if len(lines) < 2:
continue
ts_line = None
text_lines = []
for line in lines:
if '-->' in line:
ts_line = line
elif ts_line is not None:
if strip_vtt_tags:
line = re.sub(r'<[^>]+>', '', line)
cleaned = line.strip()
if cleaned:
text_lines.append(cleaned)
if not ts_line or not text_lines:
continue
start_str = ts_line.split('-->')[0].strip().replace(',', '.')
seconds = _parse_timestamp_parts(start_str.split(':'))
if seconds is not None:
markers.append({'seconds': seconds, 'text': ' '.join(text_lines)})
return markers
def parse_srt(text: str) -> list[dict]:
"""Parse SRT subtitle format into timestamp/text pairs."""
return _extract_subtitle_blocks(text)
def parse_vtt(text: str) -> list[dict]:
"""Parse WebVTT subtitle format into timestamp/text pairs."""
text = re.sub(r'^WEBVTT.*?\n', '', text, flags=re.MULTILINE)
text = re.sub(r'NOTE\n.*?\n\n', '', text, flags=re.DOTALL)
return _extract_subtitle_blocks(text, strip_vtt_tags=True)
def parse_transcript_timestamps(text: str) -> list[dict]:
"""Parse timestamped text (YouTube description format) into markers.
Supports formats like:
0:00 Introduction
00:01:30 Main Topic
1:05:30 Conclusion
00:00:00:00 SMPTE timecode
"""
markers = []
for line in text.strip().split('\n'):
line = line.strip()
if not line:
continue
match = re.match(r'^(\d{1,2}:\d{2}(?::\d{2}){0,2})\s+(.+)$', line)
if match:
seconds = _parse_timestamp_parts(match.group(1).split(':'))
if seconds is not None:
markers.append({'seconds': seconds, 'text': match.group(2).strip()})
return markers
def _detect_flash_frames(
tl: Any, *, critical_threshold: int = 2, warning_threshold: int = 6,
) -> list:
"""Find clips shorter than *warning_threshold* frames.
Returns a list of ``FlashFrame`` objects sorted by severity. Shared by
``handle_detect_flash_frames`` and ``handle_validate_timeline`` so the
detection logic lives in exactly one place.
"""
fps = tl.frame_rate
flash_frames: list[FlashFrame] = []
for clip in tl.clips:
duration_frames = int(clip.duration_seconds * fps)
if duration_frames < warning_threshold:
severity = (
FlashFrameSeverity.CRITICAL
if duration_frames < critical_threshold
else FlashFrameSeverity.WARNING
)
flash_frames.append(FlashFrame(
clip_name=clip.name, clip_id=clip.name,
start=clip.start, duration_frames=duration_frames,
duration_seconds=clip.duration_seconds, severity=severity,
))
return flash_frames
def _detect_gaps(tl: Any, *, min_gap_frames: int = 1) -> list:
"""Find inter-clip gaps of at least *min_gap_frames* length.
Returns a list of ``GapInfo`` objects. Shared by ``handle_detect_gaps``
and ``handle_validate_timeline``.
"""
fps = tl.frame_rate
min_gap_seconds = min_gap_frames / fps
gaps: list[GapInfo] = []
sorted_clips = sorted(tl.clips, key=lambda c: c.start.seconds)
for i in range(len(sorted_clips) - 1):
current_end = sorted_clips[i].end.seconds
next_start = sorted_clips[i + 1].start.seconds
gap_duration = next_start - current_end
if gap_duration >= min_gap_seconds:
gaps.append(GapInfo(
start=Timecode(frames=int(current_end * fps), frame_rate=fps),
duration_frames=int(gap_duration * fps),
duration_seconds=gap_duration,
previous_clip=sorted_clips[i].name,
next_clip=sorted_clips[i + 1].name,
))
return gaps
def _detect_duplicate_groups(tl: Any, *, mode: str = "same_source") -> list:
"""Group clips that share a source media reference.
Returns a list of ``DuplicateGroup`` objects. Shared by
``handle_detect_duplicates`` and ``handle_validate_timeline``.
"""
source_groups: dict[str, list[dict]] = {}
for clip in tl.clips:
source_key = clip.media_path or clip.name
if source_key not in source_groups:
source_groups[source_key] = []
source_groups[source_key].append({
'name': clip.name,
'start': clip.start.seconds,
'duration': clip.duration_seconds,
'source_start': clip.source_start.seconds if clip.source_start else 0,
'source_duration': clip.duration_seconds,
'timecode': format_timecode(clip.start),
})
duplicates: list[DuplicateGroup] = []
for source_key, clips in source_groups.items():
if len(clips) <= 1:
continue
group = DuplicateGroup(
source_ref=source_key,
source_name=source_key.split('/')[-1] if '/' in source_key else source_key,
clips=clips,
)
if mode == "same_source":
duplicates.append(group)
elif mode == "overlapping_ranges" and group.has_overlapping_ranges:
duplicates.append(group)
elif mode == "identical":
seen_ranges: set[tuple] = set()
identical_clips = []
for c in clips:
range_key = (c['source_start'], c['source_duration'])
if range_key in seen_ranges:
identical_clips.append(c)
seen_ranges.add(range_key)
if identical_clips:
group.clips = identical_clips
duplicates.append(group)
return duplicates
AUDIO_MEDIA_EXTENSIONS = (
'.wav', '.aif', '.aiff', '.mp3', '.m4a', '.aac', '.flac', '.mov', '.mp4',
)
_DIARIZATION_INSTALL_HINT = (
"\n\nInstall the optional diarization extra:\n\n"
" pip install 'fcp-mcp-server[diarization]'\n\n"
"and set a HuggingFace token with access to "
"pyannote/speaker-diarization-3.1 (pass hf_token= or persist one via "
"save_hf_token)."
)
_FEATURES_INSTALL_HINT = (
"\n\nInstall the optional media-intelligence extra:\n\n"
" pip install 'fcp-mcp-server[intelligence]'"
)
def _voice_analysis_config_text(config: dict) -> str:
w = config["emphasis_weights"]
text = "# Voice Analysis Settings\n\n"
text += _markdown_table(
["Setting", "Value"],
[
["Energy threshold", f"{config['energy_threshold']:.2f}"],
["Peak selection", f"top {config['peak_percentile']:.1%} of words"],
["Emphasis floor", f"{config['emphasis_floor']:.2f}"],
["Emotion detection", "on" if config["emotion_enabled"] else "off"],
["Emotion sensitivity", f"{config['emotion_sensitivity']:.2f}"],
],
) + "\n\n## Emphasis Weights\n"
text += _markdown_table(
["Factor", "Weight"],
[[k.replace("_", " ").title(), f"{v:.2f}"] for k, v in w.items()],
)
return text
def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str:
"""Apply one non-cut action to the clip that hosts it.
``clip_start`` is where that clip begins on the timeline; the writer
wants times relative to the clip's own head, so the rebase happens here
— the single place that knows about the conversion. The clip *element*
is passed through rather than its name: after a cut the pieces share a
name, and a name lookup would land every edit on the first piece.
"""
rel_start = action.start - clip_start
rel_end = action.end - clip_start
if action.kind == "zoom":
# Only forward an explicit ease — otherwise add_zoom's own default
# (a fast ramp in, instant snap back out) is what should apply.
zoom_args = {}
if action.params.get("ease") is not None:
zoom_args["ease"] = float(action.params["ease"])
if action.params.get("ease_out") is not None:
zoom_args["ease_out"] = float(action.params["ease_out"])
modifier.add_zoom(
clip_id=clip_el,
start=rel_start,
end=rel_end,
scale=float(action.params.get("scale", 1.3)),
**zoom_args,
)
return f"zoom {action.params.get('scale', 1.3):.2f}x"
if action.kind == "text":
modifier.add_text_title(
clip_el,
action.params["content"],
offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(),
)
return f"text \"{action.params['content'][:24]}\""
# marker
modifier.add_marker(
clip_id=clip_el,
timecode=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
name=action.params.get("content") or action.reason or "Voice action",
note=action.reason or None,
)
return "marker"
def _speaker_table(profiles: Sequence[dict]) -> str:
"""Who was detected, ordered by how much of the runtime each holds."""
return _markdown_table(
["ID", "Name", "Share", "Speaking", "Lines", "Avg line"],
[
[
p["id"],
p.get("name", ""),
f"{p['share']:.0%}",
format_duration(p["speaking_seconds"]),
str(p["segment_count"]),
f"{p['avg_segment']:.1f}s",
]
for p in profiles
],
)
TRANSCRIBE_MAX_MEDIA = 10
_TRANSCRIBE_INSTALL_HINT = (
"\n\nInstall the optional transcription extra:\n\n"
" pip install 'fcp-mcp-server[transcribe]'\n\n"
"or run via uvx:\n\n"
" uvx --from \"fcp-mcp-server[transcribe]\" fcp-mcp-server"
)
def _transcript_json_path(media_path: str, output_dir: str | None = None) -> Path:
"""Where the ``_transcript.json`` for ``media_path`` lives.
When ``output_dir`` (the user-selected project folder) is set, the
transcript is saved/read there instead of next to the source media.
"""
p = Path(media_path)
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
return directory / f"{p.stem}_transcript.json"
return p.with_name(p.stem + "_transcript.json")
def _load_or_transcribe(
media_path: str, model: str, language: str | None, output_dir: str | None = None
) -> tuple[dict | None, str]:
"""Load a cached ``_transcript.json`` for a media file, else transcribe and cache it.
Returns ``(transcript, "")`` or ``(None, reason)``. The cache makes
transcription a one-time cost per media file across all transcript tools.
"""
json_path = _transcript_json_path(media_path, output_dir)
if json_path.is_file():
try:
with open(json_path) as f:
data = json.load(f)
if isinstance(data, dict) and isinstance(data.get("words"), list):
return data, ""
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
pass # unreadable cache falls through to re-transcribe
result = transcribe(media_path, model_size=model, language=language)
if result is None:
return None, "untranscribable (faster-whisper not installed or media unreadable)"
anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent)
out_path = _validate_output_path(str(json_path), anchor_dir=anchor)
with open(out_path, "w") as f:
json.dump({"source": Path(media_path).name, **result}, f, indent=2)
return result, ""
def _cut_transcript_spans(modifier, clip_filter, model, language, padding, spans_fn, keep_only=False, output_dir=None):
"""Shared cut engine for transcript-driven editing.
``spans_fn(words) -> [(start, end), ...]`` in source seconds. Spans are
padded, clamped to each clip's used source window, optionally inverted
(keep_only), snapped to the frame grid, and cut with ripple.
"""
to_frame = modifier.snap_seconds_to_frame
cache: dict[str, tuple] = {}
cuts_made: list[tuple[str, int, float]] = []
skipped: list[tuple[str, str]] = []
spine_clips = [el for _, el in modifier._iter_spine_clips()]
for el in spine_clips:
name = el.get("name", "")
if clip_filter and name != clip_filter:
continue
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
media_path = media_src_to_path(src)
if not media_path or not Path(media_path).is_file():
skipped.append((name, "media file missing"))
continue
if media_path not in cache:
if len(cache) >= TRANSCRIBE_MAX_MEDIA:
skipped.append((name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)"))
continue
cache[media_path] = _load_or_transcribe(media_path, model, language, output_dir)
data, reason = cache[media_path]
if data is None:
skipped.append((name, reason))
continue
clip_source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
window_start = clip_source_start
window_end = clip_source_start + clip_duration
spans = spans_fn(data.get("words", []))
padded = merge_ranges([(s - padding, e + padding) for s, e in spans])
clamped = [
(max(s, window_start), min(e, window_end))
for s, e in padded
if min(e, window_end) > max(s, window_start)
]
if keep_only:
if not clamped:
# Never delete a whole clip just because nothing matched in it.
skipped.append((name, "no phrase matches — left untouched (keep_only)"))
continue
cut_source = invert_ranges(clamped, window_start, window_end)
else:
cut_source = clamped
cut_ranges = [
(to_frame(s - clip_source_start), to_frame(e - clip_source_start))
for s, e in cut_source
]
cut_ranges = [(a, b) for a, b in cut_ranges if b > a]
if not cut_ranges:
continue
removed = modifier.cut_clip_ranges(el, cut_ranges)
if removed > TimeValue.zero():
cuts_made.append((name, len(cut_ranges), removed.to_seconds()))
return cuts_made, skipped
def _transcript_cut_report(title, summary_lines, cuts_made, skipped, output_path, footer):
if not cuts_made:
text = f"# {title}\n\nNo cuts to make — file unchanged (nothing saved)."
if skipped:
text += "\n\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
)
if any("faster-whisper" in reason for _, reason in skipped):
text += _TRANSCRIBE_INSTALL_HINT
return _text_result(text)
total_removed = sum(seconds for _, _, seconds in cuts_made)
result = f"# {title}\n\n## Summary\n"
result += "\n".join(summary_lines) + "\n"
result += f"- **Clips Cut**: {len(cuts_made)}\n- **Total Removed**: {format_duration(total_removed)}\n"
result += "\n## Cuts\n"
result += _markdown_table(
["Clip", "Ranges Cut", "Removed"],
[[name, str(count), f"{seconds:.2f}s"] for name, count, seconds in cuts_made],
) + "\n"
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
) + "\n"
result += f"\nSaved to: {output_path}\n\n{footer}"
return _text_result(result)
+649
View File
@@ -0,0 +1,649 @@
"""Edição — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.models import MarkerType
from fcpxml.writer import FCPXMLModifier
from server_tools._shared import (
_format_batch_result,
_resolve_io_paths,
_setup_modifier,
_text_result,
format_duration,
)
TOOLS = [
Tool(
name="add_marker",
description="Add a marker at a specific timecode",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"timecode": {"type": "string", "description": "Position (00:00:10:00 or 10s)"},
"name": {"type": "string", "description": "Marker label"},
"marker_type": {"type": "string", "enum": ["standard", "chapter", "todo", "completed"], "default": "standard"},
"note": {"type": "string", "description": "Optional note"},
"output_path": {"type": "string", "description": "Output path (default: adds _modified suffix)"}
},
"required": ["filepath", "timecode", "name"]
}
),
Tool(
name="batch_add_markers",
description="Add multiple markers at once, or auto-generate at cuts/intervals",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"markers": {
"type": "array",
"items": {
"type": "object",
"properties": {
"timecode": {"type": "string"},
"name": {"type": "string"},
"marker_type": {"type": "string"},
"note": {"type": "string"}
}
},
"description": "List of markers to add"
},
"auto_at_cuts": {"type": "boolean", "description": "Add marker at every cut"},
"auto_at_intervals": {"type": "string", "description": "Add markers every N seconds (e.g., '30s')"},
"output_path": {"type": "string"}
},
"required": ["filepath"]
}
),
Tool(
name="trim_clip",
description="Trim a clip's in-point and/or out-point",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"clip_id": {"type": "string", "description": "Clip name or ID"},
"trim_start": {"type": "string", "description": "New in-point or delta (+1s, -10f)"},
"trim_end": {"type": "string", "description": "New out-point or delta"},
"ripple": {"type": "boolean", "default": True, "description": "Shift subsequent clips"},
"output_path": {"type": "string"}
},
"required": ["filepath", "clip_id"]
}
),
Tool(
name="reorder_clips",
description="Move clips to a new position in the timeline",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"clip_ids": {"type": "array", "items": {"type": "string"}, "description": "Clips to move"},
"target_position": {"type": "string", "description": "'start', 'end', timecode, or 'after:clip_id'"},
"ripple": {"type": "boolean", "default": True},
"output_path": {"type": "string"}
},
"required": ["filepath", "clip_ids", "target_position"]
}
),
Tool(
name="add_transition",
description="Add a transition between clips",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"clip_id": {"type": "string", "description": "Clip to add transition to"},
"position": {"type": "string", "enum": ["start", "end", "both"], "default": "end"},
"transition_type": {"type": "string", "enum": ["cross-dissolve", "fade-to-black", "fade-from-black", "wipe"], "default": "cross-dissolve"},
"duration": {"type": "string", "default": "00:00:00:15"},
"output_path": {"type": "string"}
},
"required": ["filepath", "clip_id"]
}
),
Tool(
name="change_speed",
description="Change clip playback speed (slow motion or speed up)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"clip_id": {"type": "string"},
"speed": {"type": "number", "description": "Speed multiplier (0.5 = half, 2.0 = double)"},
"preserve_pitch": {"type": "boolean", "default": True},
"output_path": {"type": "string"}
},
"required": ["filepath", "clip_id", "speed"]
}
),
Tool(
name="add_zoom",
description="Add a smooth ease-in/ease-out punch-in zoom to a clip, animating <adjust-transform>'s scale param via keyframes (100% -> scale -> 100%) entirely within [start, end] (clip-relative seconds, i.e. seconds from the clip's own head). The ease portions each last `ease` seconds; the zoom holds at `scale` in between. Replaces any existing zoom on the same clip rather than stacking.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"clip_id": {"type": "string", "description": "Name/ID of the clip to zoom"},
"start": {"type": "number", "description": "Clip-relative seconds where the ease-in begins"},
"end": {"type": "number", "description": "Clip-relative seconds where the ease-out ends (back to 100%)"},
"scale": {"type": "number", "default": 1.3, "description": "Zoom scale, e.g. 1.3 = 130%"},
"ease": {"type": "number", "default": 0.3, "description": "Seconds for each of the ease-in/ease-out portions (must fit: 2*ease <= end-start)"},
"position": {"type": "string", "default": "0 0", "description": "Optional pan offset \"x y\" applied for the duration of the transform"},
"output_path": {"type": "string"}
},
"required": ["filepath", "clip_id", "start", "end"]
}
),
Tool(
name="delete_clips",
description="Delete clips from timeline",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"clip_ids": {"type": "array", "items": {"type": "string"}},
"ripple": {"type": "boolean", "default": True, "description": "Close gaps after deletion"},
"output_path": {"type": "string"}
},
"required": ["filepath", "clip_ids"]
}
),
Tool(
name="split_clip",
description="Split a clip at specified timecodes",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"clip_id": {"type": "string"},
"split_points": {"type": "array", "items": {"type": "string"}, "description": "Timecodes to split at"},
"output_path": {"type": "string"}
},
"required": ["filepath", "clip_id", "split_points"]
}
),
Tool(
name="insert_clip",
description="Insert a library clip onto the timeline at a specific position",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"asset_id": {"type": "string", "description": "Asset reference ID (e.g., 'r3')"},
"asset_name": {"type": "string", "description": "Asset name (alternative to asset_id)"},
"position": {"type": "string", "description": "'start', 'end', timecode, or 'after:clip_name'"},
"duration": {"type": "string", "description": "Clip duration (if not using in/out points)"},
"in_point": {"type": "string", "description": "Source in-point for subclip"},
"out_point": {"type": "string", "description": "Source out-point for subclip"},
"ripple": {"type": "boolean", "default": True, "description": "Shift subsequent clips"},
"output_path": {"type": "string", "description": "Output path (default: adds _modified suffix)"}
},
"required": ["filepath", "position"]
}
),
Tool(
name="fix_flash_frames",
description="Automatically fix detected flash frames by extending neighbors or deleting",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"mode": {"type": "string", "enum": ["extend_previous", "extend_next", "delete", "auto"], "default": "auto", "description": "How to fix: extend previous/next clip, delete, or auto"},
"threshold_frames": {"type": "integer", "default": 6, "description": "Frames below this threshold are flash frames"},
"output_path": {"type": "string", "description": "Output path (default: adds _modified suffix)"}
},
"required": ["filepath"]
}
),
Tool(
name="rapid_trim",
description="Batch trim clips to a maximum duration for fast-paced montages",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"max_duration": {"type": "string", "description": "Maximum clip duration (e.g., '2s', '00:00:02:00')"},
"min_duration": {"type": "string", "description": "Minimum clip duration (optional)"},
"keywords": {"type": "array", "items": {"type": "string"}, "description": "Only trim clips with these keywords"},
"trim_from": {"type": "string", "enum": ["start", "end", "center"], "default": "end", "description": "Where to trim from"},
"output_path": {"type": "string", "description": "Output path (default: adds _modified suffix)"}
},
"required": ["filepath", "max_duration"]
}
),
Tool(
name="fill_gaps",
description="Automatically fill gaps in the timeline by extending adjacent clips",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"mode": {"type": "string", "enum": ["extend_previous", "extend_next", "delete"], "default": "extend_previous", "description": "How to fill gaps"},
"max_gap": {"type": "string", "description": "Only fill gaps smaller than this (e.g., '1s')"},
"output_path": {"type": "string", "description": "Output path (default: adds _modified suffix)"}
},
"required": ["filepath"]
}
),
Tool(
name="add_connected_clip",
description="Connect a library clip to an existing timeline clip (B-roll overlay, audio, title)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"parent_clip_id": {"type": "string", "description": "Name/ID of the clip to attach to"},
"asset_id": {"type": "string", "description": "Asset reference ID"},
"asset_name": {"type": "string", "description": "Asset name (alternative to asset_id)"},
"offset": {"type": "string", "default": "0s", "description": "Position relative to parent clip start"},
"duration": {"type": "string", "description": "Duration (default: full asset)"},
"lane": {"type": "integer", "default": 1, "description": "Lane number (positive=above, negative=below)"},
"output_path": {"type": "string", "description": "Output path (default: adds _modified suffix)"}
},
"required": ["filepath", "parent_clip_id"]
}
),
Tool(
name="reformat_timeline",
description="Create new FCPXML with different resolution/aspect ratio (9:16 for TikTok, 1:1 for Instagram, etc.)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"format": {"type": "string", "enum": ["9:16", "1:1", "4:5", "16:9", "4:3", "custom"], "description": "Target format preset"},
"width": {"type": "integer", "description": "Custom width (only with format='custom')"},
"height": {"type": "integer", "description": "Custom height (only with format='custom')"},
"output_path": {"type": "string", "description": "Output path (default: adds _reformatted suffix)"}
},
"required": ["filepath", "format"]
}
),
Tool(
name="add_audio",
description="Add an audio clip or music bed to the timeline",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"parent_clip_id": {"type": "string", "description": "Clip to attach audio to (omit for music bed spanning full timeline)"},
"asset_id": {"type": "string", "description": "Existing asset reference ID"},
"src": {"type": "string", "description": "Path to audio file (creates new asset)"},
"offset": {"type": "string", "description": "Position relative to parent clip start", "default": "0s"},
"duration": {"type": "string", "description": "Duration of audio clip"},
"role": {"type": "string", "description": "Audio role (dialogue, music, effects, etc.)", "default": "dialogue"},
"lane": {"type": "integer", "description": "Lane number (negative = below)", "default": -1},
"output_path": {"type": "string", "description": "Output path"},
},
"required": ["filepath"]
}
),
Tool(
name="create_compound_clip",
description="Group spine clips into a compound clip",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"clip_ids": {"type": "array", "items": {"type": "string"}, "description": "Clip IDs to group"},
"name": {"type": "string", "description": "Name for the compound clip", "default": "Compound Clip"},
"output_path": {"type": "string", "description": "Output path"},
},
"required": ["filepath", "clip_ids"]
}
),
Tool(
name="flatten_compound_clip",
description="Flatten a compound clip back into individual clips in the spine",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"ref_clip_id": {"type": "string", "description": "ID of the ref-clip to flatten"},
"output_path": {"type": "string", "description": "Output path"},
},
"required": ["filepath", "ref_clip_id"]
}
),
]
async def handle_add_marker(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments)
marker_type = MarkerType.from_string(arguments.get("marker_type", "standard"))
modifier.add_marker_at_timeline(
timecode=arguments["timecode"], name=arguments["name"],
marker_type=marker_type, note=arguments.get("note"),
)
modifier.save(output_path)
return _text_result(f"Added marker '{arguments['name']}' at {arguments['timecode']}\n\nSaved to: {output_path}")
async def handle_batch_add_markers(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments)
markers_added = modifier.batch_add_markers(
markers=arguments.get("markers", []),
auto_at_cuts=arguments.get("auto_at_cuts", False),
auto_at_intervals=arguments.get("auto_at_intervals"),
)
modifier.save(output_path)
return _text_result(f"Added {len(markers_added)} markers\n\nSaved to: {output_path}")
async def handle_trim_clip(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments)
modifier.trim_clip(
clip_id=arguments["clip_id"],
trim_start=arguments.get("trim_start"),
trim_end=arguments.get("trim_end"),
ripple=arguments.get("ripple", True),
)
modifier.save(output_path)
return _text_result(f"Trimmed clip '{arguments['clip_id']}'\n\nSaved to: {output_path}")
async def handle_reorder_clips(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments)
modifier.reorder_clips(
clip_ids=arguments["clip_ids"],
target_position=arguments["target_position"],
ripple=arguments.get("ripple", True),
)
modifier.save(output_path)
clips_moved = ", ".join(arguments["clip_ids"])
return _text_result(f"Moved clips [{clips_moved}] to {arguments['target_position']}\n\nSaved to: {output_path}")
async def handle_add_transition(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments)
modifier.add_transition(
clip_id=arguments["clip_id"],
position=arguments.get("position", "end"),
transition_type=arguments.get("transition_type", "cross-dissolve"),
duration=arguments.get("duration", "00:00:00:15"),
)
modifier.save(output_path)
return _text_result(f"Added {arguments.get('transition_type', 'cross-dissolve')} to '{arguments['clip_id']}'\n\nSaved to: {output_path}")
async def handle_change_speed(arguments: dict) -> Sequence[TextContent]:
speed = arguments["speed"]
if not isinstance(speed, (int, float)) or speed <= 0 or speed > 100:
raise ValueError(
f"Speed must be a positive number between 0 (exclusive) and 100, got {speed!r}"
)
filepath, output_path, modifier = _setup_modifier(arguments)
modifier.change_speed(
clip_id=arguments["clip_id"],
speed=speed,
preserve_pitch=arguments.get("preserve_pitch", True),
)
modifier.save(output_path)
speed_desc = f"{speed}x" if speed >= 1 else f"{int(1/speed)}x slow motion"
return _text_result(f"Changed speed of '{arguments['clip_id']}' to {speed_desc}\n\nSaved to: {output_path}")
async def handle_add_zoom(arguments: dict) -> Sequence[TextContent]:
start = float(arguments["start"])
end = float(arguments["end"])
scale = float(arguments.get("scale", 1.3))
ease = float(arguments.get("ease", 0.3))
position = arguments.get("position", "0 0")
filepath, output_path, modifier = _setup_modifier(arguments)
modifier.add_zoom(
clip_id=arguments["clip_id"], start=start, end=end,
scale=scale, ease=ease, position=position,
)
modifier.save(output_path)
return _text_result(
f"# Zoom Added\n\n"
f"- **Clip**: {arguments['clip_id']}\n"
f"- **Window**: {start}s → {end}s (clip-relative)\n"
f"- **Scale**: {int(scale * 100)}%\n"
f"- **Ease**: {ease}s in/out\n\n"
f"Saved to: {output_path}"
)
async def handle_delete_clips(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments)
modifier.delete_clip(
clip_ids=arguments["clip_ids"],
ripple=arguments.get("ripple", True),
)
modifier.save(output_path)
return _text_result(f"Deleted {len(arguments['clip_ids'])} clip(s)\n\nSaved to: {output_path}")
async def handle_split_clip(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments)
new_clips = modifier.split_clip(
clip_id=arguments["clip_id"],
split_points=arguments["split_points"],
)
modifier.save(output_path)
return _text_result(f"Split '{arguments['clip_id']}' into {len(new_clips)} clips\n\nSaved to: {output_path}")
async def handle_insert_clip(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments)
new_clip = modifier.insert_clip(
asset_id=arguments.get("asset_id"),
asset_name=arguments.get("asset_name"),
position=arguments["position"],
duration=arguments.get("duration"),
in_point=arguments.get("in_point"),
out_point=arguments.get("out_point"),
ripple=arguments.get("ripple", True),
)
modifier.save(output_path)
clip_name = new_clip.get('name', 'Unknown')
pos = arguments["position"]
return _text_result(f"Inserted '{clip_name}' at position '{pos}'\n\nSaved to: {output_path}")
async def handle_fix_flash_frames(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments, "_flash_fixed")
fixed = modifier.fix_flash_frames(
mode=arguments.get("mode", "auto"),
threshold_frames=arguments.get("threshold_frames", 6),
)
modifier.save(output_path)
if not fixed:
return _text_result("No flash frames found to fix.")
result = _format_batch_result(
title="Flash Frames Fixed",
summary={"Fixed": f"{len(fixed)} flash frames", "Mode": arguments.get('mode', 'auto')},
headers=["Clip", "Frames", "Action", "Result"],
rows=[
[f['clip_name'], f"{f['duration_frames']}f", f['action'], f"Extended: {f.get('extended_clip', 'N/A')}"]
for f in fixed
],
output_path=output_path,
)
return _text_result(result)
async def handle_rapid_trim(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments, "_rapid_trim")
trimmed = modifier.rapid_trim(
max_duration=arguments["max_duration"],
min_duration=arguments.get("min_duration"),
keywords=arguments.get("keywords"),
trim_from=arguments.get("trim_from", "end"),
)
modifier.save(output_path)
if not trimmed:
return _text_result(f"No clips exceeded {arguments['max_duration']} - nothing trimmed.")
total_before = sum(t['original_duration'] for t in trimmed)
total_after = sum(t['new_duration'] for t in trimmed)
result = _format_batch_result(
title="Rapid Trim Complete",
summary={
"Clips Trimmed": str(len(trimmed)),
"Max Duration": str(arguments['max_duration']),
"Trim From": arguments.get('trim_from', 'end'),
"Time Saved": format_duration(total_before - total_after),
},
headers=["Clip", "Before", "After"],
rows=[
[t['clip_name'], format_duration(t['original_duration']), format_duration(t['new_duration'])]
for t in trimmed
],
output_path=output_path,
)
return _text_result(result)
async def handle_fill_gaps(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments, "_gaps_filled")
filled = modifier.fill_gaps(
mode=arguments.get("mode", "extend_previous"),
max_gap=arguments.get("max_gap"),
)
modifier.save(output_path)
if not filled:
return _text_result("No gaps found to fill.")
result = _format_batch_result(
title="Gaps Filled",
summary={"Gaps Filled": str(len(filled)), "Mode": arguments.get('mode', 'extend_previous')},
headers=["Position", "Duration", "Action"],
rows=[[g['timecode'], f"{g['duration_frames']}f", g['action']] for g in filled],
output_path=output_path,
)
return _text_result(result)
async def handle_add_connected_clip(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments)
modifier.add_connected_clip(
parent_clip_id=arguments["parent_clip_id"],
asset_id=arguments.get("asset_id"),
asset_name=arguments.get("asset_name"),
offset=arguments.get("offset", "0s"),
duration=arguments.get("duration"),
lane=arguments.get("lane", 1),
)
modifier.save(output_path)
return _text_result((
f"Connected clip added to '{arguments['parent_clip_id']}' on lane {arguments.get('lane', 1)}\n\n"
f"Saved to: `{output_path}`"
))
async def handle_reformat_timeline(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_reformatted")
fmt = arguments["format"]
if fmt == "custom":
width = arguments.get("width")
height = arguments.get("height")
if not width or not height:
return _text_result("Custom format requires both 'width' and 'height' parameters.")
else:
formats = FCPXMLModifier.SOCIAL_FORMATS
if fmt not in formats:
return _text_result(f"Unknown format: {fmt}. Valid: {', '.join(formats.keys())}")
width, height = formats[fmt]
modifier = FCPXMLModifier(filepath)
modifier.reformat_resolution(width, height)
modifier.save(output_path)
return _text_result((
f"# Timeline Reformatted\n\n"
f"- **Format**: {fmt} ({width}x{height})\n"
f"- **Aspect ratio**: {width}:{height}\n\n"
f"Saved to: `{output_path}`\n\n"
f"**Next step**: Import into FCP (File > Import > XML). "
f"FCP will handle spatial conforming automatically."
))
async def handle_add_audio(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments, "_audio")
parent_clip_id = arguments.get("parent_clip_id")
if parent_clip_id:
modifier.add_audio_clip(
parent_clip_id=parent_clip_id,
asset_id=arguments.get("asset_id"),
offset=arguments.get("offset", "0s"),
duration=arguments.get("duration"),
role=arguments.get("role", "dialogue"),
lane=arguments.get("lane", -1),
src=arguments.get("src"),
)
action = f"Added audio clip to '{parent_clip_id}'"
else:
modifier.add_music_bed(
asset_id=arguments.get("asset_id"),
duration=arguments.get("duration"),
role=arguments.get("role", "music"),
src=arguments.get("src"),
)
action = "Added music bed spanning full timeline"
modifier.save(output_path)
return _text_result(f"{action}\nSaved to: `{output_path}`")
async def handle_create_compound_clip(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments, "_compound")
clip_ids = arguments["clip_ids"]
name = arguments.get("name", "Compound Clip")
modifier.create_compound_clip(clip_ids, name)
modifier.save(output_path)
return _text_result((
f"Created compound clip '{name}' from {len(clip_ids)} clips.\n"
f"Saved to: `{output_path}`"
))
async def handle_flatten_compound_clip(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments, "_flattened")
ref_clip_id = arguments["ref_clip_id"]
extracted = modifier.flatten_compound_clip(ref_clip_id)
modifier.save(output_path)
return _text_result((
f"Flattened compound clip '{ref_clip_id}' into {len(extracted)} clips.\n"
f"Saved to: `{output_path}`"
))
HANDLERS = {
"add_marker": handle_add_marker,
"batch_add_markers": handle_batch_add_markers,
"trim_clip": handle_trim_clip,
"reorder_clips": handle_reorder_clips,
"add_transition": handle_add_transition,
"change_speed": handle_change_speed,
"add_zoom": handle_add_zoom,
"delete_clips": handle_delete_clips,
"split_clip": handle_split_clip,
"insert_clip": handle_insert_clip,
"fix_flash_frames": handle_fix_flash_frames,
"rapid_trim": handle_rapid_trim,
"fill_gaps": handle_fill_gaps,
"add_connected_clip": handle_add_connected_clip,
"reformat_timeline": handle_reformat_timeline,
"add_audio": handle_add_audio,
"create_compound_clip": handle_create_compound_clip,
"flatten_compound_clip": handle_flatten_compound_clip,
}
+185
View File
@@ -0,0 +1,185 @@
"""Export / relink — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.export import DaVinciExporter
from fcpxml.writer import FCPXMLModifier
from server_tools._shared import (
_require_timeline,
_resolve_io_paths,
_setup_modifier,
_text_result,
_validate_filepath,
format_timecode,
)
TOOLS = [
Tool(
name="export_edl",
description="Generate EDL (Edit Decision List) from timeline",
inputSchema={
"type": "object",
"properties": {"filepath": {"type": "string"}},
"required": ["filepath"]
}
),
Tool(
name="export_csv",
description="Export timeline data to CSV format",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"include": {"type": "array", "items": {"type": "string"}}
},
"required": ["filepath"]
}
),
Tool(
name="export_resolve_xml",
description="Export timeline as DaVinci Resolve compatible FCPXML (simplified v1.9)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"flatten_compounds": {"type": "boolean", "default": True, "description": "Flatten compound clips for compatibility"},
"output_path": {"type": "string", "description": "Output path (default: adds _resolve suffix)"},
},
"required": ["filepath"]
}
),
Tool(
name="export_fcp7_xml",
description="Export timeline as FCP7 XML (XMEML) for Premiere Pro, DaVinci Resolve, and Avid compatibility",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"output_path": {"type": "string", "description": "Output path (default: adds _fcp7.xml suffix)"},
},
"required": ["filepath"]
}
),
Tool(
name="relink_media",
description="Bulk-rewrite media source paths (asset/media-rep src URLs) to relink moved or renamed media folders without opening FCP. Prefix-based: find='/Volumes/OldDrive/Media' replace='/Volumes/NewDrive/Media'. Use dry_run to preview.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file or .fcpxmld bundle"},
"find": {"type": "string", "description": "Old path prefix to match (plain path or file:// URL)"},
"replace": {"type": "string", "description": "New path prefix to substitute"},
"dry_run": {"type": "boolean", "description": "Preview changes without writing", "default": False},
"output_path": {"type": "string", "description": "Output path (default: adds _relinked suffix)"},
},
"required": ["filepath", "find", "replace"]
}
),
]
async def handle_export_edl(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
edl = f"TITLE: {tl.name}\nFCM: NON-DROP FRAME\n\n"
for i, c in enumerate(tl.clips, 1):
edl += f"{i:03d} AX V C {format_timecode(c.source_start)} {format_timecode(c.end)} {format_timecode(c.start)} {format_timecode(c.end)}\n"
edl += f"* FROM CLIP NAME: {c.name}\n\n"
return _text_result(f"```edl\n{edl}```")
async def handle_export_csv(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
csv = "Name,Start,End,Duration,Keywords\n"
for c in tl.clips:
kws = "|".join(k.value for k in c.keywords)
csv += f'"{c.name}",{format_timecode(c.start)},{format_timecode(c.end)},{c.duration_seconds:.3f},"{kws}"\n'
return _text_result(f"```csv\n{csv}```")
async def handle_export_resolve_xml(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_resolve")
exporter = DaVinciExporter(filepath)
exporter.export_simplified_fcpxml(
output_path,
flatten_compounds=arguments.get("flatten_compounds", True),
)
return _text_result((
f"# Exported for DaVinci Resolve\n\n"
f"- **Format**: Simplified FCPXML v1.9\n"
f"- **Compound clips flattened**: {arguments.get('flatten_compounds', True)}\n\n"
f"Saved to: `{output_path}`\n\n"
f"**Next step**: In DaVinci Resolve, go to File > Import > Timeline > Import AAF/EDL/XML"
))
async def handle_export_fcp7_xml(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_fcp7")
exporter = DaVinciExporter(filepath)
exporter.export_xmeml(output_path)
return _text_result((
f"# Exported as FCP7 XML (XMEML)\n\n"
f"- **Format**: XMEML v5\n"
f"- **Compatible with**: Premiere Pro, DaVinci Resolve, Avid Media Composer\n\n"
f"Saved to: `{output_path}`\n\n"
f"**Next step**: Import via File > Import in your target NLE"
))
async def handle_relink_media(arguments: dict) -> Sequence[TextContent]:
dry_run = arguments.get("dry_run", False)
if dry_run:
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
modifier = FCPXMLModifier(filepath)
result = modifier.relink_media(
arguments["find"], arguments["replace"], dry_run=True
)
footer = "Dry run — no file written."
else:
filepath, output_path, modifier = _setup_modifier(arguments, "_relinked")
result = modifier.relink_media(arguments["find"], arguments["replace"])
saved = modifier.save(output_path)
footer = f"Saved to: {saved}"
if not result["relinked"]:
return _text_result(
f"No media paths matched prefix '{arguments['find']}' "
f"({result['total_assets']} assets scanned). Nothing to relink."
)
lines = [
f"{'Would relink' if dry_run else 'Relinked'} "
f"{result['relinked']} media reference(s) "
f"across {result['total_assets']} asset(s):",
"",
]
missing = 0
for change in result["changes"]:
mark = "✓" if change["target_exists"] else "⚠ target missing"
if not change["target_exists"]:
missing += 1
lines.append(f" {change['asset']}: {change['new']} [{mark}]")
if missing:
lines.append("")
lines.append(
f"⚠ {missing} new path(s) do not exist on this machine — "
f"FCP will show those clips as missing until the media is present."
)
lines.append("")
lines.append(footer)
return _text_result("\n".join(lines))
HANDLERS = {
"export_edl": handle_export_edl,
"export_csv": handle_export_csv,
"export_resolve_xml": handle_export_resolve_xml,
"export_fcp7_xml": handle_export_fcp7_xml,
"relink_media": handle_relink_media,
}
+271
View File
@@ -0,0 +1,271 @@
"""Geração — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.models import SegmentSpec
from fcpxml.templates import ClipSpec, apply_template, list_templates
from server_tools._shared import (
PROJECTS_DIR,
_setup_generator,
_text_result,
_validate_output_path,
format_duration,
)
TOOLS = [
Tool(
name="auto_rough_cut",
description="Generate a rough cut from source clips based on keywords, duration, and pacing",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Source FCPXML with clips"},
"output_path": {"type": "string", "description": "Where to save rough cut"},
"target_duration": {"type": "string", "description": "Target length (3m, 00:03:00:00)"},
"pacing": {"type": "string", "enum": ["slow", "medium", "fast", "dynamic"], "default": "medium"},
"keywords": {"type": "array", "items": {"type": "string"}, "description": "Filter clips by keywords"},
"segments": {
"type": "array",
"items": {
"type": "object",
"properties": {
"name": {"type": "string"},
"keywords": {"type": "array", "items": {"type": "string"}},
"duration": {"type": "number"}
}
},
"description": "Segment structure [{name, keywords, duration_seconds}]"
},
"priority": {"type": "string", "enum": ["best", "favorites", "longest", "shortest", "random"], "default": "best"},
"favorites_only": {"type": "boolean", "default": False},
"add_transitions": {"type": "boolean", "default": False}
},
"required": ["filepath", "output_path", "target_duration"]
}
),
Tool(
name="generate_montage",
description="Create rapid-fire montages with pacing curves (accelerating, decelerating, pyramid)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Source FCPXML with clips"},
"output_path": {"type": "string", "description": "Where to save montage"},
"target_duration": {"type": "string", "description": "Total montage length (e.g., '30s', '00:00:30:00')"},
"pacing_curve": {"type": "string", "enum": ["accelerating", "decelerating", "pyramid", "constant"], "default": "accelerating", "description": "How clip duration changes over time"},
"start_duration": {"type": "number", "default": 2.0, "description": "Clip duration at start (seconds)"},
"end_duration": {"type": "number", "default": 0.5, "description": "Clip duration at end (seconds)"},
"keywords": {"type": "array", "items": {"type": "string"}, "description": "Filter clips by keywords"},
"add_transitions": {"type": "boolean", "default": False, "description": "Add quick dissolves"}
},
"required": ["filepath", "output_path", "target_duration"]
}
),
Tool(
name="generate_ab_roll",
description="Create documentary-style A/B roll edits alternating between main content and cutaways",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Source FCPXML with clips"},
"output_path": {"type": "string", "description": "Where to save A/B roll edit"},
"target_duration": {"type": "string", "description": "Total duration (e.g., '3m', '00:03:00:00')"},
"a_keywords": {"type": "array", "items": {"type": "string"}, "description": "Keywords for A-roll (main content, interviews)"},
"b_keywords": {"type": "array", "items": {"type": "string"}, "description": "Keywords for B-roll (cutaways, visuals)"},
"a_duration": {"type": "string", "default": "5s", "description": "Duration of each A-roll segment"},
"b_duration": {"type": "string", "default": "3s", "description": "Duration of each B-roll cutaway"},
"start_with": {"type": "string", "enum": ["a", "b"], "default": "a", "description": "Which roll to start with"},
"add_transitions": {"type": "boolean", "default": True, "description": "Add cross-dissolves"}
},
"required": ["filepath", "output_path", "target_duration", "a_keywords", "b_keywords"]
}
),
Tool(
name="list_templates",
description="List available timeline templates with slot definitions",
inputSchema={
"type": "object",
"properties": {},
}
),
Tool(
name="apply_template",
description="Fill a timeline template with clips and generate FCPXML",
inputSchema={
"type": "object",
"properties": {
"template_name": {"type": "string", "description": "Template name (intro_outro, lower_thirds, music_video)"},
"clips": {"type": "object", "description": "Map of slot_name -> {src, name, duration} or {asset_id, name, duration}"},
"output_path": {"type": "string", "description": "Output FCPXML path"},
"fps": {"type": "number", "description": "Frame rate", "default": 24},
},
"required": ["template_name", "clips", "output_path"]
}
),
]
async def handle_auto_rough_cut(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, generator = _setup_generator(arguments, "_roughcut")
segments = None
if arguments.get("segments"):
segments = [
SegmentSpec(
name=s.get("name", "Segment"),
keywords=s.get("keywords", []),
duration_seconds=s.get("duration", 0),
priority=s.get("priority", "best"),
)
for s in arguments["segments"]
]
result = generator.generate(
output_path=output_path,
target_duration=arguments["target_duration"],
pacing=arguments.get("pacing", "medium"),
keywords=arguments.get("keywords"),
segments=segments,
priority=arguments.get("priority", "best"),
favorites_only=arguments.get("favorites_only", False),
add_transitions=arguments.get("add_transitions", False),
)
return _text_result(f"""# Rough Cut Generated
## Summary
- **Clips Used**: {result.clips_used} of {result.clips_available} available
- **Target Duration**: {format_duration(result.target_duration)}
- **Actual Duration**: {format_duration(result.actual_duration)}
- **Average Clip**: {format_duration(result.average_clip_duration)}
## Output
Saved to: `{result.output_path}`
**Next step**: Import this FCPXML into Final Cut Pro (File > Import > XML)
""")
async def handle_generate_montage(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, generator = _setup_generator(arguments, "_montage")
result = generator.generate_montage(
output_path=output_path,
target_duration=arguments["target_duration"],
pacing_curve=arguments.get("pacing_curve", "accelerating"),
start_duration=arguments.get("start_duration", 2.0),
end_duration=arguments.get("end_duration", 0.5),
keywords=arguments.get("keywords"),
add_transitions=arguments.get("add_transitions", False),
)
curve_desc = {
'accelerating': 'slow to fast (builds energy)',
'decelerating': 'fast to slow (winds down)',
'pyramid': 'slow to fast to slow (dramatic arc)',
'constant': 'same duration throughout',
}
return _text_result(f"""# Montage Generated
## Summary
- **Clips Used**: {result['clips_used']} of {result['clips_available']} available
- **Target Duration**: {format_duration(result['target_duration'])}
- **Actual Duration**: {format_duration(result['actual_duration'])}
- **Pacing Curve**: {result['pacing_curve']} - {curve_desc.get(result['pacing_curve'], '')}
## Pacing
- **Start Clip Duration**: {format_duration(result['start_clip_duration'])}
- **End Clip Duration**: {format_duration(result['end_clip_duration'])}
## Output
Saved to: `{result['output_path']}`
""")
async def handle_generate_ab_roll(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, generator = _setup_generator(arguments, "_ab_roll")
result = generator.generate_ab_roll(
output_path=output_path,
target_duration=arguments["target_duration"],
a_keywords=arguments["a_keywords"],
b_keywords=arguments["b_keywords"],
a_duration=arguments.get("a_duration", "5s"),
b_duration=arguments.get("b_duration", "3s"),
start_with=arguments.get("start_with", "a"),
add_transitions=arguments.get("add_transitions", True),
)
return _text_result(f"""# A/B Roll Edit Generated
## Summary
- **A-Roll Segments**: {result['a_segments']} (from {result['a_clips_available']} available)
- **B-Roll Segments**: {result['b_segments']} (from {result['b_clips_available']} available)
- **Total Clips**: {result['clips_used']}
## Timing
- **Target Duration**: {format_duration(result['target_duration'])}
- **Actual Duration**: {format_duration(result['actual_duration'])}
- **A-Roll Duration**: {result['a_duration_setting']} per segment
- **B-Roll Duration**: {result['b_duration_setting']} per cutaway
## Output
Saved to: `{result['output_path']}`
**Next step**: Import this FCPXML into Final Cut Pro (File > Import > XML)
""")
async def handle_list_templates(arguments: dict) -> Sequence[TextContent]:
templates = list_templates()
lines = ["# Available Timeline Templates\n"]
for tmpl in templates:
lines.append(f"## {tmpl['name']}")
lines.append(f"{tmpl['description']}\n")
lines.append("| Slot | Type | Default Duration | Lane | Required |")
lines.append("|------|------|-----------------|------|----------|")
for s in tmpl['slots']:
lines.append(
f"| {s['name']} | {s['slot_type']} | {s['default_duration']}s "
f"| {s['lane']} | {'Yes' if s['required'] else 'No'} |"
)
lines.append("")
return _text_result("\n".join(lines))
async def handle_apply_template(arguments: dict) -> Sequence[TextContent]:
template_name = arguments["template_name"]
clips_raw = arguments["clips"]
output_path = _validate_output_path(arguments["output_path"], anchor_dir=PROJECTS_DIR)
fps = arguments.get("fps", 24)
# Convert raw clips dict to ClipSpec objects
clips_map = {}
for slot_name, spec_data in clips_raw.items():
if isinstance(spec_data, dict):
clips_map[slot_name] = ClipSpec(
asset_id=spec_data.get("asset_id"),
src=spec_data.get("src"),
name=spec_data.get("name", slot_name),
duration=spec_data.get("duration"),
)
result_path = apply_template(template_name, clips_map, output_path, fps)
return _text_result((
f"Applied template '{template_name}' with {len(clips_map)} clips.\n"
f"Saved to: `{result_path}`"
))
HANDLERS = {
"auto_rough_cut": handle_auto_rough_cut,
"generate_montage": handle_generate_montage,
"generate_ab_roll": handle_generate_ab_roll,
"list_templates": handle_list_templates,
"apply_template": handle_apply_template,
}
+110
View File
@@ -0,0 +1,110 @@
"""Live (macOS) — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
from pathlib import Path
from typing import Sequence
from mcp.types import TextContent, Tool
from server_tools._shared import (
_text_result,
_validate_filepath,
_validate_output_path,
generate_output_path,
)
TOOLS = [
Tool(
name="push_to_fcp",
description="LIVE: send an FCPXML file into the running Final Cut Pro with zero clicks (official Open Document Apple event). Creates/targets a library via import-options. Launches FCP if needed. macOS-only; first use triggers an Automation permission prompt. For true zero-click, pass a library_location ending in .fcpbundle (a new path is auto-created); omitting it makes FCP show a modal library picker.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file or .fcpxmld bundle to import"},
"library_location": {"type": "string", "description": "Target .fcpbundle library path (auto-created if it doesn't exist; the extension is normalized to .fcpbundle). Omit to import into the active library, but note FCP then shows a modal 'Open Library' picker that blocks until answered"},
"suppress_warnings": {"type": "boolean", "description": "Suppress non-fatal import warning dialogs", "default": True},
"copy_assets": {"type": "boolean", "description": "Copy media into the library (true) or link in place (false). Omit for FCP default"},
},
"required": ["filepath"]
}
),
Tool(
name="list_fcp_libraries",
description="LIVE: enumerate the running Final Cut Pro's open libraries, events, and projects via Apple's read-only scripting dictionary. Refuses to launch FCP unless allow_launch is true. macOS-only.",
inputSchema={
"type": "object",
"properties": {
"allow_launch": {"type": "boolean", "description": "Launch FCP if it isn't running", "default": False},
},
}
),
]
async def handle_push_to_fcp(arguments: dict) -> Sequence[TextContent]:
from fcpxml.live import push_to_fcp
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
# Flat files get an options-injected sibling copy (never touch the
# original); the copy path goes through the same write sandbox as
# every other derived output.
import_copy = None
if Path(filepath).suffix.lower() == '.fcpxml':
anchor = str(Path(filepath).resolve().parent)
import_copy = _validate_output_path(
generate_output_path(filepath, "_import"), anchor_dir=anchor
)
result = push_to_fcp(
filepath,
library_location=arguments.get("library_location"),
suppress_warnings=arguments.get("suppress_warnings", True),
copy_assets=arguments.get("copy_assets"),
import_copy_path=import_copy,
)
lines = [
f"Sent to Final Cut Pro: {result['sent']}",
f"FCP {'was launched' if result['launched_fcp'] else 'was already running'} — "
f"import happens in-app (libraries/events are created or merged per import-options).",
]
if arguments.get("library_location"):
lines.append(f"Target library: {arguments['library_location']}")
lines.append(
"Note: Apple offers no programmatic export — to round-trip edits "
"back, use File > Export XML in FCP."
)
return _text_result("\n".join(lines))
async def handle_list_fcp_libraries(arguments: dict) -> Sequence[TextContent]:
from fcpxml.live import list_fcp_libraries
try:
libraries = list_fcp_libraries(
allow_launch=arguments.get("allow_launch", False)
)
except RuntimeError as exc:
return _text_result(str(exc))
if not libraries:
return _text_result("Final Cut Pro is running but reports no open libraries.")
lines = [f"Open libraries in Final Cut Pro ({len(libraries)}):", ""]
for lib in libraries:
lines.append(f"📚 {lib['name']}")
for event in lib["events"]:
lines.append(f" └─ {event['name']}")
for proj in event["projects"]:
lines.append(f" • {proj}")
return _text_result("\n".join(lines))
HANDLERS = {
"push_to_fcp": handle_push_to_fcp,
"list_fcp_libraries": handle_list_fcp_libraries,
}
+458
View File
@@ -0,0 +1,458 @@
"""Beats / markers importados — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
import json
import re
from pathlib import Path
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.media_intel import media_src_to_path
from fcpxml.parser import FCPXMLParser
from fcpxml.writer import FCPXMLModifier
from server_tools._shared import (
_check_json_depth,
_load_or_transcribe,
_markdown_table,
_no_timeline,
_raw_markers_to_batch,
_resolve_io_paths,
_setup_modifier,
_text_result,
_validate_filepath,
format_duration,
parse_srt,
parse_transcript_timestamps,
parse_vtt,
)
TOOLS = [
Tool(
name="import_beat_markers",
description="Import beat markers from external audio analysis (JSON format)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"beats_path": {"type": "string", "description": "Path to beats JSON file"},
"marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "standard"},
"beat_filter": {"type": "string", "enum": ["all", "downbeat", "measure"], "default": "all", "description": "Which beats to import"},
"output_path": {"type": "string", "description": "Output path (default: adds _beats suffix)"}
},
"required": ["filepath", "beats_path"]
}
),
Tool(
name="snap_to_beats",
description="Align cuts to nearest beat markers for music-synced edits",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file with beat markers"},
"max_shift_frames": {"type": "integer", "default": 6, "description": "Maximum frames to shift a cut"},
"prefer": {"type": "string", "enum": ["earlier", "later", "nearest"], "default": "nearest", "description": "Which beat to prefer when equidistant"},
"output_path": {"type": "string", "description": "Output path (default: adds _synced suffix)"}
},
"required": ["filepath"]
}
),
Tool(
name="import_srt_markers",
description="Import SRT or VTT subtitles as chapter markers on the timeline",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"srt_path": {"type": "string", "description": "Path to SRT or VTT subtitle file"},
"mode": {"type": "string", "enum": ["all", "first_per_minute", "scene_changes"], "default": "first_per_minute", "description": "How to create markers: every subtitle, first per minute, or on text changes"},
"marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "chapter"},
"max_label_length": {"type": "integer", "default": 50, "description": "Truncate marker labels to this length"},
"output_path": {"type": "string", "description": "Output path (default: adds _subtitled suffix)"}
},
"required": ["filepath", "srt_path"]
}
),
Tool(
name="import_transcript_markers",
description="Import timestamped transcript (YouTube chapter format) as markers. Supports '0:00 Title' and 'HH:MM:SS Title' formats",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"transcript": {"type": "string", "description": "Timestamped text (one per line: '0:00 Introduction')"},
"transcript_path": {"type": "string", "description": "Path to text file with timestamps (alternative to inline transcript)"},
"marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "chapter"},
"output_path": {"type": "string", "description": "Output path (default: adds _chapters suffix)"}
},
"required": ["filepath"]
}
),
Tool(
name="transcript_markers",
description="Add a marker at the start of every transcribed segment (sentence-level), using each media file's local Whisper transcript. Maps each segment's source-media timestamp to its correct timeline position per clip, so it stays accurate across multiple clips/trims — unlike import_transcript_markers (plain timestamp text) or import_srt_markers (a caption track already synced to the whole export). Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _transcript_markers copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"clip_name": {"type": "string", "description": "Only mark the clip with this name"},
"marker_type": {"type": "string", "default": "chapter", "description": "Marker type: standard, chapter, todo, completed"},
"max_label_length": {"type": "integer", "default": 50, "description": "Truncate marker labels to this many characters (0 = no truncation)"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"output_path": {"type": "string", "description": "Output path (default: adds _transcript_markers suffix)"},
},
"required": ["filepath"]
}
),
]
async def handle_import_beat_markers(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_beats")
beats_path = _validate_filepath(arguments["beats_path"], ('.json',))
with open(beats_path, 'r') as f:
beats_data = json.load(f)
_check_json_depth(beats_data)
beat_times = []
if isinstance(beats_data, list):
beat_times = beats_data
elif isinstance(beats_data, dict):
beat_times = beats_data.get('beats', beats_data.get('times', beats_data.get('markers', [])))
beat_filter = arguments.get("beat_filter", "all")
if beat_filter == "downbeat" and isinstance(beats_data, dict):
beat_times = beats_data.get('downbeats', beat_times[::4])
elif beat_filter == "measure" and isinstance(beats_data, dict):
beat_times = beats_data.get('measures', beat_times[::4])
markers = []
marker_type = arguments.get("marker_type", "standard")
for i, beat_time in enumerate(beat_times):
if isinstance(beat_time, (int, float)):
markers.append({
'timecode': f"{beat_time}s",
'name': f"Beat {i+1}",
'marker_type': marker_type.upper(),
})
elif isinstance(beat_time, dict):
markers.append({
'timecode': f"{beat_time.get('time', beat_time.get('position', 0))}s",
'name': beat_time.get('label', f"Beat {i+1}"),
'marker_type': marker_type.upper(),
})
modifier = FCPXMLModifier(filepath)
# Songs routinely run longer than the edit — beats past the timeline's
# end are skipped (add_marker_at_timeline would raise on them).
timeline_end = modifier._timeline_duration().to_seconds()
in_range = [m for m in markers if float(m['timecode'].rstrip('s')) < timeline_end]
skipped_count = len(markers) - len(in_range)
added = modifier.batch_add_markers(markers=in_range)
modifier.save(output_path)
skipped_note = (
f"- **Skipped**: {skipped_count} beat(s) beyond the timeline end "
f"({format_duration(timeline_end)})\n" if skipped_count else ""
)
return _text_result(f"""# Beat Markers Imported
## Summary
- **Beats Found**: {len(beat_times)}
- **Markers Added**: {len(added)}
{skipped_note}- **Filter**: {beat_filter}
- **Marker Type**: {marker_type}
## Output
Saved to: `{output_path}`
*Use `snap_to_beats` to align your cuts to these markers.*
""")
async def handle_snap_to_beats(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_synced")
max_shift = arguments.get("max_shift_frames", 6)
prefer = arguments.get("prefer", "nearest")
parser = FCPXMLParser()
project = parser.parse_file(filepath)
if not project.timelines:
return _no_timeline()
tl = project.primary_timeline
fps = tl.frame_rate
markers = list(tl.markers)
for clip in tl.clips:
markers.extend(clip.markers)
if not markers:
return _text_result("No markers found. Use `import_beat_markers` first.")
marker_times = sorted([m.start.seconds for m in markers])
modifier = FCPXMLModifier(filepath)
spine = modifier._get_spine()
adjusted_count = 0
total_shift = 0
clips_list = [c for c in spine if c.tag in ('clip', 'asset-clip', 'video', 'ref-clip')]
for i, clip in enumerate(clips_list[1:], 1):
cut_offset = modifier._parse_time(clip.get('offset', '0s'))
cut_seconds = cut_offset.to_seconds()
best_marker = None
best_distance = float('inf')
for marker_time in marker_times:
distance = abs(marker_time - cut_seconds)
distance_frames = distance * fps
if distance_frames <= max_shift:
if prefer == "earlier" and marker_time <= cut_seconds:
if distance < best_distance:
best_distance = distance
best_marker = marker_time
elif prefer == "later" and marker_time >= cut_seconds:
if distance < best_distance:
best_distance = distance
best_marker = marker_time
elif prefer == "nearest":
if distance < best_distance:
best_distance = distance
best_marker = marker_time
if best_marker is not None and best_distance > 0.001:
shift = best_marker - cut_seconds
shift_frames = int(shift * fps)
prev_clip = clips_list[i - 1]
prev_dur = modifier._parse_time(prev_clip.get('duration', '0s'))
new_prev_dur = prev_dur + modifier._parse_time(f"{shift}s")
prev_clip.set('duration', new_prev_dur.to_fcpxml())
new_offset = modifier._parse_time(f"{best_marker}s")
clip.set('offset', new_offset.to_fcpxml())
adjusted_count += 1
total_shift += abs(shift_frames)
modifier.save(output_path)
avg_shift = total_shift / adjusted_count if adjusted_count > 0 else 0
return _text_result(f"""# Cuts Snapped to Beats
## Summary
- **Cuts Adjusted**: {adjusted_count}
- **Max Shift Allowed**: {max_shift} frames
- **Preference**: {prefer}
- **Average Shift**: {avg_shift:.1f} frames
## Output
Saved to: `{output_path}`
Your edits are now synced to the beat!
""")
async def handle_import_srt_markers(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_subtitled")
srt_path = _validate_filepath(arguments["srt_path"], ('.srt', '.vtt'))
mode = arguments.get("mode", "first_per_minute")
marker_type = arguments.get("marker_type", "chapter")
max_label = arguments.get("max_label_length", 50)
text = Path(srt_path).read_text(encoding='utf-8')
# Detect format and parse
if srt_path.endswith('.vtt') or text.strip().startswith('WEBVTT'):
raw_markers = parse_vtt(text)
fmt_name = "WebVTT"
else:
raw_markers = parse_srt(text)
fmt_name = "SRT"
if not raw_markers:
return _text_result(f"No subtitles found in {srt_path}")
# Apply mode filtering
filtered = []
if mode == "all":
filtered = raw_markers
elif mode == "first_per_minute":
seen_minutes = set()
for m in raw_markers:
minute = int(m['seconds'] // 60)
if minute not in seen_minutes:
seen_minutes.add(minute)
filtered.append(m)
elif mode == "scene_changes":
# Group by similar text, take first occurrence of each unique line
seen_texts = set()
for m in raw_markers:
# Normalize: lowercase, strip punctuation
normalized = re.sub(r'[^\w\s]', '', m['text'].lower()).strip()
words = normalized.split()[:3] # First 3 words as key
key = ' '.join(words)
if key and key not in seen_texts:
seen_texts.add(key)
filtered.append(m)
markers = _raw_markers_to_batch(filtered, marker_type, max_label=max_label)
modifier = FCPXMLModifier(filepath)
added = modifier.batch_add_markers(markers=markers)
modifier.save(output_path)
return _text_result(f"""# Subtitle Markers Imported
## Summary
- **Format**: {fmt_name}
- **Subtitles Parsed**: {len(raw_markers)}
- **Mode**: {mode}
- **Markers Added**: {len(added)}
- **Marker Type**: {marker_type}
## Output
Saved to: `{output_path}`
""")
async def handle_import_transcript_markers(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_chapters")
marker_type = arguments.get("marker_type", "chapter")
# Get transcript text from inline or file
transcript = arguments.get("transcript")
transcript_path = arguments.get("transcript_path")
if not transcript and not transcript_path:
return _text_result("Provide either 'transcript' (inline text) or 'transcript_path' (path to file)")
if transcript_path:
# .txt only: the parser below understands "0:00 Title" lines, not real
# SRT/VTT cue syntax — that's import_srt_markers (parse_srt/parse_vtt).
transcript_path = _validate_filepath(transcript_path, ('.txt',))
transcript = Path(transcript_path).read_text(encoding='utf-8')
raw_markers = parse_transcript_timestamps(transcript or "")
if not raw_markers:
return _text_result("No timestamps found. Expected format: '0:00 Title' or 'HH:MM:SS Title', one per line.")
markers = _raw_markers_to_batch(raw_markers, marker_type)
modifier = FCPXMLModifier(filepath)
added = modifier.batch_add_markers(markers=markers)
modifier.save(output_path)
return _text_result(f"""# Transcript Markers Imported
## Summary
- **Timestamps Found**: {len(raw_markers)}
- **Markers Added**: {len(added)}
- **Marker Type**: {marker_type}
## Markers
""" + "\n".join(f"- `{m['timecode']}` {m['name']}" for m in markers) + f"""
## Output
Saved to: `{output_path}`
""")
async def handle_transcript_markers(arguments: dict) -> Sequence[TextContent]:
"""Add a marker at the start of each transcribed segment, using each
media's cached (or freshly transcribed) local Whisper transcript.
Unlike ``import_transcript_markers`` (plain "0:00 Title" text) or
``import_srt_markers`` (a caption track already synced to the whole
exported video), this maps each segment's SOURCE-media timestamp to its
TIMELINE position per spine clip — the same source->timeline mapping
``detect_media_silence`` uses — so it stays correct across multiple
clips built from different (and differently-trimmed) source files.
"""
marker_type = arguments.get("marker_type", "chapter")
max_label = int(arguments.get("max_label_length", 50))
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
clip_filter = arguments.get("clip_name")
filepath, output_path, modifier = _setup_modifier(arguments, "_transcript_markers")
added: list[tuple[str, float, str]] = []
skipped: list[tuple[str, str]] = []
spine_clips = [el for _, el in modifier._iter_spine_clips()]
for el in spine_clips:
name = el.get("name", "")
if clip_filter and name != clip_filter:
continue
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
media_path = media_src_to_path(src)
if not media_path or not Path(media_path).is_file():
skipped.append((name, "media file missing"))
continue
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
if data is None:
skipped.append((name, reason))
continue
clip_source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
clip_offset = modifier._parse_time(el.get("offset", "0s")).to_seconds()
window_end = clip_source_start + clip_duration
for seg in data.get("segments", []):
seg_start = float(seg.get("start", 0.0))
if seg_start < clip_source_start or seg_start >= window_end:
continue
label = seg.get("text", "").strip()
if not label:
continue
if max_label and len(label) > max_label:
label = label[:max_label]
timeline_seconds = clip_offset + (seg_start - clip_source_start)
modifier.add_marker_at_timeline(
timecode=f"{timeline_seconds}s", name=label, marker_type=marker_type,
)
added.append((name, seg_start, label))
if not added:
text = "# Transcript Markers\n\nNo segments to mark — file unchanged (nothing saved)."
if skipped:
text += "\n\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
return _text_result(text)
modifier.save(output_path)
result = "# Transcript Markers Imported (local Whisper)\n\n## Summary\n"
result += f"- **Markers Added**: {len(added)}\n- **Marker Type**: {marker_type}\n\n"
result += _markdown_table(
["Clip", "Start", "Label"], [[n, f"{s:.2f}s", label] for n, s, label in added]
)
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
return _text_result(result)
HANDLERS = {
"import_beat_markers": handle_import_beat_markers,
"snap_to_beats": handle_snap_to_beats,
"import_srt_markers": handle_import_srt_markers,
"import_transcript_markers": handle_import_transcript_markers,
"transcript_markers": handle_transcript_markers,
}
+696
View File
@@ -0,0 +1,696 @@
"""QC e detecção — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.media_intel import (
detect_beats,
detect_silence,
map_silence_to_timeline,
media_src_to_path,
)
from fcpxml.model_manager import load_silence_config
from fcpxml.models import FlashFrameSeverity, TimeValue
from fcpxml.writer import FCPXMLModifier
from server_tools._shared import (
AUDIO_MEDIA_EXTENSIONS,
MAX_MEDIA_FILE_SIZE,
_detect_duplicate_groups,
_detect_flash_frames,
_detect_gaps,
_fmt_suggestions,
_format_clip_table,
_markdown_table,
_require_timeline,
_setup_modifier,
_text_result,
_validate_filepath,
_validate_output_path,
format_duration,
format_timecode,
)
TOOLS = [
Tool(
name="find_short_cuts",
description="Find clips shorter than threshold (flash frame detection)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"threshold_seconds": {"type": "number", "default": 0.5}
},
"required": ["filepath"]
}
),
Tool(
name="find_long_clips",
description="Find clips longer than threshold",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"threshold_seconds": {"type": "number", "default": 10.0}
},
"required": ["filepath"]
}
),
Tool(
name="analyze_pacing",
description="Analyze edit pacing with suggestions for improvements",
inputSchema={
"type": "object",
"properties": {"filepath": {"type": "string"}},
"required": ["filepath"]
}
),
Tool(
name="detect_flash_frames",
description="Find ultra-short clips (flash frames) that are likely errors, with severity categorization",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"critical_threshold_frames": {"type": "integer", "default": 2, "description": "Frames below this = critical (default: 2)"},
"warning_threshold_frames": {"type": "integer", "default": 6, "description": "Frames below this = warning (default: 6)"}
},
"required": ["filepath"]
}
),
Tool(
name="detect_duplicates",
description="Find clips using the same source media (potential duplicates)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"mode": {"type": "string", "enum": ["same_source", "overlapping_ranges", "identical"], "default": "same_source", "description": "Detection mode"}
},
"required": ["filepath"]
}
),
Tool(
name="detect_gaps",
description="Find unintentional gaps in the timeline",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"min_gap_frames": {"type": "integer", "default": 1, "description": "Minimum gap size to detect (default: 1 frame)"}
},
"required": ["filepath"]
}
),
Tool(
name="validate_timeline",
description="Comprehensive timeline health check for flash frames, gaps, duplicates, and issues",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"checks": {"type": "array", "items": {"type": "string", "enum": ["all", "flash_frames", "gaps", "duplicates", "offsets"]}, "default": ["all"], "description": "Which checks to run"}
},
"required": ["filepath"]
}
),
Tool(
name="detect_media_silence",
description="Detect REAL silence by analyzing each clip's source audio with ffmpeg silencedetect, mapped into timeline time. Unlike detect_silence_candidates (XML-only heuristics), this reads the actual media files referenced by the timeline. Requires ffmpeg; clips whose media is missing or unreadable are reported, not failed.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"noise_db": {"type": "number", "description": "Silence threshold in dBFS, -120 to 0. Falls back to the saved silence settings (default -30)"},
"min_silence": {"type": "number", "description": "Minimum silence duration in seconds to report. Falls back to the saved silence settings (default 0.5)"},
"clip_name": {"type": "string", "description": "Only analyze the clip with this name"},
},
"required": ["filepath"]
}
),
Tool(
name="detect_beats",
description="Detect musical beats and tempo in an audio/video file (librosa beat tracker). Writes a beats JSON next to the media file that plugs directly into import_beat_markers + snap_to_beats for beat-synced editing. Requires the optional [intelligence] extra (librosa); degrades to an install hint without it.",
inputSchema={
"type": "object",
"properties": {
"media_path": {"type": "string", "description": "Path to audio/video file (.wav, .mp3, .m4a, .aac, .aif, .flac, .mov, .mp4)"},
},
"required": ["media_path"]
}
),
Tool(
name="remove_media_silence",
description="Detect REAL silence in each clip's source audio (ffmpeg) and CUT it out of the timeline with ripple. Clips are split around silence; the silent middles are removed and everything after shifts earlier. Non-destructive: writes a _silence_removed copy. Preview with detect_media_silence first.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"noise_db": {"type": "number", "description": "Silence threshold in dBFS, -120 to 0. Falls back to the saved silence settings (default -30)"},
"min_silence": {"type": "number", "description": "Minimum silence duration in seconds to cut. Falls back to the saved silence settings (default 0.5)"},
"padding": {"type": "number", "description": "Seconds of silence to keep on each side of a cut so edits breathe (max 5). Falls back to the saved silence settings (default 0.05)"},
"clip_name": {"type": "string", "description": "Only cut silence in the clip with this name"},
"output_path": {"type": "string", "description": "Output path (default: adds _silence_removed suffix)"},
},
"required": ["filepath"]
}
),
Tool(
name="detect_silence_candidates",
description="Detect potential silence/dead air using timeline heuristics (gaps, ultra-short clips, name patterns, duration anomalies)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"min_gap_seconds": {"type": "number", "default": 0.5, "description": "Minimum gap duration to flag"},
"patterns": {"type": "array", "items": {"type": "string"}, "description": "Name patterns to match (default: gap, silence, room tone)"},
},
"required": ["filepath"]
}
),
Tool(
name="remove_silence_candidates",
description="Remove or mark detected silence candidates from timeline",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"mode": {"type": "string", "enum": ["delete", "mark"], "default": "mark", "description": "delete=remove clips/gaps, mark=add red markers"},
"min_gap_seconds": {"type": "number", "default": 0.5},
"min_confidence": {"type": "number", "default": 0.7, "description": "Only act on candidates above this confidence"},
"output_path": {"type": "string", "description": "Output path (default: adds _silence_cleaned suffix)"}
},
"required": ["filepath"]
}
),
]
async def handle_find_short_cuts(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
threshold = arguments.get("threshold_seconds", 0.5)
short = tl.get_clips_shorter_than(threshold)
if not short:
return _text_result(f"No clips shorter than {threshold}s")
return _text_result(_format_clip_table(
short, f"# Short Clips (< {threshold}s) - {len(short)} found",
))
async def handle_find_long_clips(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
threshold = arguments.get("threshold_seconds", 10.0)
long = tl.get_clips_longer_than(threshold)
if not long:
return _text_result(f"No clips longer than {threshold}s")
return _text_result(_format_clip_table(
long, f"# Long Clips (> {threshold}s) - {len(long)} found",
))
async def handle_analyze_pacing(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
if not tl.clips:
return _text_result("No clips to analyze")
durs = [c.duration_seconds for c in tl.clips]
avg = sum(durs) / len(durs)
q_len = len(durs) // 4 or 1
segments = [durs[i:i+q_len] for i in range(0, len(durs), q_len)][:4]
seg_avgs = [sum(s)/len(s) if s else 0 for s in segments]
suggestions = []
flash = [c for c in tl.clips if c.duration_seconds < 0.2]
if flash:
suggestions.append(f" {len(flash)} potential flash frames (< 0.2s)")
long = [c for c in tl.clips if c.duration_seconds > 30]
if long:
suggestions.append(f" {len(long)} long takes (> 30s) - consider trimming")
if len(seg_avgs) >= 4 and seg_avgs[3] < seg_avgs[0] * 0.7:
suggestions.append(" Pacing accelerates toward end - good for building energy")
elif len(seg_avgs) >= 4 and seg_avgs[3] > seg_avgs[0] * 1.3:
suggestions.append(" Pacing slows toward end - consider tightening")
return _text_result(f"""# Pacing Analysis: {tl.name}
## Overall
- **Avg Cut**: {format_duration(avg)}
- **Cuts/Min**: {tl.cuts_per_minute:.1f}
## By Section
| Q1 | Q2 | Q3 | Q4 |
|----|----|----|----|
| {format_duration(seg_avgs[0]) if len(seg_avgs) > 0 else 'N/A'} | {format_duration(seg_avgs[1]) if len(seg_avgs) > 1 else 'N/A'} | {format_duration(seg_avgs[2]) if len(seg_avgs) > 2 else 'N/A'} | {format_duration(seg_avgs[3]) if len(seg_avgs) > 3 else 'N/A'} |
## Suggestions
{_fmt_suggestions(suggestions)}
""")
async def handle_detect_flash_frames(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
critical_threshold = arguments.get("critical_threshold_frames", 2)
warning_threshold = arguments.get("warning_threshold_frames", 6)
flash_frames = _detect_flash_frames(
tl, critical_threshold=critical_threshold, warning_threshold=warning_threshold,
)
if not flash_frames:
return _text_result(f"No flash frames detected (threshold: {warning_threshold} frames)")
critical = [f for f in flash_frames if f.severity == FlashFrameSeverity.CRITICAL]
warnings = [f for f in flash_frames if f.severity == FlashFrameSeverity.WARNING]
result = f"""# Flash Frame Detection
## Summary
- **Critical** (< {critical_threshold} frames): {len(critical)} found
- **Warning** (< {warning_threshold} frames): {len(warnings)} found
- **Total**: {len(flash_frames)} flash frames
## Critical Flash Frames
"""
flash_headers = ["Clip", "Timecode", "Frames", "Duration"]
if critical:
result += _markdown_table(flash_headers, [
[f.clip_name, format_timecode(f.start), f"{f.duration_frames}f", format_duration(f.duration_seconds)]
for f in critical
]) + "\n"
else:
result += "_None_\n"
result += "\n## Warning Flash Frames\n"
if warnings:
result += _markdown_table(flash_headers, [
[f.clip_name, format_timecode(f.start), f"{f.duration_frames}f", format_duration(f.duration_seconds)]
for f in warnings
]) + "\n"
else:
result += "_None_\n"
result += "\n*Use `fix_flash_frames` to automatically resolve these issues.*"
return _text_result(result)
async def handle_detect_duplicates(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
mode = arguments.get("mode", "same_source")
duplicates = _detect_duplicate_groups(tl, mode=mode)
if not duplicates:
return _text_result(f"No duplicate clips found (mode: {mode})")
result = f"""# Duplicate Clip Detection
## Summary
- **Mode**: {mode}
- **Duplicate Groups**: {len(duplicates)}
- **Total Duplicate Clips**: {sum(g.count for g in duplicates)}
## Duplicate Groups
"""
for group in duplicates:
result += f"\n### {group.source_name} ({group.count} uses)\n"
result += "| Clip Name | Timeline Position | Duration |\n|-----------|-------------------|----------|\n"
for c in group.clips:
result += f"| {c['name']} | {c['timecode']} | {format_duration(c['duration'])} |\n"
return _text_result(result)
async def handle_detect_gaps(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
min_gap_frames = arguments.get("min_gap_frames", 1)
gaps = _detect_gaps(tl, min_gap_frames=min_gap_frames)
if not gaps:
return _text_result(f"No gaps detected (minimum: {min_gap_frames} frame(s))")
result = f"""# Gap Detection
## Summary
- **Gaps Found**: {len(gaps)}
- **Total Gap Duration**: {format_duration(sum(g.duration_seconds for g in gaps))}
- **Minimum Detection**: {min_gap_frames} frame(s)
## Gaps
"""
result += _markdown_table(
["Position", "Duration", "Between"],
[[gap.timecode, f"{gap.duration_frames}f ({format_duration(gap.duration_seconds)})",
f"{gap.previous_clip} -> {gap.next_clip}"] for gap in gaps],
) + "\n"
result += "\n*Use `fill_gaps` to automatically close these gaps.*"
return _text_result(result)
async def handle_validate_timeline(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
checks = arguments.get("checks", ["all"])
run_all = "all" in checks
issues: list[str] = []
flash_count = 0
gap_count = 0
duplicate_count = 0
if run_all or "flash_frames" in checks:
flashes = _detect_flash_frames(tl)
flash_count = len(flashes)
for f in flashes:
severity = "error" if f.severity == FlashFrameSeverity.CRITICAL else "warning"
issues.append(
f"- [{severity.upper()}] Flash frame: {f.clip_name} "
f"({f.duration_frames}f) at {format_timecode(f.start)}"
)
if run_all or "gaps" in checks:
detected_gaps = _detect_gaps(tl)
gap_count = len(detected_gaps)
for g in detected_gaps:
issues.append(f"- [WARNING] Gap: {g.duration_frames}f at {g.timecode}")
if run_all or "duplicates" in checks:
dup_groups = _detect_duplicate_groups(tl)
for group in dup_groups:
duplicate_count += group.count
issues.append(
f"- [INFO] Duplicate source: {group.source_name} ({group.count} uses)"
)
error_weight = 10
warning_weight = 3
info_weight = 1
errors = len([i for i in issues if "[ERROR]" in i])
warnings = len([i for i in issues if "[WARNING]" in i])
infos = len([i for i in issues if "[INFO]" in i])
penalty = (errors * error_weight) + (warnings * warning_weight) + (infos * info_weight)
health_score = max(0, 100 - penalty)
result = f"""# Timeline Validation: {tl.name}
## Health Score: {health_score}%
## Summary
| Check | Count | Status |
|-------|-------|--------|
| Flash Frames | {flash_count} | {'PASS' if flash_count == 0 else 'FAIL'} |
| Gaps | {gap_count} | {'PASS' if gap_count == 0 else 'WARN'} |
| Duplicate Sources | {duplicate_count} | {'PASS' if duplicate_count == 0 else 'INFO'} |
## Issues ({len(issues)})
"""
if issues:
result += "\n".join(issues[:20])
if len(issues) > 20:
result += f"\n... and {len(issues) - 20} more issues"
else:
result += "_No issues found!_"
result += "\n\n*Use `fix_flash_frames` and `fill_gaps` to automatically resolve issues.*"
return _text_result(result)
async def handle_detect_media_silence(arguments: dict) -> Sequence[TextContent]:
# Unpassed thresholds come from the persisted silence settings (the app's
# own slider), not a hardcoded constant, so detection previews exactly
# what removal would cut.
saved = load_silence_config()
noise_db = float(arguments.get("noise_db", saved["noise_db"]))
min_silence = float(arguments.get("min_silence", saved["min_silence"]))
# Same bounds detect_silence() enforces — validated here so a bad request
# fails before any media file is opened.
if not (-120.0 <= noise_db <= 0.0):
raise ValueError(f"noise_db must be between -120 and 0 dB, got {noise_db}")
if not (0 < min_silence <= 3600):
raise ValueError(f"min_silence must be between 0 and 3600 seconds, got {min_silence}")
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
modifier = FCPXMLModifier(filepath)
clip_filter = arguments.get("clip_name")
max_media_probes = 100
findings: list[tuple[str, float, float]] = []
skipped: list[tuple[str, str]] = []
probe_cache: dict[str, list | None] = {}
for el in [el for _, el in modifier._iter_spine_clips()]:
name = el.get("name", "")
if clip_filter and name != clip_filter:
continue
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
media_path = media_src_to_path(src)
if not media_path or not Path(media_path).is_file():
skipped.append((name, "media file missing"))
continue
if media_path not in probe_cache:
if len(probe_cache) >= max_media_probes:
skipped.append((name, f"probe cap reached ({max_media_probes} media files)"))
continue
probe_cache[media_path] = detect_silence(
media_path, noise_db=noise_db, min_duration=min_silence
)
silences = probe_cache[media_path]
if silences is None:
skipped.append((name, "unanalyzable (ffmpeg missing or media unreadable)"))
continue
source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
timeline_offset = modifier._parse_time(el.get("offset", "0s")).to_seconds()
mapped = map_silence_to_timeline(
silences, source_start, clip_duration, timeline_offset
)
findings.extend((name, start, end) for start, end in mapped)
total_silence = sum(end - start for _, start, end in findings)
result = f"""# Media Silence Detection (real audio analysis)
## Summary
- **Threshold**: {noise_db} dB for >= {min_silence}s
- **Media Files Probed**: {len(probe_cache)}
- **Silence Spans Found**: {len(findings)} ({format_duration(total_silence)} total)
"""
if findings:
result += "\n## Silence Spans (timeline time)\n"
result += _markdown_table(
["Clip", "Start", "End", "Duration"],
[[name, f"{start:.2f}s", f"{end:.2f}s", f"{end - start:.2f}s"]
for name, start, end in findings],
) + "\n"
result += "\n*To remove: `split_clip` at each boundary, then `delete_clips` with ripple.*"
if skipped:
result += "\n## Skipped Clips\n"
result += _markdown_table(
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
) + "\n"
if not findings and not skipped:
result += "\nNo silence detected in any clip's source audio."
return _text_result(result)
async def handle_detect_beats(arguments: dict) -> Sequence[TextContent]:
media_path = _validate_filepath(
arguments["media_path"], AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
)
result = detect_beats(media_path)
if result is None:
return _text_result(
"Beat detection unavailable — librosa is not installed or the file "
"could not be analyzed.\n\nInstall the optional media-intelligence "
"extra:\n\n pip install 'fcp-mcp-server[intelligence]'"
)
bpm, beats = result["bpm"], result["beats"]
beats_data = {
"source": str(Path(media_path).name),
"bpm": round(bpm, 2),
"beats": [round(b, 4) for b in beats],
"downbeats": [round(b, 4) for b in beats[::4]],
}
json_path = _validate_output_path(
str(Path(media_path).with_name(Path(media_path).stem + "_beats.json")),
anchor_dir=str(Path(media_path).parent),
)
with open(json_path, "w") as f:
json.dump(beats_data, f, indent=2)
preview = beats[:16]
result_text = f"""# Beat Detection
## Summary
- **Source**: {Path(media_path).name}
- **Estimated Tempo**: {bpm:.1f} BPM
- **Beats Detected**: {len(beats)} ({format_duration(beats[-1]) if beats else '0s'} span)
- **Beats JSON**: {json_path}
## First Beats
"""
result_text += _markdown_table(
["#", "Time"],
[[str(i + 1), f"{b:.3f}s"] for i, b in enumerate(preview)],
) + "\n"
result_text += (
f"\n*Next: `import_beat_markers` with beats_path=\"{json_path}\" to place "
"markers, then `snap_to_beats` to align your cuts.*"
)
return _text_result(result_text)
async def handle_remove_media_silence(arguments: dict) -> Sequence[TextContent]:
saved = load_silence_config()
noise_db = float(arguments.get("noise_db", saved["noise_db"]))
min_silence = float(arguments.get("min_silence", saved["min_silence"]))
padding = float(arguments.get("padding", saved["padding"]))
if not (-120.0 <= noise_db <= 0.0):
raise ValueError(f"noise_db must be between -120 and 0 dB, got {noise_db}")
if not (0 < min_silence <= 3600):
raise ValueError(f"min_silence must be between 0 and 3600 seconds, got {min_silence}")
if not (0 <= padding <= 5):
raise ValueError(f"padding must be between 0 and 5 seconds, got {padding}")
filepath, output_path, modifier = _setup_modifier(arguments, "_silence_removed")
clip_filter = arguments.get("clip_name")
to_frame_timevalue = modifier.snap_seconds_to_frame
max_media_probes = 100
cuts_made: list[tuple[str, int, float]] = []
skipped: list[tuple[str, str]] = []
probe_cache: dict[str, list | None] = {}
spine_clips = [el for _, el in modifier._iter_spine_clips()]
for el in spine_clips:
name = el.get("name", "")
if clip_filter and name != clip_filter:
continue
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
media_path = media_src_to_path(src)
if not media_path or not Path(media_path).is_file():
skipped.append((name, "media file missing"))
continue
if media_path not in probe_cache:
if len(probe_cache) >= max_media_probes:
skipped.append((name, f"probe cap reached ({max_media_probes} media files)"))
continue
probe_cache[media_path] = detect_silence(
media_path, noise_db=noise_db, min_duration=min_silence
)
silences = probe_cache[media_path]
if silences is None:
skipped.append((name, "unanalyzable (ffmpeg missing or media unreadable)"))
continue
clip_source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
cut_ranges = []
for sil_start, sil_end in silences:
# Source time -> clip-relative, padded so cuts breathe.
cut_start = max(sil_start, clip_source_start) - clip_source_start + padding
cut_end = min(sil_end, clip_source_start + clip_duration) - clip_source_start - padding
if cut_end > cut_start:
cut_ranges.append((to_frame_timevalue(cut_start), to_frame_timevalue(cut_end)))
if not cut_ranges:
continue
removed = modifier.cut_clip_ranges(el, cut_ranges)
if removed > TimeValue.zero():
cuts_made.append((name, len(cut_ranges), removed.to_seconds()))
if not cuts_made:
text = "# Media Silence Removal\n\nNo silence found to remove — file unchanged (nothing saved)."
if skipped:
text += "\n\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
)
return _text_result(text)
modifier.remove_trailing_gaps()
modifier.save(output_path)
total_removed = sum(seconds for _, _, seconds in cuts_made)
result = f"""# Media Silence Removal (real audio analysis)
## Summary
- **Threshold**: {noise_db} dB for >= {min_silence}s, padding {padding}s
- **Clips Cut**: {len(cuts_made)}
- **Total Removed**: {format_duration(total_removed)}
## Cuts
"""
result += _markdown_table(
["Clip", "Silence Spans Cut", "Removed"],
[[name, str(count), f"{seconds:.2f}s"] for name, count, seconds in cuts_made],
) + "\n"
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
) + "\n"
result += f"\nSaved to: {output_path}\n\n*Preview first next time with `detect_media_silence`. Original file untouched.*"
return _text_result(result)
async def handle_detect_silence_candidates(arguments: dict) -> Sequence[TextContent]:
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
modifier = FCPXMLModifier(filepath)
candidates = modifier.detect_silence_candidates(
min_gap_seconds=arguments.get("min_gap_seconds", 0.5),
patterns=arguments.get("patterns"),
)
if not candidates:
return _text_result("No silence candidates detected.")
result = f"# Silence Candidates Detected\n\n**Found**: {len(candidates)}\n\n"
result += "| # | Timecode | Duration | Reason | Confidence | Clip |\n"
result += "|---|----------|----------|--------|------------|------|\n"
for i, c in enumerate(candidates, 1):
result += (
f"| {i} | {c['start_timecode']} | {format_duration(c['duration_seconds'])} | "
f"{c['reason']} | {c['confidence']:.0%} | {c.get('clip_name') or '-'} |\n"
)
result += (
"\n**Note**: Detection uses timeline heuristics (gaps, ultra-short clips, name patterns). "
"Review candidates before removing — some may be intentional."
)
return _text_result(result)
async def handle_remove_silence_candidates(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments, "_silence_cleaned")
actions = modifier.remove_silence_candidates(
mode=arguments.get("mode", "mark"),
min_gap_seconds=arguments.get("min_gap_seconds", 0.5),
min_confidence=arguments.get("min_confidence", 0.7),
)
modifier.save(output_path)
if not actions:
return _text_result("No silence candidates met the confidence threshold.")
mode = arguments.get("mode", "mark")
result = f"# Silence Candidates {'Marked' if mode == 'mark' else 'Removed'}\n\n"
result += f"**Actions taken**: {len(actions)}\n\n"
for a in actions:
result += f"- **{a['action']}** {a.get('clip_name', 'gap')} ({a['reason']})\n"
result += f"\nSaved to: `{output_path}`"
return _text_result(result)
HANDLERS = {
"find_short_cuts": handle_find_short_cuts,
"find_long_clips": handle_find_long_clips,
"analyze_pacing": handle_analyze_pacing,
"detect_flash_frames": handle_detect_flash_frames,
"detect_duplicates": handle_detect_duplicates,
"detect_gaps": handle_detect_gaps,
"validate_timeline": handle_validate_timeline,
"detect_media_silence": handle_detect_media_silence,
"detect_beats": handle_detect_beats,
"remove_media_silence": handle_remove_media_silence,
"detect_silence_candidates": handle_detect_silence_candidates,
"remove_silence_candidates": handle_remove_silence_candidates,
}
+133
View File
@@ -0,0 +1,133 @@
"""Roles — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
from typing import Sequence
from mcp.types import TextContent, Tool
from server_tools._shared import (
_require_timeline,
_setup_modifier,
_text_result,
format_duration,
)
TOOLS = [
Tool(
name="assign_role",
description="Set the audio or video role on a clip (dialogue, music, effects, titles, etc.)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"clip_id": {"type": "string", "description": "Clip name or ID"},
"audio_role": {"type": "string", "description": "Audio role (e.g., dialogue, music, effects)"},
"video_role": {"type": "string", "description": "Video role (e.g., video, titles)"},
"output_path": {"type": "string", "description": "Output path (default: adds _modified suffix)"}
},
"required": ["filepath", "clip_id"]
}
),
Tool(
name="filter_by_role",
description="List all clips matching a specific audio or video role",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"role": {"type": "string", "description": "Role name to filter by"},
"role_type": {"type": "string", "enum": ["audio", "video", "any"], "default": "any", "description": "Which role type to search"},
},
"required": ["filepath", "role"]
}
),
Tool(
name="export_role_stems",
description="Export clip list grouped by role for audio mixing stem planning",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
},
"required": ["filepath"]
}
),
]
async def handle_assign_role(arguments: dict) -> Sequence[TextContent]:
filepath, output_path, modifier = _setup_modifier(arguments)
modifier.assign_role(
clip_id=arguments["clip_id"],
audio_role=arguments.get("audio_role"),
video_role=arguments.get("video_role"),
)
modifier.save(output_path)
roles_set = []
if arguments.get("audio_role"):
roles_set.append(f"audioRole={arguments['audio_role']}")
if arguments.get("video_role"):
roles_set.append(f"videoRole={arguments['video_role']}")
return _text_result((
f"Set {', '.join(roles_set)} on '{arguments['clip_id']}'\n\n"
f"Saved to: `{output_path}`"
))
async def handle_filter_by_role(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
role = arguments["role"].lower()
role_type = arguments.get("role_type", "any")
matches = []
for clip in tl.clips:
if role_type in ("audio", "any") and clip.audio_role.lower() == role:
matches.append((clip.name, "audio", clip.audio_role, format_duration(clip.duration_seconds)))
if role_type in ("video", "any") and clip.video_role.lower() == role:
matches.append((clip.name, "video", clip.video_role, format_duration(clip.duration_seconds)))
if not matches:
return _text_result(f"No clips found with role '{role}'.")
result = f"# Clips with role '{role}'\n\n"
result += "| Clip | Type | Role | Duration |\n|------|------|------|----------|\n"
for name, rtype, rval, dur in matches:
result += f"| {name} | {rtype} | {rval} | {dur} |\n"
return _text_result(result)
async def handle_export_role_stems(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
stems: dict[str, list] = {}
for clip in tl.clips:
role = clip.audio_role or "unassigned"
stems.setdefault(role, []).append(clip)
for cc in tl.connected_clips:
role = cc.role or "unassigned"
stems.setdefault(role, []).append(cc)
result = f"# Audio Stem Plan for {tl.name}\n\n"
for role, clips in sorted(stems.items()):
total_dur = sum(c.duration_seconds for c in clips)
result += f"## {role.title()} ({len(clips)} clips, {format_duration(total_dur)})\n\n"
for c in clips:
result += f"- {c.name} ({format_duration(c.duration_seconds)})\n"
result += "\n"
return _text_result(result)
HANDLERS = {
"assign_role": handle_assign_role,
"filter_by_role": handle_filter_by_role,
"export_role_stems": handle_export_role_stems,
}
+283
View File
@@ -0,0 +1,283 @@
"""Legendas dinâmicas (geração → validação) — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.media_intel import media_src_to_path
from fcpxml.model_manager import load_dynamic_subtitle_config
from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle
from fcpxml.writer import FCPXMLModifier
from server_tools._shared import (
_load_or_transcribe,
_markdown_table,
_setup_modifier,
_text_result,
_validate_filepath,
)
TOOLS = [
Tool(
name="validate_subtitle_layout",
description="Re-measure every title/subtitle in an FCPXML and report spatial collisions, frame and safe-area violations, and font fallbacks. Detects overlapping boxes only for titles on screen at the same time (half-open time intervals, so a title ending exactly as the next begins is never flagged). Returns a severity (none/render_tolerance/warning/probable/severe), the list of issues with suggested corrections, and summary counts.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"safe_margin_x": {"type": "number", "default": 0.05, "description": "Fraction of frame width to inset from each side (0.05 = 5%)"},
"safe_margin_y": {"type": "number", "default": 0.05, "description": "Fraction of frame height to inset from top/bottom"},
"min_font_size": {"type": "number", "description": "Flag titles whose emitted fontSize is below this readable minimum"},
"min_distance": {"type": "number", "description": "Flag same-block titles closer than this many pixels (insufficient_spacing)"},
"max_distance": {"type": "number", "description": "Flag same-block titles farther than this many pixels (excessive_spacing)"},
"output_format": {"type": "string", "enum": ["markdown", "json"], "default": "markdown", "description": "Report format"}
},
"required": ["filepath"]
}
),
Tool(
name="generate_dynamic_subtitles",
description="Generate progressive-composition subtitles as real, editable FCPXML title clips (the 'Text'/Basic Text template). Whisper's segments become sentences; each sentence is diagrammed as stacked blocks — supporting words grouped small in a grotesque, the sentence's key word alone and large in a display italic, body lines staggered to opposite edges. One <title> per block: each enters as its own words are spoken and stays on screen, so the sentence assembles itself, and every block clears at the same instant. Set granularity='word' for the older one-title-per-word rhythm. A sentence too tall for the band splits into successive compositions. These are TITLES, not captions: no subtitles role, so they render over the video without enabling caption display. Uses each media file's local Whisper word-level transcript (_transcript.json, auto-transcribes if missing). Style fields below (band_height through inactive_color) fall back to the style saved from the app's 'Legendas Dinâmicas' screen (~/.fcp-mcp-server/config.json via save_dynamic_subtitle_config) when omitted — pass a value here only to override that for one call. Non-destructive: writes a _dynamic_subtitles copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"language": {"type": "string", "description": "ISO language code hint (e.g. 'en'); auto-detected if omitted"},
"band_height": {"type": "number", "description": "Fraction of frame height the sentence block may fill before splitting into another block. Falls back to the saved style (default 0.22 — about three lines)"},
"block_center_y": {"type": "number", "description": "Vertical centre of the block in canvas points; negative sits below frame centre. Falls back to the saved style (default -167, just under centre)"},
"line_gap": {"type": "number", "description": "Air between stacked lines in canvas points. Lines are stacked on their real ink, so this is the whole distance beyond the glyphs themselves; negative values deliberately tuck each line into the one above. Falls back to the saved style (default 8)"},
"granularity": {"type": "string", "enum": ["phrase", "word"], "default": "phrase", "description": "'phrase': one title per LINE of the composition, key word set large (the reference look). 'word': one title per word."},
"emphasis_font": {"type": "string", "description": "Family for the key word (phrase mode). Must be installed on the editing Mac; unmeasured families fall back to estimated widths. Falls back to the saved style (default 'Playfair Display')"},
"emphasis_face": {"type": "string", "description": "Face for the key word, e.g. 'Medium Italic' or a script/calligraphic face. Falls back to the saved style (default 'Medium Italic')"},
"emphasis_size": {"type": "integer", "description": "Key-word size in canvas points, at the 2160x3840 reference frame. Falls back to the saved style (default 265)"},
"emphasis_color": {"type": "string", "description": "RGBA (0-1, space-separated) for the key word (phrase mode). Defaults to active_color, so the block reads in a single colour unless the key word is deliberately set apart"},
"text_scale": {"type": "number", "description": "Ratio between the title template's fontSize space and the canvas-point space it positions in. The \"Text\" template sizes type in frame pixels, so sizes are doubled on the way out. Falls back to the saved style (default 2.0). Lower it only if a template renders type larger than the chosen point size"},
"font": {"type": "string", "description": "Title font family (supporting lines in phrase mode). Falls back to the saved style (default 'Helvetica Neue')"},
"font_size": {"type": "integer", "description": "Supporting-line font size in canvas points, at the 2160x3840 reference frame. Falls back to the saved style (default 104)"},
"active_color": {"type": "string", "description": "RGBA (0-1, space-separated) for even-indexed lines. Falls back to the saved style (default '1 1 1 1')"},
"inactive_color": {"type": "string", "default": "0.7 0.7 0.7 1", "description": "RGBA (0-1, space-separated) for odd-indexed lines — alternates with active_color for visual variety between stacked lines"},
"output_path": {"type": "string", "description": "Output path (default: adds _dynamic_subtitles suffix)"},
},
"required": ["filepath"]
}
),
]
async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]:
"""Validate title/subtitle layout for spatial collisions and safe-area
containment (collision.validate_titles over every <title> in the file)."""
filepath = _validate_filepath(arguments["filepath"], (".fcpxml", ".fcpxmld"))
modifier = FCPXMLModifier(filepath)
report = modifier.validate_subtitle_layout(
safe_margin_x=float(arguments.get("safe_margin_x", 0.05)),
safe_margin_y=float(arguments.get("safe_margin_y", 0.05)),
min_font_size=(
float(arguments["min_font_size"])
if arguments.get("min_font_size") is not None else None
),
min_distance=(
float(arguments["min_distance"])
if arguments.get("min_distance") is not None else None
),
max_distance=(
float(arguments["max_distance"])
if arguments.get("max_distance") is not None else None
),
)
if arguments.get("output_format") == "json":
return _text_result(json.dumps(report, indent=2))
summary = report["summary"]
lines = [
"# Subtitle Layout Validation",
"",
f"## Summary (severity: {report['severity']})",
f"- **Titles**: {summary['title_count']}",
f"- **Issues**: {summary['issue_count']}",
f"- **Collisions**: {summary['spatial_collision']}",
f"- **Outside frame**: {summary['outside_frame']}",
f"- **Outside safe area**: {summary['outside_safe_area']}",
f"- **Font fallback**: {summary['font_missing']}",
f"- **Font too small**: {summary['font_too_small']}",
"",
]
issues = report["issues"]
if issues:
lines.append(f"## Issues ({len(issues)})")
for issue in issues:
sev = issue["severity"].upper()
if issue["type"] == "spatial_collision":
corr = issue["suggested_correction"]
lines.append(
f"- [{sev}] collision: \"{issue['first_title']}\" x "
f"\"{issue['second_title']}\" "
f"(overlap {issue['overlap_width']:.0f}x"
f"{issue['overlap_height']:.0f} = "
f"{issue['overlap_area']:.0f}px, ratio "
f"{issue['overlap_ratio']:.2f}, move "
f"{corr['axis']} {corr['minimum_movement']:.0f}px)"
)
else:
detail = issue.get("title", "") or issue.get("font", "")
lines.append(f"- [{sev}] {issue['type']}: {detail}".rstrip())
else:
lines.append("_No issues found — no simultaneous titles overlap._")
return _text_result("\n".join(lines))
async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextContent]:
"""Generate per-word subtitle titles laid out as a block per sentence.
Whisper's segments become sentences; each word becomes its own positioned
<title> connected clip, appearing as it is spoken and accumulating on
screen until the whole block clears at once. No compound clip.
Reuses the same SOURCE-media -> TIMELINE mapping as ``transcript_markers``
(``modifier.source_file_start`` per spine clip) so word timestamps land
at the correct position even across trimmed/multiple clips.
"""
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
clip_filter = arguments.get("clip_name")
# Anything the caller didn't explicitly pass falls back to the style
# persisted from the "Legendas Dinâmicas" screen (~/.fcp-mcp-server/
# config.json), not a hardcoded default — so the UI is the single place
# that configures the look, and every caller (app, MCP, this session)
# renders the same thing without threading 11 fields through every call.
saved = load_dynamic_subtitle_config()
body_color = arguments.get("active_color") or saved["active_color"]
config = DynamicSubtitleConfig(
style=WordStyle(
font=arguments.get("font") or saved["font"],
font_size=int(arguments.get("font_size", saved["font_size"])),
active_color=body_color,
inactive_color=arguments.get("inactive_color", "0.7 0.7 0.7 1"),
emphasis_look=WordLook(
int(arguments.get("emphasis_size", saved["emphasis_size"])),
arguments.get("emphasis_color") or saved["emphasis_color"] or body_color,
font=arguments.get("emphasis_font") or saved["emphasis_font"],
face=arguments.get("emphasis_face") or saved["emphasis_face"],
kerning=0.0,
),
body_look=WordLook(
int(arguments.get("font_size", saved["font_size"])),
body_color,
font=arguments.get("font") or saved["font"],
face="Bold",
kerning=1.2,
),
),
band_height=float(arguments.get("band_height", saved["band_height"])),
block_center_y=float(arguments.get("block_center_y", saved["block_center_y"])),
granularity=arguments.get("granularity", "phrase"),
text_scale=float(arguments.get("text_scale", saved["text_scale"])),
line_gap=float(arguments.get("line_gap", saved["line_gap"])),
)
filepath, output_path, modifier = _setup_modifier(arguments, "_dynamic_subtitles")
added: list[tuple[str, int, int]] = []
skipped: list[tuple[str, str]] = []
spine_clips = [el for _, el in modifier._iter_spine_clips()]
for el in spine_clips:
name = el.get("name", "")
if clip_filter and name != clip_filter:
continue
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
media_path = media_src_to_path(src)
if not media_path or not Path(media_path).is_file():
skipped.append((name, "media file missing"))
continue
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
if data is None:
skipped.append((name, reason))
continue
clip_source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
window_end = clip_source_start + clip_duration
clip_words = [
{
"word": w.get("word", ""),
"start": float(w.get("start", 0.0)) - clip_source_start,
"end": float(w.get("end", 0.0)) - clip_source_start,
}
for w in data.get("words", [])
if clip_source_start <= float(w.get("start", 0.0)) < window_end
]
if not clip_words:
skipped.append((name, "no words in clip's source range"))
continue
# Sentence boundaries, rebased the same way, so each sentence becomes
# its own block of titles that builds up and then clears together.
# Overlap rather than containment: a segment straddling the clip's
# in-point still governs the words that made the cut.
clip_segments = [
{
"start": float(s.get("start", 0.0)) - clip_source_start,
"end": float(s.get("end", 0.0)) - clip_source_start,
}
for s in data.get("segments", [])
if float(s.get("end", 0.0)) > clip_source_start
and float(s.get("start", 0.0)) < window_end
]
# Pass the element itself, not `name` — after ripple-cut/silence
# removal every fragment of an originally-named clip keeps the same
# `name`, so a name lookup here would resolve every clip in this
# loop to whichever one `self.clips` last indexed, stacking every
# clip's captions onto a single wrong spine element instead of each
# clip's own. See Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-17.
lines = modifier.generate_dynamic_subtitles(
el, clip_words, config, segments=clip_segments
)
added.append((name, len(lines), len(clip_words)))
if not added:
text = "# Dynamic Subtitles\n\nNo captions generated — file unchanged (nothing saved)."
if skipped:
text += "\n\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
return _text_result(text)
modifier.save(output_path)
total_lines = sum(lines for _, lines, _ in added)
total_words = sum(words for _, _, words in added)
result = "# Dynamic Subtitles Generated (local Whisper)\n\n## Summary\n"
result += (
f"- **Clips Captioned**: {len(added)}\n"
f"- **Caption Lines (Title Clips)**: {total_lines}\n"
f"- **Total Words**: {total_words}\n\n"
)
result += _markdown_table(
["Clip", "Caption Lines", "Words"],
[[n, str(lines), str(words)] for n, lines, words in added],
)
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
return _text_result(result)
HANDLERS = {
"validate_subtitle_layout": handle_validate_subtitle_layout,
"generate_dynamic_subtitles": handle_generate_dynamic_subtitles,
}
+400
View File
@@ -0,0 +1,400 @@
"""Timeline & análise (Projeto) — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.diff import compare_timelines
from fcpxml.models import MarkerType
from fcpxml.parser import FCPXMLParser
from fcpxml.writer import list_effects
from server_tools._shared import (
_SANDBOX_ENABLED,
PROJECTS_DIR,
_require_timeline,
_text_result,
_validate_directory,
_validate_filepath,
find_fcpxml_files,
format_duration,
format_timecode,
)
TOOLS = [
Tool(
name="list_projects",
description="List all FCPXML projects in directory",
inputSchema={
"type": "object",
"properties": {
"directory": {"type": "string", "description": "Directory to search (default: ~/Movies)"}
}
}
),
Tool(
name="analyze_timeline",
description="Get comprehensive timeline statistics including duration, resolution, clip count, pacing metrics",
inputSchema={
"type": "object",
"properties": {"filepath": {"type": "string", "description": "Path to FCPXML file"}},
"required": ["filepath"]
}
),
Tool(
name="list_clips",
description="List all clips with timecodes, durations, and metadata",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"limit": {"type": "integer", "description": "Max clips to return"}
},
"required": ["filepath"]
}
),
Tool(
name="list_markers",
description="Extract markers (chapter, todo, standard) with timestamps",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string"},
"marker_type": {"type": "string", "enum": ["all", "chapter", "todo", "standard", "completed"]},
"format": {"type": "string", "enum": ["detailed", "youtube", "simple"]}
},
"required": ["filepath"]
}
),
Tool(
name="list_keywords",
description="Extract all keywords/tags from project",
inputSchema={
"type": "object",
"properties": {"filepath": {"type": "string"}},
"required": ["filepath"]
}
),
Tool(
name="list_library_clips",
description="List all available clips in the library (source media, not yet on timeline)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"keywords": {"type": "array", "items": {"type": "string"}, "description": "Filter by keywords"},
"limit": {"type": "integer", "description": "Max clips to return"}
},
"required": ["filepath"]
}
),
Tool(
name="list_connected_clips",
description="List all connected clips (B-roll, titles, audio) with their lanes and parent clips",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"lane": {"type": "integer", "description": "Filter by lane number (positive=above, negative=below)"},
},
"required": ["filepath"]
}
),
Tool(
name="list_compound_clips",
description="List compound clips (ref-clips) and their nested content",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
},
"required": ["filepath"]
}
),
Tool(
name="list_roles",
description="List all audio/video roles used in the timeline with clip counts",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
},
"required": ["filepath"]
}
),
Tool(
name="diff_timelines",
description="Compare two FCPXML files and report differences in clips, markers, transitions, and format",
inputSchema={
"type": "object",
"properties": {
"filepath_a": {"type": "string", "description": "Path to first FCPXML file (baseline)"},
"filepath_b": {"type": "string", "description": "Path to second FCPXML file (comparison)"},
},
"required": ["filepath_a", "filepath_b"]
}
),
Tool(
name="list_effects",
description="List all available FCP transition effects with slugs and UUIDs",
inputSchema={
"type": "object",
"properties": {},
}
),
]
async def handle_list_projects(arguments: dict) -> Sequence[TextContent]:
directory = arguments.get("directory", PROJECTS_DIR)
resolved_dir = _validate_directory(
directory, allowed_root=PROJECTS_DIR if _SANDBOX_ENABLED else None
)
files = find_fcpxml_files(resolved_dir)
if not files:
return _text_result(f"No FCPXML files found in {directory}")
return _text_result(f"Found {len(files)} FCPXML file(s):\n" + "\n".join(f" - {f}" for f in files))
async def handle_analyze_timeline(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
durs = [c.duration_seconds for c in tl.clips]
avg, med, mn, mx = (0, 0, 0, 0) if not durs else (
sum(durs)/len(durs), sorted(durs)[len(durs)//2], min(durs), max(durs))
return _text_result(f"""# Timeline Analysis: {tl.name}
## Overview
- **Duration**: {format_duration(tl.duration.seconds)}
- **Resolution**: {tl.width}x{tl.height} @ {tl.frame_rate}fps
## Clip Statistics
- **Total Clips**: {tl.total_clips}
- **Total Cuts**: {tl.total_cuts}
- **Transitions**: {len(tl.transitions)}
## Pacing
- **Average**: {format_duration(avg)}
- **Median**: {format_duration(med)}
- **Shortest**: {format_duration(mn)}
- **Longest**: {format_duration(mx)}
- **Cuts/Minute**: {tl.cuts_per_minute:.1f}
## Markers
- **Total**: {len(tl.markers)}
- **Chapters**: {len([m for m in tl.markers if m.marker_type == MarkerType.CHAPTER])}
""")
async def handle_list_clips(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
limit = arguments.get("limit")
clips = tl.clips[:limit] if limit else tl.clips
result = f"# Clips in {tl.name}\n\n| # | Name | Start | Duration | Keywords |\n|---|------|-------|----------|----------|\n"
for i, c in enumerate(clips, 1):
kws = ", ".join(k.value for k in c.keywords) if c.keywords else "-"
result += f"| {i} | {c.name} | {format_timecode(c.start)} | {format_duration(c.duration_seconds)} | {kws} |\n"
return _text_result(result)
async def handle_list_markers(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
markers = list(tl.markers)
for clip in tl.clips:
markers.extend(clip.markers)
marker_type = arguments.get("marker_type", "all")
if marker_type != "all":
markers = [m for m in markers if m.marker_type == MarkerType.from_string(marker_type)]
markers.sort(key=lambda m: m.start.frames)
fmt = arguments.get("format", "detailed")
if fmt == "youtube":
result = "# YouTube Chapters\n\n" + "\n".join(f"{m.to_youtube_timestamp()} {m.name}" for m in markers)
elif fmt == "simple":
result = "\n".join(f"{format_timecode(m.start)} - {m.name}" for m in markers)
else:
result = f"# Markers ({len(markers)})\n\n| TC | Name | Type |\n|---|------|------|\n"
result += "\n".join(f"| {format_timecode(m.start)} | {m.name} | {m.marker_type.value} |" for m in markers)
return _text_result(result)
async def handle_list_keywords(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
keywords = {}
for clip in tl.clips:
for kw in clip.keywords:
keywords.setdefault(kw.value, []).append(clip.name)
if not keywords:
return _text_result("No keywords found")
result = f"# Keywords ({len(keywords)})\n\n"
for kw, clips in sorted(keywords.items()):
result += f"**{kw}** ({len(clips)} clips)\n"
return _text_result(result)
async def handle_list_library_clips(arguments: dict) -> Sequence[TextContent]:
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
parser = FCPXMLParser()
parser.parse_file(filepath)
keywords = arguments.get("keywords")
library_clips = parser.get_library_clips(keywords=keywords)
limit = arguments.get("limit")
if limit:
library_clips = library_clips[:limit]
if not library_clips:
return _text_result("No library clips found")
result = f"# Library Clips ({len(library_clips)} available)\n\n"
result += "| ID | Name | Duration | Has Video | Has Audio |\n"
result += "|----|------|----------|-----------|----------|\n"
for c in library_clips:
result += f"| {c['asset_id']} | {c['name']} | {format_duration(c['duration_seconds'])} | {'Y' if c['has_video'] else 'N'} | {'Y' if c['has_audio'] else 'N'} |\n"
result += "\n*Use `insert_clip` to add these to your timeline.*"
return _text_result(result)
async def handle_list_connected_clips(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
lane_filter = arguments.get("lane")
clips = tl.connected_clips
if lane_filter is not None:
clips = [c for c in clips if c.lane == lane_filter]
if not clips:
return _text_result("No connected clips found in timeline.")
result = f"# Connected Clips in {tl.name}\n\n**Total**: {len(clips)}\n\n"
result += "| # | Name | Lane | Type | Duration | Parent | Role |\n"
result += "|---|------|------|------|----------|--------|------|\n"
for i, c in enumerate(clips, 1):
result += (
f"| {i} | {c.name} | {c.lane} | {c.clip_type} | "
f"{format_duration(c.duration_seconds)} | {c.parent_clip_name} | "
f"{c.role or '-'} |\n"
)
return _text_result(result)
async def handle_list_compound_clips(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
if not tl.compound_clips:
return _text_result("No compound clips found in timeline.")
result = f"# Compound Clips in {tl.name}\n\n"
for i, cc in enumerate(tl.compound_clips, 1):
result += f"### {i}. {cc.name}\n"
result += f"- **Ref ID**: {cc.ref_id}\n"
result += f"- **Duration**: {format_duration(cc.duration_seconds)}\n"
result += f"- **Clips inside**: {len(cc.clips)}\n\n"
return _text_result(result)
async def handle_list_roles(arguments: dict) -> Sequence[TextContent]:
project, tl = _require_timeline(arguments["filepath"])
audio_roles: dict[str, int] = {}
video_roles: dict[str, int] = {}
for clip in tl.clips:
if clip.audio_role:
audio_roles[clip.audio_role] = audio_roles.get(clip.audio_role, 0) + 1
if clip.video_role:
video_roles[clip.video_role] = video_roles.get(clip.video_role, 0) + 1
for cc in tl.connected_clips:
if cc.role:
# Determine type from clip_type
if cc.clip_type in ('audio', 'audio-clip'):
audio_roles[cc.role] = audio_roles.get(cc.role, 0) + 1
else:
video_roles[cc.role] = video_roles.get(cc.role, 0) + 1
result = f"# Roles in {tl.name}\n\n"
if audio_roles:
result += "## Audio Roles\n\n| Role | Clips |\n|------|-------|\n"
for role, count in sorted(audio_roles.items()):
result += f"| {role} | {count} |\n"
else:
result += "## Audio Roles\n\nNo audio roles assigned.\n"
result += "\n"
if video_roles:
result += "## Video Roles\n\n| Role | Clips |\n|------|-------|\n"
for role, count in sorted(video_roles.items()):
result += f"| {role} | {count} |\n"
else:
result += "## Video Roles\n\nNo video roles assigned.\n"
return _text_result(result)
async def handle_diff_timelines(arguments: dict) -> Sequence[TextContent]:
filepath_a = _validate_filepath(arguments["filepath_a"], ('.fcpxml', '.fcpxmld'))
filepath_b = _validate_filepath(arguments["filepath_b"], ('.fcpxml', '.fcpxmld'))
diff = compare_timelines(filepath_a, filepath_b)
if not diff.has_changes:
return _text_result((
f"# Timeline Diff: No Changes\n\n"
f"**{diff.timeline_a_name}** vs **{diff.timeline_b_name}** are identical."
))
result = (
f"# Timeline Diff\n\n"
f"**Baseline**: {diff.timeline_a_name}\n"
f"**Comparison**: {diff.timeline_b_name}\n"
f"**Total changes**: {diff.total_changes}\n\n"
)
if diff.format_changes:
result += "## Format Changes\n\n"
for change in diff.format_changes:
result += f"- {change}\n"
result += "\n"
clip_changes = [d for d in diff.clip_diffs if d.action != "unchanged"]
if clip_changes:
result += "## Clip Changes\n\n| Action | Clip | Details |\n|--------|------|--------|\n"
for d in clip_changes:
result += f"| {d.action.upper()} | {d.clip_name} | {d.details} |\n"
result += "\n"
if diff.marker_diffs:
result += "## Marker Changes\n\n| Action | Marker | Details |\n|--------|--------|--------|\n"
for d in diff.marker_diffs:
result += f"| {d.action.upper()} | {d.marker_name} | {d.details} |\n"
result += "\n"
if diff.transition_diffs:
result += "## Transition Changes\n\n"
for change in diff.transition_diffs:
result += f"- {change}\n"
return _text_result(result)
async def handle_list_effects(arguments: dict) -> Sequence[TextContent]:
effects = list_effects()
lines = ["# Available FCP Transition Effects\n"]
for eff in effects:
lines.append(f"- **{eff['slug']}**: {eff['name']} (`{eff['uuid']}`)")
return _text_result("\n".join(lines))
HANDLERS = {
"list_projects": handle_list_projects,
"analyze_timeline": handle_analyze_timeline,
"list_clips": handle_list_clips,
"list_markers": handle_list_markers,
"list_keywords": handle_list_keywords,
"list_library_clips": handle_list_library_clips,
"list_connected_clips": handle_list_connected_clips,
"list_compound_clips": handle_list_compound_clips,
"list_roles": handle_list_roles,
"diff_timelines": handle_diff_timelines,
"list_effects": handle_list_effects,
}
+235
View File
@@ -0,0 +1,235 @@
"""Transcrição & edição por transcrição — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
from pathlib import Path
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.media_intel import media_src_to_path
from fcpxml.transcribe import (
DEFAULT_FILLERS,
find_filler_spans,
find_phrase_spans,
merge_ranges,
segments_to_srt,
)
from server_tools._shared import (
_TRANSCRIBE_INSTALL_HINT,
TRANSCRIBE_MAX_MEDIA,
_cut_transcript_spans,
_load_or_transcribe,
_markdown_table,
_require_timeline,
_setup_modifier,
_text_result,
_transcript_cut_report,
_validate_output_path,
format_duration,
)
TOOLS = [
Tool(
name="transcribe_media",
description="Transcribe each clip's source media locally with word-level timestamps (faster-whisper). Writes a _transcript.json next to each media file (reused by edit_by_transcript / remove_filler_words so media is only transcribed once) and optionally an SRT for captions. Requires the optional [transcribe] extra; degrades to an install hint without it.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"clip_name": {"type": "string", "description": "Only transcribe the clip with this name"},
"model": {"type": "string", "default": "base", "description": "Whisper model size: tiny, base, small, medium, large-v3 (default base; larger = slower + more accurate)"},
"language": {"type": "string", "description": "ISO language code hint (e.g. 'en'); auto-detected if omitted"},
"write_srt": {"type": "boolean", "default": False, "description": "Also write a _transcript.srt next to each media file (plugs into import_srt_markers)"},
},
"required": ["filepath"]
}
),
Tool(
name="edit_by_transcript",
description="Text-based editing: cut timeline content by what was SAID. mode=remove cuts every occurrence of the given phrases (with ripple); mode=keep_only keeps only the matched phrases and cuts everything else in each matched clip (clips with no matches are left untouched). Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _transcript_edit copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"phrases": {"type": "array", "items": {"type": "string"}, "description": "Spoken phrases to match (case/punctuation-insensitive)"},
"mode": {"type": "string", "enum": ["remove", "keep_only"], "default": "remove", "description": "remove=cut matches out; keep_only=keep only matches"},
"clip_name": {"type": "string", "description": "Only edit the clip with this name"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"padding": {"type": "number", "default": 0.0, "description": "Seconds to widen each cut on both sides (0-2, default 0)"},
"output_path": {"type": "string", "description": "Output path (default: adds _transcript_edit suffix)"},
},
"required": ["filepath", "phrases"]
}
),
Tool(
name="remove_filler_words",
description="Cut filler words (um, uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: um, uh, uhh, umm, erm, ehm, mmm, hmm, mhm)"},
"clip_name": {"type": "string", "description": "Only clean the clip with this name"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"padding": {"type": "number", "default": 0.02, "description": "Seconds to widen each cut on both sides (0-2, default 0.02)"},
"output_path": {"type": "string", "description": "Output path (default: adds _defillered suffix)"},
},
"required": ["filepath"]
}
),
]
async def handle_transcribe_media(arguments: dict) -> Sequence[TextContent]:
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
write_srt = bool(arguments.get("write_srt", False))
_, tl = _require_timeline(arguments["filepath"])
clip_filter = arguments.get("clip_name")
done: dict[str, dict | None] = {}
skipped: list[tuple[str, str]] = []
rows: list[list[str]] = []
srt_paths: list[str] = []
for clip in tl.clips:
if clip_filter and clip.name != clip_filter:
continue
media_path = media_src_to_path(clip.media_path or "")
if not media_path or not Path(media_path).is_file():
skipped.append((clip.name, "media file missing"))
continue
if media_path in done:
continue
if len(done) >= TRANSCRIBE_MAX_MEDIA:
skipped.append((clip.name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)"))
continue
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
done[media_path] = data
if data is None:
skipped.append((clip.name, reason))
continue
if write_srt and data.get("segments"):
srt_name = Path(media_path).stem + "_transcript.srt"
srt_anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent)
srt_path = _validate_output_path(
str(Path(srt_anchor) / srt_name),
anchor_dir=srt_anchor,
)
with open(srt_path, "w") as f:
f.write(segments_to_srt(data["segments"]))
srt_paths.append(srt_path)
preview = data.get("text", "")[:160]
rows.append([
Path(media_path).name,
data.get("language", "?"),
str(len(data.get("words", []))),
format_duration(float(data.get("duration", 0.0))),
preview + ("…" if len(data.get("text", "")) > 160 else ""),
])
result = f"""# Media Transcription (local Whisper)
## Summary
- **Model**: {model}
- **Media Files Transcribed**: {len(rows)}
"""
if rows:
result += "\n## Transcripts (saved as _transcript.json next to each media file)\n"
result += _markdown_table(
["Media", "Language", "Words", "Duration", "Preview"], rows
) + "\n"
result += (
"\n*Next: `edit_by_transcript` to cut by what was said, or "
"`remove_filler_words` to clean ums/uhs. Transcripts are cached — "
"media is only transcribed once.*"
)
if srt_paths:
result += "\n\n## SRT Files\n" + "\n".join(f"- {p}" for p in srt_paths)
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
) + "\n"
if not rows and any("faster-whisper" in reason for _, reason in skipped):
result += _TRANSCRIBE_INSTALL_HINT
return _text_result(result)
async def handle_edit_by_transcript(arguments: dict) -> Sequence[TextContent]:
phrases = arguments.get("phrases") or []
if not isinstance(phrases, list) or not all(isinstance(p, str) for p in phrases):
raise ValueError("phrases must be a list of strings")
phrases = [p for p in phrases if p.strip()]
if not phrases:
raise ValueError("phrases must contain at least one non-empty string")
mode = arguments.get("mode", "remove")
if mode not in ("remove", "keep_only"):
raise ValueError(f"mode must be 'remove' or 'keep_only', got {mode!r}")
padding = float(arguments.get("padding", 0.0))
if not (0 <= padding <= 2):
raise ValueError(f"padding must be between 0 and 2 seconds, got {padding}")
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
filepath, output_path, modifier = _setup_modifier(arguments, "_transcript_edit")
def spans_fn(words):
return merge_ranges(
[span for phrase in phrases for span in find_phrase_spans(words, phrase)]
)
cuts_made, skipped = _cut_transcript_spans(
modifier, arguments.get("clip_name"), model, language, padding,
spans_fn, keep_only=(mode == "keep_only"), output_dir=output_dir,
)
if cuts_made:
modifier.save(output_path)
verb = "kept only" if mode == "keep_only" else "removed"
return _transcript_cut_report(
"Transcript Edit",
[f"- **Mode**: {mode} ({verb} the matched phrases)",
f"- **Phrases**: {', '.join(repr(p) for p in phrases)}",
f"- **Padding**: {padding}s"],
cuts_made, skipped, output_path,
"*Transcripts are cached as _transcript.json. Original file untouched.*",
)
async def handle_remove_filler_words(arguments: dict) -> Sequence[TextContent]:
fillers = arguments.get("fillers") or list(DEFAULT_FILLERS)
if not isinstance(fillers, list) or not all(isinstance(f, str) for f in fillers):
raise ValueError("fillers must be a list of strings")
padding = float(arguments.get("padding", 0.02))
if not (0 <= padding <= 2):
raise ValueError(f"padding must be between 0 and 2 seconds, got {padding}")
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
filepath, output_path, modifier = _setup_modifier(arguments, "_defillered")
cuts_made, skipped = _cut_transcript_spans(
modifier, arguments.get("clip_name"), model, language, padding,
lambda words: merge_ranges(find_filler_spans(words, fillers)),
output_dir=output_dir,
)
if cuts_made:
modifier.save(output_path)
return _transcript_cut_report(
"Filler Word Removal",
[f"- **Fillers**: {', '.join(fillers)}", f"- **Padding**: {padding}s"],
cuts_made, skipped, output_path,
"*Transcripts are cached as _transcript.json. Original file untouched.*",
)
HANDLERS = {
"transcribe_media": handle_transcribe_media,
"edit_by_transcript": handle_edit_by_transcript,
"remove_filler_words": handle_remove_filler_words,
}
+751
View File
@@ -0,0 +1,751 @@
"""Voz (análise → decisão → aplicação) — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import List, Optional, Sequence, Tuple
from mcp.types import TextContent, Tool
from fcpxml.diarize import assign_speakers, build_speakers, diarization_capability, diarize
from fcpxml.emphasis import EmphasisWeights
from fcpxml.media_intel import media_src_to_path
from fcpxml.model_manager import (
load_hf_token,
load_num_speakers,
load_voice_analysis_config,
save_voice_analysis_config,
)
from fcpxml.models import TimeValue
from fcpxml.voice_actions import parse_actions, resolve_actions, speaker_cut_actions
from fcpxml.voice_features import extract_energy, extract_pitch, features_capability
from fcpxml.voice_timeline import (
build_voice_timeline,
enrich_words,
load_voice_timeline,
restrict_to_kept,
save_voice_timeline,
select_peaks,
suggest_zoom_windows,
voice_timeline_path,
)
from fcpxml.writer import FCPXMLModifier
from server_tools._shared import (
_DIARIZATION_INSTALL_HINT,
_FEATURES_INSTALL_HINT,
_TRANSCRIBE_INSTALL_HINT,
AUDIO_MEDIA_EXTENSIONS,
MAX_MEDIA_FILE_SIZE,
_apply_placed_action,
_load_or_transcribe,
_markdown_table,
_setup_modifier,
_speaker_table,
_text_result,
_validate_filepath,
_validate_output_path,
_voice_analysis_config_text,
format_duration,
)
TOOLS = [
Tool(
name="diarize_media",
description="Identify WHO is speaking (speaker diarization) in an audio/video file using pyannote.audio, and assign SPEAKER_NN labels to each word/segment of its cached transcript. Writes a _diarization.json next to the media file. Requires the optional [diarization] extra (pyannote.audio) and a HuggingFace token with access to pyannote/speaker-diarization-3.1 (set once via save_hf_token or the HF_TOKEN argument); degrades to an install/token hint without them. Transcribes first if no _transcript.json is cached yet.",
inputSchema={
"type": "object",
"properties": {
"media_path": {"type": "string", "description": "Path to audio/video file (.wav, .mp3, .m4a, .aac, .aif, .flac, .mov, .mp4)"},
"hf_token": {"type": "string", "description": "HuggingFace token with pyannote/speaker-diarization-3.1 access (default: the persisted token from save_hf_token, if any)"},
"num_speakers": {"type": "string", "description": "Known number of speakers, if you know it (speeds up and improves accuracy). Leave empty to auto-detect."},
"model": {"type": "string", "default": "base", "description": "Whisper model size to use if transcription is needed (default base)"},
"language": {"type": "string", "description": "ISO language code hint for transcription, if needed"},
},
"required": ["media_path"]
}
),
Tool(
name="analyze_voice_features",
description="Analyze HOW a voice is speaking: pitch, energy, local speech rate, pauses, and a combined emphasis index (0-1) per transcribed word, using the persisted Voice Analysis settings (energy threshold, emphasis weights, emphasis cutoff — see save_voice_analysis_config). Writes a _voice_features.json next to the media file. Requires the optional [intelligence] extra (librosa); degrades to an install hint without it. Transcribes first if no _transcript.json is cached yet.",
inputSchema={
"type": "object",
"properties": {
"media_path": {"type": "string", "description": "Path to audio/video file (.wav, .mp3, .m4a, .aac, .aif, .flac, .mov, .mp4)"},
"model": {"type": "string", "default": "base", "description": "Whisper model size to use if transcription is needed (default base)"},
"language": {"type": "string", "description": "ISO language code hint for transcription, if needed"},
},
"required": ["media_path"]
}
),
Tool(
name="build_voice_timeline",
description="Build the consolidated voice timeline: WHAT was said (transcript), WHO said it (diarization), and HOW it was said (pitch/energy/rate/pauses -> emphasis index), merged into one AI-readable JSON written next to the media as _voice_timeline.json. This is the source of truth for automated editing — layered as summary -> segments -> words, with all acoustic values normalized 0-1 and documented inline, so a model can read the narrative shape and decide how to direct the edit. Every layer degrades independently: no librosa means acoustic values are 0, no HuggingFace token means a single default speaker; the document shape never changes.",
inputSchema={
"type": "object",
"properties": {
"media_path": {"type": "string", "description": "Path to audio/video file (.wav, .mp3, .m4a, .aac, .aif, .flac, .mov, .mp4)"},
"model": {"type": "string", "default": "base", "description": "Whisper model size to use if transcription is needed (default base)"},
"language": {"type": "string", "description": "ISO language code hint for transcription, if needed"},
"hf_token": {"type": "string", "description": "HuggingFace token for speaker diarization (default: the persisted token; omit to skip diarization)"},
"num_speakers": {"type": "string", "description": "Known number of speakers, if any (default: the persisted setting, else auto-detect)"},
"output_dir": {"type": "string", "description": "Folder to write _voice_timeline.json into (default: next to the media file)"},
},
"required": ["media_path"]
}
),
Tool(
name="remove_speakers",
description="Cut everything one or more speakers say out of the timeline — the standard cleanup on an interview shoot, where the interviewer or a crew member talks during the take and only the subject should survive. Reads the media's _voice_timeline.json (build it first with build_voice_timeline, which must have run with diarization so speakers are separated). Call without speaker_ids to just LIST who was detected, with speaking share and sample lines, so you can tell who is who before cutting anything. Non-destructive: writes a _voice_edit copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"media_path": {"type": "string", "description": "Media whose _voice_timeline.json holds the speakers (default: the timeline's first clip media)"},
"speaker_ids": {"type": "array", "items": {"type": "string"}, "description": "Speakers to REMOVE (e.g. [\"SPEAKER_01\"]). Omit to only list the detected speakers without editing."},
"padding": {"type": "number", "default": 0.15, "description": "Seconds trimmed inside each cut so the kept speaker's first syllable is never clipped (default 0.15)"},
"output_path": {"type": "string", "description": "Output path (default: adds _voice_edit suffix)"},
},
"required": ["filepath"]
}
),
Tool(
name="refine_voice_timeline",
description="Re-analyze a voice timeline over only the material that survives a set of cuts, then propose punch-in windows over it. Emphasis is RELATIVE — energy is scored against the loudest word of the recording — so once the loudest moment is cut (a laugh, an aside to the crew), every remaining score is measured against something the viewer will never see and the ranking points at the wrong words. Run this after deciding cuts and before deciding zooms. Cheap: it re-normalizes the already-measured numbers, never re-reads the audio. Times stay in ORIGINAL source seconds, so the result feeds straight back into apply_voice_actions.",
inputSchema={
"type": "object",
"properties": {
"media_path": {"type": "string", "description": "Media whose _voice_timeline.json will be refined (build it first with build_voice_timeline)"},
"cuts": {
"type": "array",
"description": "The ranges being REMOVED, in original source seconds. Pass the cut actions you already decided; anything overlapping them is excluded from the re-analysis.",
"items": {
"type": "object",
"properties": {
"start": {"type": "number", "description": "Start in original source seconds"},
"end": {"type": "number", "description": "End in original source seconds"},
},
"required": ["start", "end"],
},
},
"min_gap": {"type": "number", "default": 8.0, "description": "Minimum seconds between two proposed zooms — effects stacked close together read as nervous editing (default 8.0)"},
"max_zooms": {"type": "integer", "description": "Cap on how many zoom candidates to return (default: no cap — cut the list by rhythm yourself)"},
"save": {"type": "boolean", "default": False, "description": "Also write the refined timeline as _voice_timeline_refined.json next to the media"},
"output_dir": {"type": "string", "description": "Folder holding _voice_timeline.json (default: next to the media file)"},
},
"required": ["media_path", "cuts"]
}
),
Tool(
name="apply_voice_actions",
description="Apply a list of editing decisions (from the rules engine, or from a model that read the _voice_timeline.json) to a timeline, producing FCPXML. Actions are validated first and reported per row, so one malformed decision never discards the edit. All action times are in ORIGINAL source seconds: cuts are resolved first and every other action is moved onto its post-cut position automatically, so decisions never land on the wrong frame. Actions pointing into removed material are dropped and reported, not silently slid. Non-destructive: writes a _voice_edit copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"actions": {
"type": "array",
"description": "The decision list. Each item: {kind, start, end, params, reason, speaker}. kind is cut | zoom | text | marker. Times in original source seconds. zoom takes params.scale (1.0-3.0, default 1.3); text requires params.content.",
"items": {
"type": "object",
"properties": {
"kind": {"type": "string", "enum": ["cut", "zoom", "text", "marker"]},
"start": {"type": "number", "description": "Start in original source seconds"},
"end": {"type": "number", "description": "End in original source seconds"},
"params": {"type": "object", "description": "kind-specific: {scale} for zoom, {content} for text"},
"reason": {"type": "string", "description": "Why this decision was made — kept for review"},
"speaker": {"type": "string", "description": "Speaker id this decision relates to, if any"},
},
"required": ["kind", "start", "end"],
},
},
"output_path": {"type": "string", "description": "Output path (default: adds _voice_edit suffix)"},
},
"required": ["filepath", "actions"]
}
),
Tool(
name="get_voice_analysis_config",
description="Read the persisted Voice Analysis settings: energy threshold, emphasis-index weights (energy/pitch_variation/rate_variation/pause_before/duration), emphasis cutoff for punch-in candidates, and emotion detection toggle/sensitivity. Shared with the MacApp settings screen (~/.fcp-mcp-server/config.json).",
inputSchema={"type": "object", "properties": {}}
),
Tool(
name="save_voice_analysis_config",
description="Persist Voice Analysis settings. Only the fields you pass are changed; omitted fields keep their current value. emphasis_weights don't need to sum to 1 (normalized internally). Shared with the MacApp settings screen (~/.fcp-mcp-server/config.json).",
inputSchema={
"type": "object",
"properties": {
"energy_threshold": {"type": "number", "description": "0-1, how loud (normalized RMS) counts as 'high energy' (default 0.5)"},
"emphasis_weights": {
"type": "object",
"description": "Any subset of {energy, pitch_variation, rate_variation, pause_before, duration} weights for the emphasis index",
"properties": {
"energy": {"type": "number"},
"pitch_variation": {"type": "number"},
"rate_variation": {"type": "number"},
"pause_before": {"type": "number"},
"duration": {"type": "number"},
},
},
"peak_percentile": {"type": "number", "description": "Fraction of words selected as peaks, 0-1 (default 0.02 = top 2%). Selection is relative because the emphasis index's real range depends on the material — measured on a real interview it never passed 0.55."},
"emphasis_floor": {"type": "number", "description": "0-1 minimum emphasis for a peak, guarding genuinely flat audio (default 0.25)"},
"emotion_enabled": {"type": "boolean", "description": "Whether emotion detection runs as part of voice analysis (default false)"},
"emotion_sensitivity": {"type": "number", "description": "0-1 confidence threshold to accept an emotion label (default 0.5)"},
},
}
),
]
async def handle_diarize_media(arguments: dict) -> Sequence[TextContent]:
media_path = _validate_filepath(
arguments["media_path"], AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
)
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
num_speakers = str(arguments.get("num_speakers") or "").strip()
model = arguments.get("model", "base")
language = arguments.get("language")
ok, message = diarization_capability(token)
if not ok:
return _text_result(f"# Speaker Diarization\n\n{message}{_DIARIZATION_INSTALL_HINT}")
transcript, reason = _load_or_transcribe(media_path, model, language)
if transcript is None:
return _text_result(
f"# Speaker Diarization\n\nCould not obtain a transcript to diarize "
f"({reason}).{_TRANSCRIBE_INSTALL_HINT}"
)
tracks = diarize(media_path, token, num_speakers)
if tracks is None:
return _text_result(
"# Speaker Diarization\n\nDiarization failed — check the HuggingFace "
"token has accepted the pyannote/speaker-diarization-3.1 model terms, "
"and that the media file is readable."
)
segments, words = assign_speakers(
transcript.get("segments", []), transcript.get("words", []), tracks
)
speakers = build_speakers(segments)
diarization_data = {
"source": Path(media_path).name,
"speakers": speakers,
"segments": segments,
"words": words,
}
json_path = _validate_output_path(
str(Path(media_path).with_name(Path(media_path).stem + "_diarization.json")),
anchor_dir=str(Path(media_path).parent),
)
with open(json_path, "w") as f:
json.dump(diarization_data, f, indent=2)
result_text = f"""# Speaker Diarization
## Summary
- **Source**: {Path(media_path).name}
- **Speakers Detected**: {len(speakers)}
- **Segments**: {len(segments)}
- **Diarization JSON**: {json_path}
## Speakers
"""
result_text += _markdown_table(
["ID", "Name"], [[s["id"], s["name"]] for s in speakers]
) + "\n"
result_text += (
"\n*Next: `build_voice_timeline` to cross this with acoustic features, "
"or use the segments/words directly for speaker-aware editing.*"
)
return _text_result(result_text)
async def handle_analyze_voice_features(arguments: dict) -> Sequence[TextContent]:
media_path = _validate_filepath(
arguments["media_path"], AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
)
model = arguments.get("model", "base")
language = arguments.get("language")
ok, message = features_capability()
if not ok:
return _text_result(f"# Voice Feature Analysis\n\n{message}{_FEATURES_INSTALL_HINT}")
transcript, reason = _load_or_transcribe(media_path, model, language)
if transcript is None:
return _text_result(
f"# Voice Feature Analysis\n\nCould not obtain a transcript to analyze "
f"({reason}).{_TRANSCRIBE_INSTALL_HINT}"
)
words = transcript.get("words", [])
if not words:
return _text_result("# Voice Feature Analysis\n\nNo words in transcript — nothing to analyze.")
config = load_voice_analysis_config()
weights = EmphasisWeights.from_dict(config["emphasis_weights"])
enriched = enrich_words(
words, extract_pitch(media_path), extract_energy(media_path), weights
)
energy_threshold = config["energy_threshold"]
high_energy_words = [w for w in enriched if w["energy_norm"] >= energy_threshold]
high_emphasis_words = select_peaks(
enriched, config["peak_percentile"], config["emphasis_floor"]
)
features_data = {"source": Path(media_path).name, "config": config, "words": enriched}
json_path = _validate_output_path(
str(Path(media_path).with_name(Path(media_path).stem + "_voice_features.json")),
anchor_dir=str(Path(media_path).parent),
)
with open(json_path, "w") as f:
json.dump(features_data, f, indent=2)
result_text = f"""# Voice Feature Analysis
## Summary
- **Source**: {Path(media_path).name}
- **Words Analyzed**: {len(enriched)}
- **High-Energy Words** (>= {energy_threshold:.2f}): {len(high_energy_words)}
- **Peak Words** (top {config["peak_percentile"]:.1%}): {len(high_emphasis_words)}
- **Features JSON**: {json_path}
## Top Emphasis Words
"""
top = sorted(enriched, key=lambda w: w["emphasis"], reverse=True)[:10]
result_text += _markdown_table(
["Word", "Time", "Emphasis", "Energy", "Pitch Δ"],
[
[
w.get("word", ""),
f"{w.get('start', 0):.2f}s",
f"{w['emphasis']:.2f}",
f"{w['energy_norm']:.2f}",
f"{w['pitch_delta']:.2f}",
]
for w in top
],
) + "\n"
result_text += (
"\n*Thresholds and emphasis weights are configurable in Voice Analysis "
"settings (`save_voice_analysis_config`). Next: `diarize_media` to add "
"speaker labels.*"
)
return _text_result(result_text)
async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
media_path = _validate_filepath(
arguments["media_path"], AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
)
model = arguments.get("model", "base")
language = arguments.get("language")
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
transcript, reason = _load_or_transcribe(media_path, model, language)
if transcript is None:
return _text_result(
f"# Voice Timeline\n\nCould not obtain a transcript "
f"({reason}).{_TRANSCRIBE_INSTALL_HINT}"
)
config = load_voice_analysis_config()
timeline = build_voice_timeline(
media_path,
transcript,
hf_token=token,
num_speakers=num_speakers,
weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
peak_percentile=config["peak_percentile"],
emphasis_floor=config["emphasis_floor"],
)
output_dir = arguments.get("output_dir")
json_path = Path(_validate_output_path(
str(voice_timeline_path(media_path, output_dir)),
anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent),
))
save_voice_timeline(timeline, json_path)
summary = timeline["summary"]
layers = timeline["layers"]
result_text = f"""# Voice Timeline
## Summary
- **Source**: {timeline["source"]}
- **Duration**: {format_duration(summary["duration"])}
- **Speakers**: {summary["speaker_count"]}
- **Segments**: {summary["segment_count"]} ({summary["word_count"]} words)
- **Average Emphasis**: {summary["avg_emphasis"]:.2f}
- **Peak Moments** ({summary["peak_selection"]}): {summary["peak_count"]}
- **Timeline JSON**: {json_path}
## Analysis Layers
"""
result_text += _markdown_table(
["Layer", "Status"],
[
["Transcript", "yes" if layers["transcript"] else "empty"],
[
"Acoustics (pitch/energy)",
"yes" if layers["acoustics"] else "FAILED — every acoustic value is 0",
],
["Speakers", "yes" if layers["speakers"] else "not run — single default speaker"],
],
) + "\n"
if summary["peak_moments"]:
result_text += "\n## Peak Moments\n"
result_text += _markdown_table(
["Time", "Word", "Speaker", "Emphasis"],
[
[
f"{m['time']:.2f}s",
m["text"],
m["speaker"],
f"{m['emphasis']:.2f}",
]
for m in summary["peak_moments"][:10]
],
) + "\n"
result_text += (
"\n*The JSON is layered summary -> segments -> words with normalized "
"0-1 values, ready to hand to a model for edit direction.*"
)
return _text_result(result_text)
async def handle_remove_speakers(arguments: dict) -> Sequence[TextContent]:
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
speaker_ids = arguments.get("speaker_ids") or []
media_path = arguments.get("media_path")
if not media_path:
modifier = FCPXMLModifier(filepath)
for _, clip_el in modifier._iter_spine_clips():
src = modifier.resources.get(clip_el.get("ref", ""), {}).get("src", "")
candidate = media_src_to_path(src)
if candidate and Path(candidate).is_file():
media_path = candidate
break
if not media_path:
return _text_result(
"# Speakers\n\nNo source media found for this timeline — pass `media_path` explicitly."
)
timeline = _read_voice_timeline(media_path, arguments.get("output_dir"))
if timeline is None:
return _text_result(
f"# Speakers\n\nNo voice timeline for `{Path(media_path).name}` yet.\n\n"
"Run `build_voice_timeline` on it first."
)
profiles = timeline.get("speakers", [])
header = f"# Speakers in {timeline.get('source', '')}\n\n" + _speaker_table(profiles) + "\n"
if len(profiles) < 2:
header += (
"\n> Only one speaker is present. Either the recording really has one "
"voice, or diarization did not run — check the Models tab for the "
"HuggingFace token.\n"
)
for p in profiles:
samples = [s for s in p.get("samples", []) if s]
if samples:
header += f"\n**{p['id']}** ({p.get('name', '')}) says things like:\n"
header += "".join(f"> {s}\n" for s in samples[:2])
if not speaker_ids:
return _text_result(
header
+ "\n*Nothing was edited. Re-run with `speaker_ids` naming who to REMOVE — "
"typically the interviewer or crew, keeping the subject.*"
)
known = {p["id"] for p in profiles}
unknown = [s for s in speaker_ids if s not in known]
if unknown:
return _text_result(
header + f"\n**Unknown speaker(s): {', '.join(unknown)}** — nothing was edited."
)
if set(speaker_ids) >= known:
return _text_result(
header + "\n**That would remove every speaker**, leaving nothing — nothing was edited."
)
actions = speaker_cut_actions(
timeline, speaker_ids, padding=float(arguments.get("padding", 0.15))
)
if not actions:
return _text_result(header + "\n No speech found for those speakers — nothing was edited.")
result = await handle_apply_voice_actions({
**arguments,
"actions": [a.as_dict() for a in actions],
})
removed = sum(a.duration for a in actions)
return _text_result(
header
+ f"\n## Removed\n- **Speakers cut**: {', '.join(speaker_ids)}\n"
+ f"- **Speech removed**: {format_duration(removed)} across {len(actions)} segments\n\n"
+ result[0].text
)
def _read_voice_timeline(media_path: str, output_dir: Optional[str] = None) -> Optional[dict]:
"""Load the cached voice timeline, project folder first.
``build_voice_timeline`` writes to the chosen project folder when one is
set and beside the media otherwise, so a reader that only checks one of
the two reports "no voice timeline yet" for a file that exists. Checking
both also keeps timelines built before the project folder existed
readable.
"""
if output_dir:
timeline = load_voice_timeline(voice_timeline_path(media_path, output_dir))
if timeline is not None:
return timeline
return load_voice_timeline(voice_timeline_path(media_path))
async def handle_refine_voice_timeline(arguments: dict) -> Sequence[TextContent]:
media_path = _validate_filepath(
arguments["media_path"], AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
)
timeline = _read_voice_timeline(media_path, arguments.get("output_dir"))
if timeline is None:
return _text_result(
f"# Refined Voice Timeline\n\nNo voice timeline for "
f"`{Path(media_path).name}` yet.\n\nRun `build_voice_timeline` on it first."
)
cut_ranges: List[Tuple[float, float]] = []
rejected: List[str] = []
for i, raw in enumerate(arguments.get("cuts") or []):
try:
start, end = float(raw["start"]), float(raw["end"])
except (TypeError, ValueError, KeyError):
rejected.append(f"cut #{i}: start/end must be numbers")
continue
if end <= start:
rejected.append(f"cut #{i}: end ({end}) must be after start ({start})")
continue
cut_ranges.append((start, end))
config = load_voice_analysis_config()
refined = restrict_to_kept(
timeline,
cut_ranges,
weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
peak_percentile=config["peak_percentile"],
emphasis_floor=config["emphasis_floor"],
)
before, after = timeline["summary"], refined["summary"]
zooms = suggest_zoom_windows(
refined,
min_gap=float(arguments.get("min_gap", 8.0)),
max_zooms=arguments.get("max_zooms"),
)
result = f"""# Refined Voice Timeline
Re-normalized over the material that survives {len(cut_ranges)} cut(s).
"""
result += _markdown_table(
["Measure", "Raw recording", "Survivors only"],
[
["Duration", format_duration(before["duration"]), format_duration(after["duration"])],
["Segments", str(before["segment_count"]), str(after["segment_count"])],
["Words", str(before["word_count"]), str(after["word_count"])],
["Average emphasis", f"{before['avg_emphasis']:.3f}", f"{after['avg_emphasis']:.3f}"],
["Peak moments", str(before["peak_count"]), str(after["peak_count"])],
],
) + "\n"
if after["word_count"] == 0:
result += "\n> The cuts removed every word — nothing left to analyze.\n"
if after["peak_moments"]:
result += "\n## Peak Moments (re-ranked)\n"
result += _markdown_table(
["Time", "Word", "Speaker", "Emphasis"],
[
[f"{m['time']:.2f}s", m["text"], m["speaker"], f"{m['emphasis']:.2f}"]
for m in after["peak_moments"][:10]
],
) + "\n"
if zooms:
result += "\n## Zoom Candidates\n"
result += _markdown_table(
["Start", "End", "Word", "Emphasis", "Line"],
[
[f"{z['start']:.2f}s", f"{z['end']:.2f}s", z["word"],
f"{z['emphasis']:.2f}", z["line"][:60]]
for z in zooms
],
) + "\n"
else:
result += "\n## Zoom Candidates\n\nNone — no content words survived the cuts.\n"
if arguments.get("save"):
p = Path(media_path)
json_path = Path(_validate_output_path(
str(p.with_name(p.stem + "_voice_timeline_refined.json")),
anchor_dir=str(p.parent),
))
save_voice_timeline(refined, json_path)
result += f"\n**Refined JSON**: {json_path}\n"
if rejected:
result += "\n## Rejected cuts\n" + "\n".join(f"- {r}" for r in rejected) + "\n"
result += (
"\n*Candidates, not obligations — cut the list by rhythm. Times are in "
"original source seconds, ready for `apply_voice_actions`.*"
)
return _text_result(result)
async def handle_apply_voice_actions(arguments: dict) -> Sequence[TextContent]:
raw_actions = arguments.get("actions")
if raw_actions is None:
return _text_result("# Voice Actions\n\nNo `actions` provided — nothing to apply.")
actions, errors = parse_actions(raw_actions)
if not actions:
text = "# Voice Actions\n\nNo valid actions to apply."
if errors:
text += "\n\n## Rejected\n" + "\n".join(f"- {e}" for e in errors)
return _text_result(text)
cut_ranges, placed, dropped = resolve_actions(actions)
filepath, output_path, modifier = _setup_modifier(arguments, "_voice_edit")
def source_window(clip_el) -> tuple[float, float]:
"""The span of source media a spine clip actually uses."""
start = modifier.source_file_start(clip_el).to_seconds()
duration = modifier._parse_time(clip_el.get("duration", "0s")).to_seconds()
return start, start + duration
applied: list[list[str]] = []
unplaced: list[str] = []
# Placements go on before cuts: they are anchored in source coordinates,
# and cutting afterwards ripples the spine around them.
# Cuts go FIRST. Cutting splits a clip into pieces and rewrites the
# spine around them, which would duplicate a zoom onto every piece and
# lose markers entirely. Cutting first means placements land on final,
# stable clips — and `resolve_actions` already moved their times onto
# the post-cut timeline, so they still point at the same moment.
cuts_made = 0
for _, clip_el in modifier._iter_spine_clips():
clip_start, clip_end = source_window(clip_el)
to_frame = modifier.snap_seconds_to_frame
ranges = [
(to_frame(max(s, clip_start) - clip_start), to_frame(min(e, clip_end) - clip_start))
for s, e in cut_ranges
if min(e, clip_end) > max(s, clip_start)
]
ranges = [(a, b) for a, b in ranges if b > a]
if ranges and modifier.cut_clip_ranges(clip_el, ranges) > TimeValue.zero():
cuts_made += len(ranges)
def timeline_window(clip_el) -> tuple[float, float]:
"""Where a spine clip sits on the timeline, in seconds."""
offset = modifier._parse_time(clip_el.get("offset", "0s")).to_seconds()
duration = modifier._parse_time(clip_el.get("duration", "0s")).to_seconds()
return offset, offset + duration
for action in placed:
host = next(
(
clip_el
for _, clip_el in modifier._iter_spine_clips()
if timeline_window(clip_el)[0] <= action.start < timeline_window(clip_el)[1]
),
None,
)
if host is None:
unplaced.append(
f"{action.kind} @ {action.start:.2f}s — falls outside the edited timeline"
)
continue
try:
what = _apply_placed_action(modifier, host, action, timeline_window(host)[0])
applied.append([f"{action.start:.2f}s", what, host.get("name", ""), action.reason])
except (ValueError, KeyError) as exc:
unplaced.append(f"{action.kind} @ {action.start:.2f}s — {exc}")
# Rename the project so it does not land in the library indistinguishable
# from the original. FCP imports by the name in the XML, so an untouched
# name puts two same-named projects in the same event — and the edit looks
# like it did nothing, because the original is what gets opened.
project_name = ""
project_el = modifier.root.find(".//project")
if project_el is not None:
project_name = f"{project_el.get('name', 'Projeto')} — corte por voz"
project_el.set("name", project_name)
modifier.save(output_path)
result = f"""# Voice Actions Applied
## Summary
- **Actions received**: {len(actions)}
- **Placed** (zoom/text/marker): {len(applied)}
- **Cuts applied**: {cuts_made}
- **Project name**: {project_name or '(unchanged)'}
- **Saved to**: `{output_path}`
"""
if applied:
result += "## Applied\n" + _markdown_table(
["Time", "Action", "Clip", "Reason"], applied[:40]
) + "\n"
if dropped:
result += "\n## Dropped (pointed into removed material)\n" + "\n".join(
f"- {a.kind} @ {a.start:.2f}s — {a.reason or 'no reason given'}" for a in dropped
) + "\n"
if unplaced:
result += "\n## Not placed\n" + "\n".join(f"- {u}" for u in unplaced) + "\n"
if errors:
result += "\n## Rejected\n" + "\n".join(f"- {e}" for e in errors) + "\n"
result += "\n*Non-destructive: the original file is untouched.*"
return _text_result(result)
async def handle_get_voice_analysis_config(arguments: dict) -> Sequence[TextContent]:
return _text_result(_voice_analysis_config_text(load_voice_analysis_config()))
async def handle_save_voice_analysis_config(arguments: dict) -> Sequence[TextContent]:
config = save_voice_analysis_config(
energy_threshold=arguments.get("energy_threshold"),
emphasis_weights=arguments.get("emphasis_weights"),
peak_percentile=arguments.get("peak_percentile"),
emphasis_floor=arguments.get("emphasis_floor"),
emotion_enabled=arguments.get("emotion_enabled"),
emotion_sensitivity=arguments.get("emotion_sensitivity"),
)
return _text_result(_voice_analysis_config_text(config))
HANDLERS = {
"diarize_media": handle_diarize_media,
"analyze_voice_features": handle_analyze_voice_features,
"build_voice_timeline": handle_build_voice_timeline,
"remove_speakers": handle_remove_speakers,
"refine_voice_timeline": handle_refine_voice_timeline,
"apply_voice_actions": handle_apply_voice_actions,
"get_voice_analysis_config": handle_get_voice_analysis_config,
"save_voice_analysis_config": handle_save_voice_analysis_config,
}