459 lines
19 KiB
Python
459 lines
19 KiB
Python
"""Beats / markers importados — tool schemas and handlers.
|
|
|
|
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
from pathlib import Path
|
|
from typing import Sequence
|
|
|
|
from mcp.types import TextContent, Tool
|
|
|
|
from fcpxml.media_intel import media_src_to_path
|
|
from fcpxml.parser import FCPXMLParser
|
|
from fcpxml.writer import FCPXMLModifier
|
|
from server_tools._shared import (
|
|
_check_json_depth,
|
|
_load_or_transcribe,
|
|
_markdown_table,
|
|
_no_timeline,
|
|
_raw_markers_to_batch,
|
|
_resolve_io_paths,
|
|
_setup_modifier,
|
|
_text_result,
|
|
_validate_filepath,
|
|
format_duration,
|
|
parse_srt,
|
|
parse_transcript_timestamps,
|
|
parse_vtt,
|
|
)
|
|
|
|
TOOLS = [
|
|
Tool(
|
|
name="import_beat_markers",
|
|
description="Import beat markers from external audio analysis (JSON format)",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"beats_path": {"type": "string", "description": "Path to beats JSON file"},
|
|
"marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "standard"},
|
|
"beat_filter": {"type": "string", "enum": ["all", "downbeat", "measure"], "default": "all", "description": "Which beats to import"},
|
|
"output_path": {"type": "string", "description": "Output path (default: adds _beats suffix)"}
|
|
},
|
|
"required": ["filepath", "beats_path"]
|
|
}
|
|
),
|
|
Tool(
|
|
name="snap_to_beats",
|
|
description="Align cuts to nearest beat markers for music-synced edits",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file with beat markers"},
|
|
"max_shift_frames": {"type": "integer", "default": 6, "description": "Maximum frames to shift a cut"},
|
|
"prefer": {"type": "string", "enum": ["earlier", "later", "nearest"], "default": "nearest", "description": "Which beat to prefer when equidistant"},
|
|
"output_path": {"type": "string", "description": "Output path (default: adds _synced suffix)"}
|
|
},
|
|
"required": ["filepath"]
|
|
}
|
|
),
|
|
Tool(
|
|
name="import_srt_markers",
|
|
description="Import SRT or VTT subtitles as chapter markers on the timeline",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"srt_path": {"type": "string", "description": "Path to SRT or VTT subtitle file"},
|
|
"mode": {"type": "string", "enum": ["all", "first_per_minute", "scene_changes"], "default": "first_per_minute", "description": "How to create markers: every subtitle, first per minute, or on text changes"},
|
|
"marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "chapter"},
|
|
"max_label_length": {"type": "integer", "default": 50, "description": "Truncate marker labels to this length"},
|
|
"output_path": {"type": "string", "description": "Output path (default: adds _subtitled suffix)"}
|
|
},
|
|
"required": ["filepath", "srt_path"]
|
|
}
|
|
),
|
|
Tool(
|
|
name="import_transcript_markers",
|
|
description="Import timestamped transcript (YouTube chapter format) as markers. Supports '0:00 Title' and 'HH:MM:SS Title' formats",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"transcript": {"type": "string", "description": "Timestamped text (one per line: '0:00 Introduction')"},
|
|
"transcript_path": {"type": "string", "description": "Path to text file with timestamps (alternative to inline transcript)"},
|
|
"marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "chapter"},
|
|
"output_path": {"type": "string", "description": "Output path (default: adds _chapters suffix)"}
|
|
},
|
|
"required": ["filepath"]
|
|
}
|
|
),
|
|
Tool(
|
|
name="transcript_markers",
|
|
description="Add a marker at the start of every transcribed segment (sentence-level), using each media file's local Whisper transcript. Maps each segment's source-media timestamp to its correct timeline position per clip, so it stays accurate across multiple clips/trims — unlike import_transcript_markers (plain timestamp text) or import_srt_markers (a caption track already synced to the whole export). Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _transcript_markers copy.",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"clip_name": {"type": "string", "description": "Only mark the clip with this name"},
|
|
"marker_type": {"type": "string", "default": "chapter", "description": "Marker type: standard, chapter, todo, completed"},
|
|
"max_label_length": {"type": "integer", "default": 50, "description": "Truncate marker labels to this many characters (0 = no truncation)"},
|
|
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
|
"output_path": {"type": "string", "description": "Output path (default: adds _transcript_markers suffix)"},
|
|
},
|
|
"required": ["filepath"]
|
|
}
|
|
),
|
|
]
|
|
|
|
|
|
async def handle_import_beat_markers(arguments: dict) -> Sequence[TextContent]:
|
|
filepath, output_path = _resolve_io_paths(arguments, "_beats")
|
|
beats_path = _validate_filepath(arguments["beats_path"], ('.json',))
|
|
|
|
with open(beats_path, 'r') as f:
|
|
beats_data = json.load(f)
|
|
_check_json_depth(beats_data)
|
|
|
|
beat_times = []
|
|
if isinstance(beats_data, list):
|
|
beat_times = beats_data
|
|
elif isinstance(beats_data, dict):
|
|
beat_times = beats_data.get('beats', beats_data.get('times', beats_data.get('markers', [])))
|
|
|
|
beat_filter = arguments.get("beat_filter", "all")
|
|
if beat_filter == "downbeat" and isinstance(beats_data, dict):
|
|
beat_times = beats_data.get('downbeats', beat_times[::4])
|
|
elif beat_filter == "measure" and isinstance(beats_data, dict):
|
|
beat_times = beats_data.get('measures', beat_times[::4])
|
|
|
|
markers = []
|
|
marker_type = arguments.get("marker_type", "standard")
|
|
for i, beat_time in enumerate(beat_times):
|
|
if isinstance(beat_time, (int, float)):
|
|
markers.append({
|
|
'timecode': f"{beat_time}s",
|
|
'name': f"Beat {i+1}",
|
|
'marker_type': marker_type.upper(),
|
|
})
|
|
elif isinstance(beat_time, dict):
|
|
markers.append({
|
|
'timecode': f"{beat_time.get('time', beat_time.get('position', 0))}s",
|
|
'name': beat_time.get('label', f"Beat {i+1}"),
|
|
'marker_type': marker_type.upper(),
|
|
})
|
|
|
|
modifier = FCPXMLModifier(filepath)
|
|
|
|
# Songs routinely run longer than the edit — beats past the timeline's
|
|
# end are skipped (add_marker_at_timeline would raise on them).
|
|
timeline_end = modifier._timeline_duration().to_seconds()
|
|
in_range = [m for m in markers if float(m['timecode'].rstrip('s')) < timeline_end]
|
|
skipped_count = len(markers) - len(in_range)
|
|
|
|
added = modifier.batch_add_markers(markers=in_range)
|
|
modifier.save(output_path)
|
|
|
|
skipped_note = (
|
|
f"- **Skipped**: {skipped_count} beat(s) beyond the timeline end "
|
|
f"({format_duration(timeline_end)})\n" if skipped_count else ""
|
|
)
|
|
return _text_result(f"""# Beat Markers Imported
|
|
|
|
## Summary
|
|
- **Beats Found**: {len(beat_times)}
|
|
- **Markers Added**: {len(added)}
|
|
{skipped_note}- **Filter**: {beat_filter}
|
|
- **Marker Type**: {marker_type}
|
|
|
|
## Output
|
|
Saved to: `{output_path}`
|
|
|
|
*Use `snap_to_beats` to align your cuts to these markers.*
|
|
""")
|
|
|
|
|
|
async def handle_snap_to_beats(arguments: dict) -> Sequence[TextContent]:
|
|
filepath, output_path = _resolve_io_paths(arguments, "_synced")
|
|
max_shift = arguments.get("max_shift_frames", 6)
|
|
prefer = arguments.get("prefer", "nearest")
|
|
|
|
parser = FCPXMLParser()
|
|
project = parser.parse_file(filepath)
|
|
if not project.timelines:
|
|
return _no_timeline()
|
|
|
|
tl = project.primary_timeline
|
|
fps = tl.frame_rate
|
|
|
|
markers = list(tl.markers)
|
|
for clip in tl.clips:
|
|
markers.extend(clip.markers)
|
|
|
|
if not markers:
|
|
return _text_result("No markers found. Use `import_beat_markers` first.")
|
|
|
|
marker_times = sorted([m.start.seconds for m in markers])
|
|
|
|
modifier = FCPXMLModifier(filepath)
|
|
spine = modifier._get_spine()
|
|
adjusted_count = 0
|
|
total_shift = 0
|
|
|
|
clips_list = [c for c in spine if c.tag in ('clip', 'asset-clip', 'video', 'ref-clip')]
|
|
|
|
for i, clip in enumerate(clips_list[1:], 1):
|
|
cut_offset = modifier._parse_time(clip.get('offset', '0s'))
|
|
cut_seconds = cut_offset.to_seconds()
|
|
|
|
best_marker = None
|
|
best_distance = float('inf')
|
|
|
|
for marker_time in marker_times:
|
|
distance = abs(marker_time - cut_seconds)
|
|
distance_frames = distance * fps
|
|
|
|
if distance_frames <= max_shift:
|
|
if prefer == "earlier" and marker_time <= cut_seconds:
|
|
if distance < best_distance:
|
|
best_distance = distance
|
|
best_marker = marker_time
|
|
elif prefer == "later" and marker_time >= cut_seconds:
|
|
if distance < best_distance:
|
|
best_distance = distance
|
|
best_marker = marker_time
|
|
elif prefer == "nearest":
|
|
if distance < best_distance:
|
|
best_distance = distance
|
|
best_marker = marker_time
|
|
|
|
if best_marker is not None and best_distance > 0.001:
|
|
shift = best_marker - cut_seconds
|
|
shift_frames = int(shift * fps)
|
|
|
|
prev_clip = clips_list[i - 1]
|
|
prev_dur = modifier._parse_time(prev_clip.get('duration', '0s'))
|
|
new_prev_dur = prev_dur + modifier._parse_time(f"{shift}s")
|
|
prev_clip.set('duration', new_prev_dur.to_fcpxml())
|
|
|
|
new_offset = modifier._parse_time(f"{best_marker}s")
|
|
clip.set('offset', new_offset.to_fcpxml())
|
|
|
|
adjusted_count += 1
|
|
total_shift += abs(shift_frames)
|
|
|
|
modifier.save(output_path)
|
|
avg_shift = total_shift / adjusted_count if adjusted_count > 0 else 0
|
|
|
|
return _text_result(f"""# Cuts Snapped to Beats
|
|
|
|
## Summary
|
|
- **Cuts Adjusted**: {adjusted_count}
|
|
- **Max Shift Allowed**: {max_shift} frames
|
|
- **Preference**: {prefer}
|
|
- **Average Shift**: {avg_shift:.1f} frames
|
|
|
|
## Output
|
|
Saved to: `{output_path}`
|
|
|
|
Your edits are now synced to the beat!
|
|
""")
|
|
|
|
|
|
async def handle_import_srt_markers(arguments: dict) -> Sequence[TextContent]:
|
|
filepath, output_path = _resolve_io_paths(arguments, "_subtitled")
|
|
srt_path = _validate_filepath(arguments["srt_path"], ('.srt', '.vtt'))
|
|
mode = arguments.get("mode", "first_per_minute")
|
|
marker_type = arguments.get("marker_type", "chapter")
|
|
max_label = arguments.get("max_label_length", 50)
|
|
|
|
text = Path(srt_path).read_text(encoding='utf-8')
|
|
|
|
# Detect format and parse
|
|
if srt_path.endswith('.vtt') or text.strip().startswith('WEBVTT'):
|
|
raw_markers = parse_vtt(text)
|
|
fmt_name = "WebVTT"
|
|
else:
|
|
raw_markers = parse_srt(text)
|
|
fmt_name = "SRT"
|
|
|
|
if not raw_markers:
|
|
return _text_result(f"No subtitles found in {srt_path}")
|
|
|
|
# Apply mode filtering
|
|
filtered = []
|
|
if mode == "all":
|
|
filtered = raw_markers
|
|
elif mode == "first_per_minute":
|
|
seen_minutes = set()
|
|
for m in raw_markers:
|
|
minute = int(m['seconds'] // 60)
|
|
if minute not in seen_minutes:
|
|
seen_minutes.add(minute)
|
|
filtered.append(m)
|
|
elif mode == "scene_changes":
|
|
# Group by similar text, take first occurrence of each unique line
|
|
seen_texts = set()
|
|
for m in raw_markers:
|
|
# Normalize: lowercase, strip punctuation
|
|
normalized = re.sub(r'[^\w\s]', '', m['text'].lower()).strip()
|
|
words = normalized.split()[:3] # First 3 words as key
|
|
key = ' '.join(words)
|
|
if key and key not in seen_texts:
|
|
seen_texts.add(key)
|
|
filtered.append(m)
|
|
|
|
markers = _raw_markers_to_batch(filtered, marker_type, max_label=max_label)
|
|
|
|
modifier = FCPXMLModifier(filepath)
|
|
added = modifier.batch_add_markers(markers=markers)
|
|
modifier.save(output_path)
|
|
|
|
return _text_result(f"""# Subtitle Markers Imported
|
|
|
|
## Summary
|
|
- **Format**: {fmt_name}
|
|
- **Subtitles Parsed**: {len(raw_markers)}
|
|
- **Mode**: {mode}
|
|
- **Markers Added**: {len(added)}
|
|
- **Marker Type**: {marker_type}
|
|
|
|
## Output
|
|
Saved to: `{output_path}`
|
|
""")
|
|
|
|
|
|
async def handle_import_transcript_markers(arguments: dict) -> Sequence[TextContent]:
|
|
filepath, output_path = _resolve_io_paths(arguments, "_chapters")
|
|
marker_type = arguments.get("marker_type", "chapter")
|
|
|
|
# Get transcript text from inline or file
|
|
transcript = arguments.get("transcript")
|
|
transcript_path = arguments.get("transcript_path")
|
|
|
|
if not transcript and not transcript_path:
|
|
return _text_result("Provide either 'transcript' (inline text) or 'transcript_path' (path to file)")
|
|
|
|
if transcript_path:
|
|
# .txt only: the parser below understands "0:00 Title" lines, not real
|
|
# SRT/VTT cue syntax — that's import_srt_markers (parse_srt/parse_vtt).
|
|
transcript_path = _validate_filepath(transcript_path, ('.txt',))
|
|
transcript = Path(transcript_path).read_text(encoding='utf-8')
|
|
|
|
raw_markers = parse_transcript_timestamps(transcript or "")
|
|
|
|
if not raw_markers:
|
|
return _text_result("No timestamps found. Expected format: '0:00 Title' or 'HH:MM:SS Title', one per line.")
|
|
|
|
markers = _raw_markers_to_batch(raw_markers, marker_type)
|
|
|
|
modifier = FCPXMLModifier(filepath)
|
|
added = modifier.batch_add_markers(markers=markers)
|
|
modifier.save(output_path)
|
|
|
|
return _text_result(f"""# Transcript Markers Imported
|
|
|
|
## Summary
|
|
- **Timestamps Found**: {len(raw_markers)}
|
|
- **Markers Added**: {len(added)}
|
|
- **Marker Type**: {marker_type}
|
|
|
|
## Markers
|
|
""" + "\n".join(f"- `{m['timecode']}` {m['name']}" for m in markers) + f"""
|
|
|
|
## Output
|
|
Saved to: `{output_path}`
|
|
""")
|
|
|
|
|
|
async def handle_transcript_markers(arguments: dict) -> Sequence[TextContent]:
|
|
"""Add a marker at the start of each transcribed segment, using each
|
|
media's cached (or freshly transcribed) local Whisper transcript.
|
|
|
|
Unlike ``import_transcript_markers`` (plain "0:00 Title" text) or
|
|
``import_srt_markers`` (a caption track already synced to the whole
|
|
exported video), this maps each segment's SOURCE-media timestamp to its
|
|
TIMELINE position per spine clip — the same source->timeline mapping
|
|
``detect_media_silence`` uses — so it stays correct across multiple
|
|
clips built from different (and differently-trimmed) source files.
|
|
"""
|
|
marker_type = arguments.get("marker_type", "chapter")
|
|
max_label = int(arguments.get("max_label_length", 50))
|
|
model = arguments.get("model", "base")
|
|
language = arguments.get("language")
|
|
output_dir = arguments.get("output_dir")
|
|
clip_filter = arguments.get("clip_name")
|
|
|
|
filepath, output_path, modifier = _setup_modifier(arguments, "_transcript_markers")
|
|
|
|
added: list[tuple[str, float, str]] = []
|
|
skipped: list[tuple[str, str]] = []
|
|
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
|
for el in spine_clips:
|
|
name = el.get("name", "")
|
|
if clip_filter and name != clip_filter:
|
|
continue
|
|
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
|
media_path = media_src_to_path(src)
|
|
if not media_path or not Path(media_path).is_file():
|
|
skipped.append((name, "media file missing"))
|
|
continue
|
|
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
|
if data is None:
|
|
skipped.append((name, reason))
|
|
continue
|
|
|
|
clip_source_start = modifier.source_file_start(el).to_seconds()
|
|
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
|
clip_offset = modifier._parse_time(el.get("offset", "0s")).to_seconds()
|
|
window_end = clip_source_start + clip_duration
|
|
|
|
for seg in data.get("segments", []):
|
|
seg_start = float(seg.get("start", 0.0))
|
|
if seg_start < clip_source_start or seg_start >= window_end:
|
|
continue
|
|
label = seg.get("text", "").strip()
|
|
if not label:
|
|
continue
|
|
if max_label and len(label) > max_label:
|
|
label = label[:max_label]
|
|
timeline_seconds = clip_offset + (seg_start - clip_source_start)
|
|
modifier.add_marker_at_timeline(
|
|
timecode=f"{timeline_seconds}s", name=label, marker_type=marker_type,
|
|
)
|
|
added.append((name, seg_start, label))
|
|
|
|
if not added:
|
|
text = "# Transcript Markers\n\nNo segments to mark — file unchanged (nothing saved)."
|
|
if skipped:
|
|
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
|
)
|
|
return _text_result(text)
|
|
|
|
modifier.save(output_path)
|
|
result = "# Transcript Markers Imported (local Whisper)\n\n## Summary\n"
|
|
result += f"- **Markers Added**: {len(added)}\n- **Marker Type**: {marker_type}\n\n"
|
|
result += _markdown_table(
|
|
["Clip", "Start", "Label"], [[n, f"{s:.2f}s", label] for n, s, label in added]
|
|
)
|
|
if skipped:
|
|
result += "\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
|
)
|
|
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
|
|
return _text_result(result)
|
|
|
|
|
|
HANDLERS = {
|
|
"import_beat_markers": handle_import_beat_markers,
|
|
"snap_to_beats": handle_snap_to_beats,
|
|
"import_srt_markers": handle_import_srt_markers,
|
|
"import_transcript_markers": handle_import_transcript_markers,
|
|
"transcript_markers": handle_transcript_markers,
|
|
}
|