"""Beats / markers importados — tool schemas and handlers. Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog. """ from __future__ import annotations import json import re from pathlib import Path from typing import Sequence from mcp.types import TextContent, Tool from fcpxml.media_intel import media_src_to_path from fcpxml.parser import FCPXMLParser from fcpxml.writer import FCPXMLModifier from server_tools._shared import ( _check_json_depth, _load_or_transcribe, _markdown_table, _no_timeline, _raw_markers_to_batch, _resolve_io_paths, _setup_modifier, _text_result, _validate_filepath, format_duration, parse_srt, parse_transcript_timestamps, parse_vtt, ) TOOLS = [ Tool( name="import_beat_markers", description="Import beat markers from external audio analysis (JSON format)", inputSchema={ "type": "object", "properties": { "filepath": {"type": "string", "description": "Path to FCPXML file"}, "beats_path": {"type": "string", "description": "Path to beats JSON file"}, "marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "standard"}, "beat_filter": {"type": "string", "enum": ["all", "downbeat", "measure"], "default": "all", "description": "Which beats to import"}, "output_path": {"type": "string", "description": "Output path (default: adds _beats suffix)"} }, "required": ["filepath", "beats_path"] } ), Tool( name="snap_to_beats", description="Align cuts to nearest beat markers for music-synced edits", inputSchema={ "type": "object", "properties": { "filepath": {"type": "string", "description": "Path to FCPXML file with beat markers"}, "max_shift_frames": {"type": "integer", "default": 6, "description": "Maximum frames to shift a cut"}, "prefer": {"type": "string", "enum": ["earlier", "later", "nearest"], "default": "nearest", "description": "Which beat to prefer when equidistant"}, "output_path": {"type": "string", "description": "Output path (default: adds _synced suffix)"} }, "required": ["filepath"] } ), Tool( name="import_srt_markers", description="Import SRT or VTT subtitles as chapter markers on the timeline", inputSchema={ "type": "object", "properties": { "filepath": {"type": "string", "description": "Path to FCPXML file"}, "srt_path": {"type": "string", "description": "Path to SRT or VTT subtitle file"}, "mode": {"type": "string", "enum": ["all", "first_per_minute", "scene_changes"], "default": "first_per_minute", "description": "How to create markers: every subtitle, first per minute, or on text changes"}, "marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "chapter"}, "max_label_length": {"type": "integer", "default": 50, "description": "Truncate marker labels to this length"}, "output_path": {"type": "string", "description": "Output path (default: adds _subtitled suffix)"} }, "required": ["filepath", "srt_path"] } ), Tool( name="import_transcript_markers", description="Import timestamped transcript (YouTube chapter format) as markers. Supports '0:00 Title' and 'HH:MM:SS Title' formats", inputSchema={ "type": "object", "properties": { "filepath": {"type": "string", "description": "Path to FCPXML file"}, "transcript": {"type": "string", "description": "Timestamped text (one per line: '0:00 Introduction')"}, "transcript_path": {"type": "string", "description": "Path to text file with timestamps (alternative to inline transcript)"}, "marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "chapter"}, "output_path": {"type": "string", "description": "Output path (default: adds _chapters suffix)"} }, "required": ["filepath"] } ), Tool( name="transcript_markers", description="Add a marker at the start of every transcribed segment (sentence-level), using each media file's local Whisper transcript. Maps each segment's source-media timestamp to its correct timeline position per clip, so it stays accurate across multiple clips/trims — unlike import_transcript_markers (plain timestamp text) or import_srt_markers (a caption track already synced to the whole export). Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _transcript_markers copy.", inputSchema={ "type": "object", "properties": { "filepath": {"type": "string", "description": "Path to FCPXML file"}, "clip_name": {"type": "string", "description": "Only mark the clip with this name"}, "marker_type": {"type": "string", "default": "chapter", "description": "Marker type: standard, chapter, todo, completed"}, "max_label_length": {"type": "integer", "default": 50, "description": "Truncate marker labels to this many characters (0 = no truncation)"}, "model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"}, "output_path": {"type": "string", "description": "Output path (default: adds _transcript_markers suffix)"}, }, "required": ["filepath"] } ), ] async def handle_import_beat_markers(arguments: dict) -> Sequence[TextContent]: filepath, output_path = _resolve_io_paths(arguments, "_beats") beats_path = _validate_filepath(arguments["beats_path"], ('.json',)) with open(beats_path, 'r') as f: beats_data = json.load(f) _check_json_depth(beats_data) beat_times = [] if isinstance(beats_data, list): beat_times = beats_data elif isinstance(beats_data, dict): beat_times = beats_data.get('beats', beats_data.get('times', beats_data.get('markers', []))) beat_filter = arguments.get("beat_filter", "all") if beat_filter == "downbeat" and isinstance(beats_data, dict): beat_times = beats_data.get('downbeats', beat_times[::4]) elif beat_filter == "measure" and isinstance(beats_data, dict): beat_times = beats_data.get('measures', beat_times[::4]) markers = [] marker_type = arguments.get("marker_type", "standard") for i, beat_time in enumerate(beat_times): if isinstance(beat_time, (int, float)): markers.append({ 'timecode': f"{beat_time}s", 'name': f"Beat {i+1}", 'marker_type': marker_type.upper(), }) elif isinstance(beat_time, dict): markers.append({ 'timecode': f"{beat_time.get('time', beat_time.get('position', 0))}s", 'name': beat_time.get('label', f"Beat {i+1}"), 'marker_type': marker_type.upper(), }) modifier = FCPXMLModifier(filepath) # Songs routinely run longer than the edit — beats past the timeline's # end are skipped (add_marker_at_timeline would raise on them). timeline_end = modifier._timeline_duration().to_seconds() in_range = [m for m in markers if float(m['timecode'].rstrip('s')) < timeline_end] skipped_count = len(markers) - len(in_range) added = modifier.batch_add_markers(markers=in_range) modifier.save(output_path) skipped_note = ( f"- **Skipped**: {skipped_count} beat(s) beyond the timeline end " f"({format_duration(timeline_end)})\n" if skipped_count else "" ) return _text_result(f"""# Beat Markers Imported ## Summary - **Beats Found**: {len(beat_times)} - **Markers Added**: {len(added)} {skipped_note}- **Filter**: {beat_filter} - **Marker Type**: {marker_type} ## Output Saved to: `{output_path}` *Use `snap_to_beats` to align your cuts to these markers.* """) async def handle_snap_to_beats(arguments: dict) -> Sequence[TextContent]: filepath, output_path = _resolve_io_paths(arguments, "_synced") max_shift = arguments.get("max_shift_frames", 6) prefer = arguments.get("prefer", "nearest") parser = FCPXMLParser() project = parser.parse_file(filepath) if not project.timelines: return _no_timeline() tl = project.primary_timeline fps = tl.frame_rate markers = list(tl.markers) for clip in tl.clips: markers.extend(clip.markers) if not markers: return _text_result("No markers found. Use `import_beat_markers` first.") marker_times = sorted([m.start.seconds for m in markers]) modifier = FCPXMLModifier(filepath) spine = modifier._get_spine() adjusted_count = 0 total_shift = 0 clips_list = [c for c in spine if c.tag in ('clip', 'asset-clip', 'video', 'ref-clip')] for i, clip in enumerate(clips_list[1:], 1): cut_offset = modifier._parse_time(clip.get('offset', '0s')) cut_seconds = cut_offset.to_seconds() best_marker = None best_distance = float('inf') for marker_time in marker_times: distance = abs(marker_time - cut_seconds) distance_frames = distance * fps if distance_frames <= max_shift: if prefer == "earlier" and marker_time <= cut_seconds: if distance < best_distance: best_distance = distance best_marker = marker_time elif prefer == "later" and marker_time >= cut_seconds: if distance < best_distance: best_distance = distance best_marker = marker_time elif prefer == "nearest": if distance < best_distance: best_distance = distance best_marker = marker_time if best_marker is not None and best_distance > 0.001: shift = best_marker - cut_seconds shift_frames = int(shift * fps) prev_clip = clips_list[i - 1] prev_dur = modifier._parse_time(prev_clip.get('duration', '0s')) new_prev_dur = prev_dur + modifier._parse_time(f"{shift}s") prev_clip.set('duration', new_prev_dur.to_fcpxml()) new_offset = modifier._parse_time(f"{best_marker}s") clip.set('offset', new_offset.to_fcpxml()) adjusted_count += 1 total_shift += abs(shift_frames) modifier.save(output_path) avg_shift = total_shift / adjusted_count if adjusted_count > 0 else 0 return _text_result(f"""# Cuts Snapped to Beats ## Summary - **Cuts Adjusted**: {adjusted_count} - **Max Shift Allowed**: {max_shift} frames - **Preference**: {prefer} - **Average Shift**: {avg_shift:.1f} frames ## Output Saved to: `{output_path}` Your edits are now synced to the beat! """) async def handle_import_srt_markers(arguments: dict) -> Sequence[TextContent]: filepath, output_path = _resolve_io_paths(arguments, "_subtitled") srt_path = _validate_filepath(arguments["srt_path"], ('.srt', '.vtt')) mode = arguments.get("mode", "first_per_minute") marker_type = arguments.get("marker_type", "chapter") max_label = arguments.get("max_label_length", 50) text = Path(srt_path).read_text(encoding='utf-8') # Detect format and parse if srt_path.endswith('.vtt') or text.strip().startswith('WEBVTT'): raw_markers = parse_vtt(text) fmt_name = "WebVTT" else: raw_markers = parse_srt(text) fmt_name = "SRT" if not raw_markers: return _text_result(f"No subtitles found in {srt_path}") # Apply mode filtering filtered = [] if mode == "all": filtered = raw_markers elif mode == "first_per_minute": seen_minutes = set() for m in raw_markers: minute = int(m['seconds'] // 60) if minute not in seen_minutes: seen_minutes.add(minute) filtered.append(m) elif mode == "scene_changes": # Group by similar text, take first occurrence of each unique line seen_texts = set() for m in raw_markers: # Normalize: lowercase, strip punctuation normalized = re.sub(r'[^\w\s]', '', m['text'].lower()).strip() words = normalized.split()[:3] # First 3 words as key key = ' '.join(words) if key and key not in seen_texts: seen_texts.add(key) filtered.append(m) markers = _raw_markers_to_batch(filtered, marker_type, max_label=max_label) modifier = FCPXMLModifier(filepath) added = modifier.batch_add_markers(markers=markers) modifier.save(output_path) return _text_result(f"""# Subtitle Markers Imported ## Summary - **Format**: {fmt_name} - **Subtitles Parsed**: {len(raw_markers)} - **Mode**: {mode} - **Markers Added**: {len(added)} - **Marker Type**: {marker_type} ## Output Saved to: `{output_path}` """) async def handle_import_transcript_markers(arguments: dict) -> Sequence[TextContent]: filepath, output_path = _resolve_io_paths(arguments, "_chapters") marker_type = arguments.get("marker_type", "chapter") # Get transcript text from inline or file transcript = arguments.get("transcript") transcript_path = arguments.get("transcript_path") if not transcript and not transcript_path: return _text_result("Provide either 'transcript' (inline text) or 'transcript_path' (path to file)") if transcript_path: # .txt only: the parser below understands "0:00 Title" lines, not real # SRT/VTT cue syntax — that's import_srt_markers (parse_srt/parse_vtt). transcript_path = _validate_filepath(transcript_path, ('.txt',)) transcript = Path(transcript_path).read_text(encoding='utf-8') raw_markers = parse_transcript_timestamps(transcript or "") if not raw_markers: return _text_result("No timestamps found. Expected format: '0:00 Title' or 'HH:MM:SS Title', one per line.") markers = _raw_markers_to_batch(raw_markers, marker_type) modifier = FCPXMLModifier(filepath) added = modifier.batch_add_markers(markers=markers) modifier.save(output_path) return _text_result(f"""# Transcript Markers Imported ## Summary - **Timestamps Found**: {len(raw_markers)} - **Markers Added**: {len(added)} - **Marker Type**: {marker_type} ## Markers """ + "\n".join(f"- `{m['timecode']}` {m['name']}" for m in markers) + f""" ## Output Saved to: `{output_path}` """) async def handle_transcript_markers(arguments: dict) -> Sequence[TextContent]: """Add a marker at the start of each transcribed segment, using each media's cached (or freshly transcribed) local Whisper transcript. Unlike ``import_transcript_markers`` (plain "0:00 Title" text) or ``import_srt_markers`` (a caption track already synced to the whole exported video), this maps each segment's SOURCE-media timestamp to its TIMELINE position per spine clip — the same source->timeline mapping ``detect_media_silence`` uses — so it stays correct across multiple clips built from different (and differently-trimmed) source files. """ marker_type = arguments.get("marker_type", "chapter") max_label = int(arguments.get("max_label_length", 50)) model = arguments.get("model", "base") language = arguments.get("language") output_dir = arguments.get("output_dir") clip_filter = arguments.get("clip_name") filepath, output_path, modifier = _setup_modifier(arguments, "_transcript_markers") added: list[tuple[str, float, str]] = [] skipped: list[tuple[str, str]] = [] spine_clips = [el for _, el in modifier._iter_spine_clips()] for el in spine_clips: name = el.get("name", "") if clip_filter and name != clip_filter: continue src = modifier.resources.get(el.get("ref", ""), {}).get("src", "") media_path = media_src_to_path(src) if not media_path or not Path(media_path).is_file(): skipped.append((name, "media file missing")) continue data, reason = _load_or_transcribe(media_path, model, language, output_dir) if data is None: skipped.append((name, reason)) continue clip_source_start = modifier.source_file_start(el).to_seconds() clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() clip_offset = modifier._parse_time(el.get("offset", "0s")).to_seconds() window_end = clip_source_start + clip_duration for seg in data.get("segments", []): seg_start = float(seg.get("start", 0.0)) if seg_start < clip_source_start or seg_start >= window_end: continue label = seg.get("text", "").strip() if not label: continue if max_label and len(label) > max_label: label = label[:max_label] timeline_seconds = clip_offset + (seg_start - clip_source_start) modifier.add_marker_at_timeline( timecode=f"{timeline_seconds}s", name=label, marker_type=marker_type, ) added.append((name, seg_start, label)) if not added: text = "# Transcript Markers\n\nNo segments to mark — file unchanged (nothing saved)." if skipped: text += "\n\n## Skipped Clips\n" + _markdown_table( ["Clip", "Reason"], [[n, r] for n, r in skipped] ) return _text_result(text) modifier.save(output_path) result = "# Transcript Markers Imported (local Whisper)\n\n## Summary\n" result += f"- **Markers Added**: {len(added)}\n- **Marker Type**: {marker_type}\n\n" result += _markdown_table( ["Clip", "Start", "Label"], [[n, f"{s:.2f}s", label] for n, s, label in added] ) if skipped: result += "\n## Skipped Clips\n" + _markdown_table( ["Clip", "Reason"], [[n, r] for n, r in skipped] ) result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*" return _text_result(result) HANDLERS = { "import_beat_markers": handle_import_beat_markers, "snap_to_beats": handle_snap_to_beats, "import_srt_markers": handle_import_srt_markers, "import_transcript_markers": handle_import_transcript_markers, "transcript_markers": handle_transcript_markers, }