Files
gart/code/server_tools/markers_import.py
T

459 lines
19 KiB
Python

"""Beats / markers importados — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
import json
import re
from pathlib import Path
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.media_intel import media_src_to_path
from fcpxml.parser import FCPXMLParser
from fcpxml.writer import FCPXMLModifier
from server_tools._shared import (
_check_json_depth,
_load_or_transcribe,
_markdown_table,
_no_timeline,
_raw_markers_to_batch,
_resolve_io_paths,
_setup_modifier,
_text_result,
_validate_filepath,
format_duration,
parse_srt,
parse_transcript_timestamps,
parse_vtt,
)
TOOLS = [
Tool(
name="import_beat_markers",
description="Import beat markers from external audio analysis (JSON format)",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"beats_path": {"type": "string", "description": "Path to beats JSON file"},
"marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "standard"},
"beat_filter": {"type": "string", "enum": ["all", "downbeat", "measure"], "default": "all", "description": "Which beats to import"},
"output_path": {"type": "string", "description": "Output path (default: adds _beats suffix)"}
},
"required": ["filepath", "beats_path"]
}
),
Tool(
name="snap_to_beats",
description="Align cuts to nearest beat markers for music-synced edits",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file with beat markers"},
"max_shift_frames": {"type": "integer", "default": 6, "description": "Maximum frames to shift a cut"},
"prefer": {"type": "string", "enum": ["earlier", "later", "nearest"], "default": "nearest", "description": "Which beat to prefer when equidistant"},
"output_path": {"type": "string", "description": "Output path (default: adds _synced suffix)"}
},
"required": ["filepath"]
}
),
Tool(
name="import_srt_markers",
description="Import SRT or VTT subtitles as chapter markers on the timeline",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"srt_path": {"type": "string", "description": "Path to SRT or VTT subtitle file"},
"mode": {"type": "string", "enum": ["all", "first_per_minute", "scene_changes"], "default": "first_per_minute", "description": "How to create markers: every subtitle, first per minute, or on text changes"},
"marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "chapter"},
"max_label_length": {"type": "integer", "default": 50, "description": "Truncate marker labels to this length"},
"output_path": {"type": "string", "description": "Output path (default: adds _subtitled suffix)"}
},
"required": ["filepath", "srt_path"]
}
),
Tool(
name="import_transcript_markers",
description="Import timestamped transcript (YouTube chapter format) as markers. Supports '0:00 Title' and 'HH:MM:SS Title' formats",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"transcript": {"type": "string", "description": "Timestamped text (one per line: '0:00 Introduction')"},
"transcript_path": {"type": "string", "description": "Path to text file with timestamps (alternative to inline transcript)"},
"marker_type": {"type": "string", "enum": ["standard", "chapter"], "default": "chapter"},
"output_path": {"type": "string", "description": "Output path (default: adds _chapters suffix)"}
},
"required": ["filepath"]
}
),
Tool(
name="transcript_markers",
description="Add a marker at the start of every transcribed segment (sentence-level), using each media file's local Whisper transcript. Maps each segment's source-media timestamp to its correct timeline position per clip, so it stays accurate across multiple clips/trims — unlike import_transcript_markers (plain timestamp text) or import_srt_markers (a caption track already synced to the whole export). Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _transcript_markers copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"clip_name": {"type": "string", "description": "Only mark the clip with this name"},
"marker_type": {"type": "string", "default": "chapter", "description": "Marker type: standard, chapter, todo, completed"},
"max_label_length": {"type": "integer", "default": 50, "description": "Truncate marker labels to this many characters (0 = no truncation)"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"output_path": {"type": "string", "description": "Output path (default: adds _transcript_markers suffix)"},
},
"required": ["filepath"]
}
),
]
async def handle_import_beat_markers(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_beats")
beats_path = _validate_filepath(arguments["beats_path"], ('.json',))
with open(beats_path, 'r') as f:
beats_data = json.load(f)
_check_json_depth(beats_data)
beat_times = []
if isinstance(beats_data, list):
beat_times = beats_data
elif isinstance(beats_data, dict):
beat_times = beats_data.get('beats', beats_data.get('times', beats_data.get('markers', [])))
beat_filter = arguments.get("beat_filter", "all")
if beat_filter == "downbeat" and isinstance(beats_data, dict):
beat_times = beats_data.get('downbeats', beat_times[::4])
elif beat_filter == "measure" and isinstance(beats_data, dict):
beat_times = beats_data.get('measures', beat_times[::4])
markers = []
marker_type = arguments.get("marker_type", "standard")
for i, beat_time in enumerate(beat_times):
if isinstance(beat_time, (int, float)):
markers.append({
'timecode': f"{beat_time}s",
'name': f"Beat {i+1}",
'marker_type': marker_type.upper(),
})
elif isinstance(beat_time, dict):
markers.append({
'timecode': f"{beat_time.get('time', beat_time.get('position', 0))}s",
'name': beat_time.get('label', f"Beat {i+1}"),
'marker_type': marker_type.upper(),
})
modifier = FCPXMLModifier(filepath)
# Songs routinely run longer than the edit — beats past the timeline's
# end are skipped (add_marker_at_timeline would raise on them).
timeline_end = modifier._timeline_duration().to_seconds()
in_range = [m for m in markers if float(m['timecode'].rstrip('s')) < timeline_end]
skipped_count = len(markers) - len(in_range)
added = modifier.batch_add_markers(markers=in_range)
modifier.save(output_path)
skipped_note = (
f"- **Skipped**: {skipped_count} beat(s) beyond the timeline end "
f"({format_duration(timeline_end)})\n" if skipped_count else ""
)
return _text_result(f"""# Beat Markers Imported
## Summary
- **Beats Found**: {len(beat_times)}
- **Markers Added**: {len(added)}
{skipped_note}- **Filter**: {beat_filter}
- **Marker Type**: {marker_type}
## Output
Saved to: `{output_path}`
*Use `snap_to_beats` to align your cuts to these markers.*
""")
async def handle_snap_to_beats(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_synced")
max_shift = arguments.get("max_shift_frames", 6)
prefer = arguments.get("prefer", "nearest")
parser = FCPXMLParser()
project = parser.parse_file(filepath)
if not project.timelines:
return _no_timeline()
tl = project.primary_timeline
fps = tl.frame_rate
markers = list(tl.markers)
for clip in tl.clips:
markers.extend(clip.markers)
if not markers:
return _text_result("No markers found. Use `import_beat_markers` first.")
marker_times = sorted([m.start.seconds for m in markers])
modifier = FCPXMLModifier(filepath)
spine = modifier._get_spine()
adjusted_count = 0
total_shift = 0
clips_list = [c for c in spine if c.tag in ('clip', 'asset-clip', 'video', 'ref-clip')]
for i, clip in enumerate(clips_list[1:], 1):
cut_offset = modifier._parse_time(clip.get('offset', '0s'))
cut_seconds = cut_offset.to_seconds()
best_marker = None
best_distance = float('inf')
for marker_time in marker_times:
distance = abs(marker_time - cut_seconds)
distance_frames = distance * fps
if distance_frames <= max_shift:
if prefer == "earlier" and marker_time <= cut_seconds:
if distance < best_distance:
best_distance = distance
best_marker = marker_time
elif prefer == "later" and marker_time >= cut_seconds:
if distance < best_distance:
best_distance = distance
best_marker = marker_time
elif prefer == "nearest":
if distance < best_distance:
best_distance = distance
best_marker = marker_time
if best_marker is not None and best_distance > 0.001:
shift = best_marker - cut_seconds
shift_frames = int(shift * fps)
prev_clip = clips_list[i - 1]
prev_dur = modifier._parse_time(prev_clip.get('duration', '0s'))
new_prev_dur = prev_dur + modifier._parse_time(f"{shift}s")
prev_clip.set('duration', new_prev_dur.to_fcpxml())
new_offset = modifier._parse_time(f"{best_marker}s")
clip.set('offset', new_offset.to_fcpxml())
adjusted_count += 1
total_shift += abs(shift_frames)
modifier.save(output_path)
avg_shift = total_shift / adjusted_count if adjusted_count > 0 else 0
return _text_result(f"""# Cuts Snapped to Beats
## Summary
- **Cuts Adjusted**: {adjusted_count}
- **Max Shift Allowed**: {max_shift} frames
- **Preference**: {prefer}
- **Average Shift**: {avg_shift:.1f} frames
## Output
Saved to: `{output_path}`
Your edits are now synced to the beat!
""")
async def handle_import_srt_markers(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_subtitled")
srt_path = _validate_filepath(arguments["srt_path"], ('.srt', '.vtt'))
mode = arguments.get("mode", "first_per_minute")
marker_type = arguments.get("marker_type", "chapter")
max_label = arguments.get("max_label_length", 50)
text = Path(srt_path).read_text(encoding='utf-8')
# Detect format and parse
if srt_path.endswith('.vtt') or text.strip().startswith('WEBVTT'):
raw_markers = parse_vtt(text)
fmt_name = "WebVTT"
else:
raw_markers = parse_srt(text)
fmt_name = "SRT"
if not raw_markers:
return _text_result(f"No subtitles found in {srt_path}")
# Apply mode filtering
filtered = []
if mode == "all":
filtered = raw_markers
elif mode == "first_per_minute":
seen_minutes = set()
for m in raw_markers:
minute = int(m['seconds'] // 60)
if minute not in seen_minutes:
seen_minutes.add(minute)
filtered.append(m)
elif mode == "scene_changes":
# Group by similar text, take first occurrence of each unique line
seen_texts = set()
for m in raw_markers:
# Normalize: lowercase, strip punctuation
normalized = re.sub(r'[^\w\s]', '', m['text'].lower()).strip()
words = normalized.split()[:3] # First 3 words as key
key = ' '.join(words)
if key and key not in seen_texts:
seen_texts.add(key)
filtered.append(m)
markers = _raw_markers_to_batch(filtered, marker_type, max_label=max_label)
modifier = FCPXMLModifier(filepath)
added = modifier.batch_add_markers(markers=markers)
modifier.save(output_path)
return _text_result(f"""# Subtitle Markers Imported
## Summary
- **Format**: {fmt_name}
- **Subtitles Parsed**: {len(raw_markers)}
- **Mode**: {mode}
- **Markers Added**: {len(added)}
- **Marker Type**: {marker_type}
## Output
Saved to: `{output_path}`
""")
async def handle_import_transcript_markers(arguments: dict) -> Sequence[TextContent]:
filepath, output_path = _resolve_io_paths(arguments, "_chapters")
marker_type = arguments.get("marker_type", "chapter")
# Get transcript text from inline or file
transcript = arguments.get("transcript")
transcript_path = arguments.get("transcript_path")
if not transcript and not transcript_path:
return _text_result("Provide either 'transcript' (inline text) or 'transcript_path' (path to file)")
if transcript_path:
# .txt only: the parser below understands "0:00 Title" lines, not real
# SRT/VTT cue syntax — that's import_srt_markers (parse_srt/parse_vtt).
transcript_path = _validate_filepath(transcript_path, ('.txt',))
transcript = Path(transcript_path).read_text(encoding='utf-8')
raw_markers = parse_transcript_timestamps(transcript or "")
if not raw_markers:
return _text_result("No timestamps found. Expected format: '0:00 Title' or 'HH:MM:SS Title', one per line.")
markers = _raw_markers_to_batch(raw_markers, marker_type)
modifier = FCPXMLModifier(filepath)
added = modifier.batch_add_markers(markers=markers)
modifier.save(output_path)
return _text_result(f"""# Transcript Markers Imported
## Summary
- **Timestamps Found**: {len(raw_markers)}
- **Markers Added**: {len(added)}
- **Marker Type**: {marker_type}
## Markers
""" + "\n".join(f"- `{m['timecode']}` {m['name']}" for m in markers) + f"""
## Output
Saved to: `{output_path}`
""")
async def handle_transcript_markers(arguments: dict) -> Sequence[TextContent]:
"""Add a marker at the start of each transcribed segment, using each
media's cached (or freshly transcribed) local Whisper transcript.
Unlike ``import_transcript_markers`` (plain "0:00 Title" text) or
``import_srt_markers`` (a caption track already synced to the whole
exported video), this maps each segment's SOURCE-media timestamp to its
TIMELINE position per spine clip — the same source->timeline mapping
``detect_media_silence`` uses — so it stays correct across multiple
clips built from different (and differently-trimmed) source files.
"""
marker_type = arguments.get("marker_type", "chapter")
max_label = int(arguments.get("max_label_length", 50))
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
clip_filter = arguments.get("clip_name")
filepath, output_path, modifier = _setup_modifier(arguments, "_transcript_markers")
added: list[tuple[str, float, str]] = []
skipped: list[tuple[str, str]] = []
spine_clips = [el for _, el in modifier._iter_spine_clips()]
for el in spine_clips:
name = el.get("name", "")
if clip_filter and name != clip_filter:
continue
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
media_path = media_src_to_path(src)
if not media_path or not Path(media_path).is_file():
skipped.append((name, "media file missing"))
continue
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
if data is None:
skipped.append((name, reason))
continue
clip_source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
clip_offset = modifier._parse_time(el.get("offset", "0s")).to_seconds()
window_end = clip_source_start + clip_duration
for seg in data.get("segments", []):
seg_start = float(seg.get("start", 0.0))
if seg_start < clip_source_start or seg_start >= window_end:
continue
label = seg.get("text", "").strip()
if not label:
continue
if max_label and len(label) > max_label:
label = label[:max_label]
timeline_seconds = clip_offset + (seg_start - clip_source_start)
modifier.add_marker_at_timeline(
timecode=f"{timeline_seconds}s", name=label, marker_type=marker_type,
)
added.append((name, seg_start, label))
if not added:
text = "# Transcript Markers\n\nNo segments to mark — file unchanged (nothing saved)."
if skipped:
text += "\n\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
return _text_result(text)
modifier.save(output_path)
result = "# Transcript Markers Imported (local Whisper)\n\n## Summary\n"
result += f"- **Markers Added**: {len(added)}\n- **Marker Type**: {marker_type}\n\n"
result += _markdown_table(
["Clip", "Start", "Label"], [[n, f"{s:.2f}s", label] for n, s, label in added]
)
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
return _text_result(result)
HANDLERS = {
"import_beat_markers": handle_import_beat_markers,
"snap_to_beats": handle_snap_to_beats,
"import_srt_markers": handle_import_srt_markers,
"import_transcript_markers": handle_import_transcript_markers,
"transcript_markers": handle_transcript_markers,
}