"""Transcrição & edição por transcrição — tool schemas and handlers. Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog. """ from __future__ import annotations from pathlib import Path from typing import Sequence from mcp.types import TextContent, Tool from fcpxml.media_intel import media_src_to_path from fcpxml.transcribe import ( DEFAULT_FILLERS, find_filler_spans, find_phrase_spans, merge_ranges, segments_to_srt, ) from server_tools._shared import ( _TRANSCRIBE_INSTALL_HINT, TRANSCRIBE_MAX_MEDIA, _cut_transcript_spans, _load_or_transcribe, _markdown_table, _require_timeline, _setup_modifier, _text_result, _transcript_cut_report, _validate_output_path, format_duration, ) TOOLS = [ Tool( name="transcribe_media", description="Transcribe each clip's source media locally with word-level timestamps (faster-whisper). Writes a _transcript.json next to each media file (reused by edit_by_transcript / remove_filler_words so media is only transcribed once) and optionally an SRT for captions. Requires the optional [transcribe] extra; degrades to an install hint without it.", inputSchema={ "type": "object", "properties": { "filepath": {"type": "string", "description": "Path to FCPXML file"}, "clip_name": {"type": "string", "description": "Only transcribe the clip with this name"}, "model": {"type": "string", "default": "base", "description": "Whisper model size: tiny, base, small, medium, large-v3 (default base; larger = slower + more accurate)"}, "language": {"type": "string", "description": "ISO language code hint (e.g. 'en'); auto-detected if omitted"}, "write_srt": {"type": "boolean", "default": False, "description": "Also write a _transcript.srt next to each media file (plugs into import_srt_markers)"}, }, "required": ["filepath"] } ), Tool( name="edit_by_transcript", description="Text-based editing: cut timeline content by what was SAID. mode=remove cuts every occurrence of the given phrases (with ripple); mode=keep_only keeps only the matched phrases and cuts everything else in each matched clip (clips with no matches are left untouched). Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _transcript_edit copy.", inputSchema={ "type": "object", "properties": { "filepath": {"type": "string", "description": "Path to FCPXML file"}, "phrases": {"type": "array", "items": {"type": "string"}, "description": "Spoken phrases to match (case/punctuation-insensitive)"}, "mode": {"type": "string", "enum": ["remove", "keep_only"], "default": "remove", "description": "remove=cut matches out; keep_only=keep only matches"}, "clip_name": {"type": "string", "description": "Only edit the clip with this name"}, "model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"}, "padding": {"type": "number", "default": 0.0, "description": "Seconds to widen each cut on both sides (0-2, default 0)"}, "output_path": {"type": "string", "description": "Output path (default: adds _transcript_edit suffix)"}, }, "required": ["filepath", "phrases"] } ), Tool( name="remove_filler_words", description="Cut filler interjections (uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'um', 'uma', 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.", inputSchema={ "type": "object", "properties": { "filepath": {"type": "string", "description": "Path to FCPXML file"}, "fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: uh, uhh, umm, erm, ehm, mmm, hmm, mhm; pass um/uma explicitly if desired)"}, "clip_name": {"type": "string", "description": "Only clean the clip with this name"}, "model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"}, "padding": {"type": "number", "default": 0.02, "description": "Seconds to widen each cut on both sides (0-2, default 0.02)"}, "output_path": {"type": "string", "description": "Output path (default: adds _defillered suffix)"}, }, "required": ["filepath"] } ), ] async def handle_transcribe_media(arguments: dict) -> Sequence[TextContent]: model = arguments.get("model", "base") language = arguments.get("language") output_dir = arguments.get("output_dir") write_srt = bool(arguments.get("write_srt", False)) _, tl = _require_timeline(arguments["filepath"]) clip_filter = arguments.get("clip_name") done: dict[str, dict | None] = {} skipped: list[tuple[str, str]] = [] rows: list[list[str]] = [] srt_paths: list[str] = [] for clip in tl.clips: if clip_filter and clip.name != clip_filter: continue media_path = media_src_to_path(clip.media_path or "") if not media_path or not Path(media_path).is_file(): skipped.append((clip.name, "media file missing")) continue if media_path in done: continue if len(done) >= TRANSCRIBE_MAX_MEDIA: skipped.append((clip.name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)")) continue data, reason = _load_or_transcribe(media_path, model, language, output_dir) done[media_path] = data if data is None: skipped.append((clip.name, reason)) continue if write_srt and data.get("segments"): srt_name = Path(media_path).stem + "_transcript.srt" srt_anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent) srt_path = _validate_output_path( str(Path(srt_anchor) / srt_name), anchor_dir=srt_anchor, ) with open(srt_path, "w") as f: f.write(segments_to_srt(data["segments"])) srt_paths.append(srt_path) preview = data.get("text", "")[:160] rows.append([ Path(media_path).name, data.get("language", "?"), str(len(data.get("words", []))), format_duration(float(data.get("duration", 0.0))), preview + ("…" if len(data.get("text", "")) > 160 else ""), ]) result = f"""# Media Transcription (local Whisper) ## Summary - **Model**: {model} - **Media Files Transcribed**: {len(rows)} """ if rows: result += "\n## Transcripts (saved as _transcript.json next to each media file)\n" result += _markdown_table( ["Media", "Language", "Words", "Duration", "Preview"], rows ) + "\n" result += ( "\n*Next: `edit_by_transcript` to cut by what was said, or " "`remove_filler_words` to clean ums/uhs. Transcripts are cached — " "media is only transcribed once.*" ) if srt_paths: result += "\n\n## SRT Files\n" + "\n".join(f"- {p}" for p in srt_paths) if skipped: result += "\n## Skipped Clips\n" + _markdown_table( ["Clip", "Reason"], [[name, reason] for name, reason in skipped] ) + "\n" if not rows and any("faster-whisper" in reason for _, reason in skipped): result += _TRANSCRIBE_INSTALL_HINT return _text_result(result) async def handle_edit_by_transcript(arguments: dict) -> Sequence[TextContent]: phrases = arguments.get("phrases") or [] if not isinstance(phrases, list) or not all(isinstance(p, str) for p in phrases): raise ValueError("phrases must be a list of strings") phrases = [p for p in phrases if p.strip()] if not phrases: raise ValueError("phrases must contain at least one non-empty string") mode = arguments.get("mode", "remove") if mode not in ("remove", "keep_only"): raise ValueError(f"mode must be 'remove' or 'keep_only', got {mode!r}") padding = float(arguments.get("padding", 0.0)) if not (0 <= padding <= 2): raise ValueError(f"padding must be between 0 and 2 seconds, got {padding}") model = arguments.get("model", "base") language = arguments.get("language") output_dir = arguments.get("output_dir") filepath, output_path, modifier = _setup_modifier(arguments, "_transcript_edit") def spans_fn(words): return merge_ranges( [span for phrase in phrases for span in find_phrase_spans(words, phrase)] ) cuts_made, skipped = _cut_transcript_spans( modifier, arguments.get("clip_name"), model, language, padding, spans_fn, keep_only=(mode == "keep_only"), output_dir=output_dir, ) if cuts_made: modifier.save(output_path) verb = "kept only" if mode == "keep_only" else "removed" return _transcript_cut_report( "Transcript Edit", [f"- **Mode**: {mode} ({verb} the matched phrases)", f"- **Phrases**: {', '.join(repr(p) for p in phrases)}", f"- **Padding**: {padding}s"], cuts_made, skipped, output_path, "*Transcripts are cached as _transcript.json. Original file untouched.*", ) async def handle_remove_filler_words(arguments: dict) -> Sequence[TextContent]: fillers = arguments.get("fillers") or list(DEFAULT_FILLERS) if not isinstance(fillers, list) or not all(isinstance(f, str) for f in fillers): raise ValueError("fillers must be a list of strings") padding = float(arguments.get("padding", 0.02)) if not (0 <= padding <= 2): raise ValueError(f"padding must be between 0 and 2 seconds, got {padding}") model = arguments.get("model", "base") language = arguments.get("language") output_dir = arguments.get("output_dir") filepath, output_path, modifier = _setup_modifier(arguments, "_defillered") cuts_made, skipped = _cut_transcript_spans( modifier, arguments.get("clip_name"), model, language, padding, lambda words: merge_ranges(find_filler_spans(words, fillers)), output_dir=output_dir, ) if cuts_made: modifier.save(output_path) return _transcript_cut_report( "Filler Word Removal", [f"- **Fillers**: {', '.join(fillers)}", f"- **Padding**: {padding}s"], cuts_made, skipped, output_path, "*Transcripts are cached as _transcript.json. Original file untouched.*", ) HANDLERS = { "transcribe_media": handle_transcribe_media, "edit_by_transcript": handle_edit_by_transcript, "remove_filler_words": handle_remove_filler_words, }