Files

236 lines
11 KiB
Python

"""Transcrição & edição por transcrição — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
from pathlib import Path
from typing import Sequence
from mcp.types import TextContent, Tool
from fcpxml.media_intel import media_src_to_path
from fcpxml.transcribe import (
DEFAULT_FILLERS,
find_filler_spans,
find_phrase_spans,
merge_ranges,
segments_to_srt,
)
from server_tools._shared import (
_TRANSCRIBE_INSTALL_HINT,
TRANSCRIBE_MAX_MEDIA,
_cut_transcript_spans,
_load_or_transcribe,
_markdown_table,
_require_timeline,
_setup_modifier,
_text_result,
_transcript_cut_report,
_validate_output_path,
format_duration,
)
TOOLS = [
Tool(
name="transcribe_media",
description="Transcribe each clip's source media locally with word-level timestamps (faster-whisper). Writes a _transcript.json next to each media file (reused by edit_by_transcript / remove_filler_words so media is only transcribed once) and optionally an SRT for captions. Requires the optional [transcribe] extra; degrades to an install hint without it.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"clip_name": {"type": "string", "description": "Only transcribe the clip with this name"},
"model": {"type": "string", "default": "base", "description": "Whisper model size: tiny, base, small, medium, large-v3 (default base; larger = slower + more accurate)"},
"language": {"type": "string", "description": "ISO language code hint (e.g. 'en'); auto-detected if omitted"},
"write_srt": {"type": "boolean", "default": False, "description": "Also write a _transcript.srt next to each media file (plugs into import_srt_markers)"},
},
"required": ["filepath"]
}
),
Tool(
name="edit_by_transcript",
description="Text-based editing: cut timeline content by what was SAID. mode=remove cuts every occurrence of the given phrases (with ripple); mode=keep_only keeps only the matched phrases and cuts everything else in each matched clip (clips with no matches are left untouched). Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _transcript_edit copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"phrases": {"type": "array", "items": {"type": "string"}, "description": "Spoken phrases to match (case/punctuation-insensitive)"},
"mode": {"type": "string", "enum": ["remove", "keep_only"], "default": "remove", "description": "remove=cut matches out; keep_only=keep only matches"},
"clip_name": {"type": "string", "description": "Only edit the clip with this name"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"padding": {"type": "number", "default": 0.0, "description": "Seconds to widen each cut on both sides (0-2, default 0)"},
"output_path": {"type": "string", "description": "Output path (default: adds _transcript_edit suffix)"},
},
"required": ["filepath", "phrases"]
}
),
Tool(
name="remove_filler_words",
description="Cut filler words (um, uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: um, uh, uhh, umm, erm, ehm, mmm, hmm, mhm)"},
"clip_name": {"type": "string", "description": "Only clean the clip with this name"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"padding": {"type": "number", "default": 0.02, "description": "Seconds to widen each cut on both sides (0-2, default 0.02)"},
"output_path": {"type": "string", "description": "Output path (default: adds _defillered suffix)"},
},
"required": ["filepath"]
}
),
]
async def handle_transcribe_media(arguments: dict) -> Sequence[TextContent]:
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
write_srt = bool(arguments.get("write_srt", False))
_, tl = _require_timeline(arguments["filepath"])
clip_filter = arguments.get("clip_name")
done: dict[str, dict | None] = {}
skipped: list[tuple[str, str]] = []
rows: list[list[str]] = []
srt_paths: list[str] = []
for clip in tl.clips:
if clip_filter and clip.name != clip_filter:
continue
media_path = media_src_to_path(clip.media_path or "")
if not media_path or not Path(media_path).is_file():
skipped.append((clip.name, "media file missing"))
continue
if media_path in done:
continue
if len(done) >= TRANSCRIBE_MAX_MEDIA:
skipped.append((clip.name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)"))
continue
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
done[media_path] = data
if data is None:
skipped.append((clip.name, reason))
continue
if write_srt and data.get("segments"):
srt_name = Path(media_path).stem + "_transcript.srt"
srt_anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent)
srt_path = _validate_output_path(
str(Path(srt_anchor) / srt_name),
anchor_dir=srt_anchor,
)
with open(srt_path, "w") as f:
f.write(segments_to_srt(data["segments"]))
srt_paths.append(srt_path)
preview = data.get("text", "")[:160]
rows.append([
Path(media_path).name,
data.get("language", "?"),
str(len(data.get("words", []))),
format_duration(float(data.get("duration", 0.0))),
preview + ("…" if len(data.get("text", "")) > 160 else ""),
])
result = f"""# Media Transcription (local Whisper)
## Summary
- **Model**: {model}
- **Media Files Transcribed**: {len(rows)}
"""
if rows:
result += "\n## Transcripts (saved as _transcript.json next to each media file)\n"
result += _markdown_table(
["Media", "Language", "Words", "Duration", "Preview"], rows
) + "\n"
result += (
"\n*Next: `edit_by_transcript` to cut by what was said, or "
"`remove_filler_words` to clean ums/uhs. Transcripts are cached — "
"media is only transcribed once.*"
)
if srt_paths:
result += "\n\n## SRT Files\n" + "\n".join(f"- {p}" for p in srt_paths)
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
) + "\n"
if not rows and any("faster-whisper" in reason for _, reason in skipped):
result += _TRANSCRIBE_INSTALL_HINT
return _text_result(result)
async def handle_edit_by_transcript(arguments: dict) -> Sequence[TextContent]:
phrases = arguments.get("phrases") or []
if not isinstance(phrases, list) or not all(isinstance(p, str) for p in phrases):
raise ValueError("phrases must be a list of strings")
phrases = [p for p in phrases if p.strip()]
if not phrases:
raise ValueError("phrases must contain at least one non-empty string")
mode = arguments.get("mode", "remove")
if mode not in ("remove", "keep_only"):
raise ValueError(f"mode must be 'remove' or 'keep_only', got {mode!r}")
padding = float(arguments.get("padding", 0.0))
if not (0 <= padding <= 2):
raise ValueError(f"padding must be between 0 and 2 seconds, got {padding}")
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
filepath, output_path, modifier = _setup_modifier(arguments, "_transcript_edit")
def spans_fn(words):
return merge_ranges(
[span for phrase in phrases for span in find_phrase_spans(words, phrase)]
)
cuts_made, skipped = _cut_transcript_spans(
modifier, arguments.get("clip_name"), model, language, padding,
spans_fn, keep_only=(mode == "keep_only"), output_dir=output_dir,
)
if cuts_made:
modifier.save(output_path)
verb = "kept only" if mode == "keep_only" else "removed"
return _transcript_cut_report(
"Transcript Edit",
[f"- **Mode**: {mode} ({verb} the matched phrases)",
f"- **Phrases**: {', '.join(repr(p) for p in phrases)}",
f"- **Padding**: {padding}s"],
cuts_made, skipped, output_path,
"*Transcripts are cached as _transcript.json. Original file untouched.*",
)
async def handle_remove_filler_words(arguments: dict) -> Sequence[TextContent]:
fillers = arguments.get("fillers") or list(DEFAULT_FILLERS)
if not isinstance(fillers, list) or not all(isinstance(f, str) for f in fillers):
raise ValueError("fillers must be a list of strings")
padding = float(arguments.get("padding", 0.02))
if not (0 <= padding <= 2):
raise ValueError(f"padding must be between 0 and 2 seconds, got {padding}")
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
filepath, output_path, modifier = _setup_modifier(arguments, "_defillered")
cuts_made, skipped = _cut_transcript_spans(
modifier, arguments.get("clip_name"), model, language, padding,
lambda words: merge_ranges(find_filler_spans(words, fillers)),
output_dir=output_dir,
)
if cuts_made:
modifier.save(output_path)
return _transcript_cut_report(
"Filler Word Removal",
[f"- **Fillers**: {', '.join(fillers)}", f"- **Padding**: {padding}s"],
cuts_made, skipped, output_path,
"*Transcripts are cached as _transcript.json. Original file untouched.*",
)
HANDLERS = {
"transcribe_media": handle_transcribe_media,
"edit_by_transcript": handle_edit_by_transcript,
"remove_filler_words": handle_remove_filler_words,
}