236 lines
11 KiB
Python
236 lines
11 KiB
Python
"""Transcrição & edição por transcrição — tool schemas and handlers.
|
|
|
|
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
from typing import Sequence
|
|
|
|
from mcp.types import TextContent, Tool
|
|
|
|
from fcpxml.media_intel import media_src_to_path
|
|
from fcpxml.transcribe import (
|
|
DEFAULT_FILLERS,
|
|
find_filler_spans,
|
|
find_phrase_spans,
|
|
merge_ranges,
|
|
segments_to_srt,
|
|
)
|
|
from server_tools._shared import (
|
|
_TRANSCRIBE_INSTALL_HINT,
|
|
TRANSCRIBE_MAX_MEDIA,
|
|
_cut_transcript_spans,
|
|
_load_or_transcribe,
|
|
_markdown_table,
|
|
_require_timeline,
|
|
_setup_modifier,
|
|
_text_result,
|
|
_transcript_cut_report,
|
|
_validate_output_path,
|
|
format_duration,
|
|
)
|
|
|
|
TOOLS = [
|
|
Tool(
|
|
name="transcribe_media",
|
|
description="Transcribe each clip's source media locally with word-level timestamps (faster-whisper). Writes a _transcript.json next to each media file (reused by edit_by_transcript / remove_filler_words so media is only transcribed once) and optionally an SRT for captions. Requires the optional [transcribe] extra; degrades to an install hint without it.",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"clip_name": {"type": "string", "description": "Only transcribe the clip with this name"},
|
|
"model": {"type": "string", "default": "base", "description": "Whisper model size: tiny, base, small, medium, large-v3 (default base; larger = slower + more accurate)"},
|
|
"language": {"type": "string", "description": "ISO language code hint (e.g. 'en'); auto-detected if omitted"},
|
|
"write_srt": {"type": "boolean", "default": False, "description": "Also write a _transcript.srt next to each media file (plugs into import_srt_markers)"},
|
|
},
|
|
"required": ["filepath"]
|
|
}
|
|
),
|
|
Tool(
|
|
name="edit_by_transcript",
|
|
description="Text-based editing: cut timeline content by what was SAID. mode=remove cuts every occurrence of the given phrases (with ripple); mode=keep_only keeps only the matched phrases and cuts everything else in each matched clip (clips with no matches are left untouched). Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _transcript_edit copy.",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"phrases": {"type": "array", "items": {"type": "string"}, "description": "Spoken phrases to match (case/punctuation-insensitive)"},
|
|
"mode": {"type": "string", "enum": ["remove", "keep_only"], "default": "remove", "description": "remove=cut matches out; keep_only=keep only matches"},
|
|
"clip_name": {"type": "string", "description": "Only edit the clip with this name"},
|
|
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
|
"padding": {"type": "number", "default": 0.0, "description": "Seconds to widen each cut on both sides (0-2, default 0)"},
|
|
"output_path": {"type": "string", "description": "Output path (default: adds _transcript_edit suffix)"},
|
|
},
|
|
"required": ["filepath", "phrases"]
|
|
}
|
|
),
|
|
Tool(
|
|
name="remove_filler_words",
|
|
description="Cut filler words (um, uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.",
|
|
inputSchema={
|
|
"type": "object",
|
|
"properties": {
|
|
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
|
"fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: um, uh, uhh, umm, erm, ehm, mmm, hmm, mhm)"},
|
|
"clip_name": {"type": "string", "description": "Only clean the clip with this name"},
|
|
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
|
"padding": {"type": "number", "default": 0.02, "description": "Seconds to widen each cut on both sides (0-2, default 0.02)"},
|
|
"output_path": {"type": "string", "description": "Output path (default: adds _defillered suffix)"},
|
|
},
|
|
"required": ["filepath"]
|
|
}
|
|
),
|
|
]
|
|
|
|
|
|
async def handle_transcribe_media(arguments: dict) -> Sequence[TextContent]:
|
|
model = arguments.get("model", "base")
|
|
language = arguments.get("language")
|
|
output_dir = arguments.get("output_dir")
|
|
write_srt = bool(arguments.get("write_srt", False))
|
|
_, tl = _require_timeline(arguments["filepath"])
|
|
clip_filter = arguments.get("clip_name")
|
|
|
|
done: dict[str, dict | None] = {}
|
|
skipped: list[tuple[str, str]] = []
|
|
rows: list[list[str]] = []
|
|
srt_paths: list[str] = []
|
|
for clip in tl.clips:
|
|
if clip_filter and clip.name != clip_filter:
|
|
continue
|
|
media_path = media_src_to_path(clip.media_path or "")
|
|
if not media_path or not Path(media_path).is_file():
|
|
skipped.append((clip.name, "media file missing"))
|
|
continue
|
|
if media_path in done:
|
|
continue
|
|
if len(done) >= TRANSCRIBE_MAX_MEDIA:
|
|
skipped.append((clip.name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)"))
|
|
continue
|
|
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
|
done[media_path] = data
|
|
if data is None:
|
|
skipped.append((clip.name, reason))
|
|
continue
|
|
if write_srt and data.get("segments"):
|
|
srt_name = Path(media_path).stem + "_transcript.srt"
|
|
srt_anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent)
|
|
srt_path = _validate_output_path(
|
|
str(Path(srt_anchor) / srt_name),
|
|
anchor_dir=srt_anchor,
|
|
)
|
|
with open(srt_path, "w") as f:
|
|
f.write(segments_to_srt(data["segments"]))
|
|
srt_paths.append(srt_path)
|
|
preview = data.get("text", "")[:160]
|
|
rows.append([
|
|
Path(media_path).name,
|
|
data.get("language", "?"),
|
|
str(len(data.get("words", []))),
|
|
format_duration(float(data.get("duration", 0.0))),
|
|
preview + ("…" if len(data.get("text", "")) > 160 else ""),
|
|
])
|
|
|
|
result = f"""# Media Transcription (local Whisper)
|
|
|
|
## Summary
|
|
- **Model**: {model}
|
|
- **Media Files Transcribed**: {len(rows)}
|
|
"""
|
|
if rows:
|
|
result += "\n## Transcripts (saved as _transcript.json next to each media file)\n"
|
|
result += _markdown_table(
|
|
["Media", "Language", "Words", "Duration", "Preview"], rows
|
|
) + "\n"
|
|
result += (
|
|
"\n*Next: `edit_by_transcript` to cut by what was said, or "
|
|
"`remove_filler_words` to clean ums/uhs. Transcripts are cached — "
|
|
"media is only transcribed once.*"
|
|
)
|
|
if srt_paths:
|
|
result += "\n\n## SRT Files\n" + "\n".join(f"- {p}" for p in srt_paths)
|
|
if skipped:
|
|
result += "\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
|
|
) + "\n"
|
|
if not rows and any("faster-whisper" in reason for _, reason in skipped):
|
|
result += _TRANSCRIBE_INSTALL_HINT
|
|
return _text_result(result)
|
|
|
|
|
|
async def handle_edit_by_transcript(arguments: dict) -> Sequence[TextContent]:
|
|
phrases = arguments.get("phrases") or []
|
|
if not isinstance(phrases, list) or not all(isinstance(p, str) for p in phrases):
|
|
raise ValueError("phrases must be a list of strings")
|
|
phrases = [p for p in phrases if p.strip()]
|
|
if not phrases:
|
|
raise ValueError("phrases must contain at least one non-empty string")
|
|
mode = arguments.get("mode", "remove")
|
|
if mode not in ("remove", "keep_only"):
|
|
raise ValueError(f"mode must be 'remove' or 'keep_only', got {mode!r}")
|
|
padding = float(arguments.get("padding", 0.0))
|
|
if not (0 <= padding <= 2):
|
|
raise ValueError(f"padding must be between 0 and 2 seconds, got {padding}")
|
|
model = arguments.get("model", "base")
|
|
language = arguments.get("language")
|
|
output_dir = arguments.get("output_dir")
|
|
|
|
filepath, output_path, modifier = _setup_modifier(arguments, "_transcript_edit")
|
|
|
|
def spans_fn(words):
|
|
return merge_ranges(
|
|
[span for phrase in phrases for span in find_phrase_spans(words, phrase)]
|
|
)
|
|
|
|
cuts_made, skipped = _cut_transcript_spans(
|
|
modifier, arguments.get("clip_name"), model, language, padding,
|
|
spans_fn, keep_only=(mode == "keep_only"), output_dir=output_dir,
|
|
)
|
|
if cuts_made:
|
|
modifier.save(output_path)
|
|
verb = "kept only" if mode == "keep_only" else "removed"
|
|
return _transcript_cut_report(
|
|
"Transcript Edit",
|
|
[f"- **Mode**: {mode} ({verb} the matched phrases)",
|
|
f"- **Phrases**: {', '.join(repr(p) for p in phrases)}",
|
|
f"- **Padding**: {padding}s"],
|
|
cuts_made, skipped, output_path,
|
|
"*Transcripts are cached as _transcript.json. Original file untouched.*",
|
|
)
|
|
|
|
|
|
async def handle_remove_filler_words(arguments: dict) -> Sequence[TextContent]:
|
|
fillers = arguments.get("fillers") or list(DEFAULT_FILLERS)
|
|
if not isinstance(fillers, list) or not all(isinstance(f, str) for f in fillers):
|
|
raise ValueError("fillers must be a list of strings")
|
|
padding = float(arguments.get("padding", 0.02))
|
|
if not (0 <= padding <= 2):
|
|
raise ValueError(f"padding must be between 0 and 2 seconds, got {padding}")
|
|
model = arguments.get("model", "base")
|
|
language = arguments.get("language")
|
|
output_dir = arguments.get("output_dir")
|
|
|
|
filepath, output_path, modifier = _setup_modifier(arguments, "_defillered")
|
|
|
|
cuts_made, skipped = _cut_transcript_spans(
|
|
modifier, arguments.get("clip_name"), model, language, padding,
|
|
lambda words: merge_ranges(find_filler_spans(words, fillers)),
|
|
output_dir=output_dir,
|
|
)
|
|
if cuts_made:
|
|
modifier.save(output_path)
|
|
return _transcript_cut_report(
|
|
"Filler Word Removal",
|
|
[f"- **Fillers**: {', '.join(fillers)}", f"- **Padding**: {padding}s"],
|
|
cuts_made, skipped, output_path,
|
|
"*Transcripts are cached as _transcript.json. Original file untouched.*",
|
|
)
|
|
|
|
|
|
HANDLERS = {
|
|
"transcribe_media": handle_transcribe_media,
|
|
"edit_by_transcript": handle_edit_by_transcript,
|
|
"remove_filler_words": handle_remove_filler_words,
|
|
}
|