Files

830 lines
32 KiB
Python

"""Shared internal helpers used by tool handlers across categories.
Extracted from server.py — validation, formatting, and small parsing utilities
that more than one server_tools/*.py module needs.
"""
from __future__ import annotations
import json
import os
import re
from pathlib import Path
from typing import Any, Sequence
from mcp.types import TextContent
from fcpxml.media_intel import media_src_to_path
from fcpxml.models import (
DuplicateGroup,
FlashFrame,
FlashFrameSeverity,
GapInfo,
Timecode,
TimeValue,
)
from fcpxml.parser import FCPXMLParser
from fcpxml.rough_cut import RoughCutGenerator
from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe
from fcpxml.writer import FCPXMLModifier
PROJECTS_DIR = os.environ.get("FCP_PROJECTS_DIR", os.path.expanduser("~/Movies"))
_SANDBOX_ENABLED = "FCP_PROJECTS_DIR" in os.environ
MAX_FILE_SIZE = 100 * 1024 * 1024
MAX_MEDIA_FILE_SIZE = 32 * 1024 * 1024 * 1024
_MAX_JSON_DEPTH = 50
def _check_json_depth(obj: object, _depth: int = 0) -> None:
"""Reject JSON structures nested beyond _MAX_JSON_DEPTH.
Prevents denial-of-service via deeply nested objects that exhaust the
call stack or memory during downstream processing. Called after
json.load() since Python's json module has no built-in depth limit.
"""
if _depth > _MAX_JSON_DEPTH:
raise ValueError(
f"JSON nesting depth exceeds {_MAX_JSON_DEPTH} — "
"file may be malformed or adversarial"
)
if isinstance(obj, dict):
for v in obj.values():
_check_json_depth(v, _depth + 1)
elif isinstance(obj, list):
for item in obj:
_check_json_depth(item, _depth + 1)
def _validate_filepath(
filepath: str,
allowed_extensions: tuple[str, ...] | None = None,
max_size: int = MAX_FILE_SIZE,
) -> str:
"""Validate a user-provided file path against traversal and size attacks.
Resolves symlinks, blocks null bytes, enforces extension whitelist, and
checks file size before any parsing takes place.
``max_size`` defaults to the document limit; callers handling source
media pass ``MAX_MEDIA_FILE_SIZE``, since media is streamed rather than
parsed into memory (see the constant for why).
Raises:
ValueError: For invalid paths (null bytes, bad extensions, oversized).
FileNotFoundError: When the resolved path does not exist.
"""
if '\x00' in filepath:
raise ValueError("Invalid file path: null byte detected")
resolved = Path(filepath).resolve()
if not resolved.exists():
raise FileNotFoundError(f"File not found: {filepath}")
# .fcpxmld bundles are directories (a package wrapping Info.fcpxml plus
# sidecar data files for object tracking / Cinematic mode). The size
# check applies to the inner Info.fcpxml, which is what gets parsed.
if resolved.is_dir():
if resolved.suffix.lower() != '.fcpxmld':
raise ValueError(f"Not a regular file: {filepath}")
inner = resolved / 'Info.fcpxml'
if not inner.is_file():
raise ValueError(f"Invalid bundle (no Info.fcpxml): {filepath}")
size_target = inner
elif not resolved.is_file():
raise ValueError(f"Not a regular file: {filepath}")
else:
size_target = resolved
if allowed_extensions and resolved.suffix.lower() not in allowed_extensions:
raise ValueError(
f"Invalid file type '{resolved.suffix}'. "
f"Allowed: {', '.join(allowed_extensions)}"
)
if size_target.stat().st_size > max_size:
size_mb = size_target.stat().st_size / (1024 * 1024)
raise ValueError(f"File too large ({size_mb:.1f} MB). Maximum: {max_size // (1024 * 1024)} MB")
return str(resolved)
def _validate_output_path(output_path: str, *, anchor_dir: str | None = None) -> str:
"""Validate an output path with optional sandbox enforcement.
Resolves traversal, blocks null bytes, ensures parent exists, and — when
*anchor_dir* is provided — verifies the resolved output lives under that
directory. This prevents LLM-generated tool calls from writing to
arbitrary filesystem locations (e.g. ``/etc/cron.d/backdoor``).
Args:
output_path: The raw output path to validate.
anchor_dir: If set, the resolved output must be a child of this
directory. Typically the parent directory of the input file so
outputs stay co-located with their sources.
Raises:
ValueError: For null bytes, missing parent, or sandbox escape.
"""
if '\x00' in output_path:
raise ValueError("Invalid output path: null byte detected")
resolved = Path(output_path).resolve()
if not resolved.parent.exists():
raise ValueError(f"Output directory does not exist: {resolved.parent}")
if anchor_dir is not None:
anchor = Path(anchor_dir).resolve()
try:
resolved.relative_to(anchor)
except ValueError:
raise ValueError(
f"Output path escapes allowed directory: "
f"{resolved} is not under {anchor}"
)
return str(resolved)
def _validate_directory(directory: str, *, allowed_root: str | None = None) -> str:
"""Validate a user-provided directory path against traversal and injection.
Resolves symlinks, blocks null bytes, and verifies the path is a real
directory. When *allowed_root* is given, the resolved path must be a
descendant of (or equal to) that root — preventing filesystem enumeration
beyond the project workspace.
Raises:
ValueError: For invalid paths (null bytes, not a directory, sandbox escape).
"""
if '\x00' in directory:
raise ValueError("Invalid directory path: null byte detected")
resolved = Path(directory).resolve()
if not resolved.is_dir():
raise ValueError(f"Not a valid directory: {directory}")
if allowed_root is not None:
root = Path(allowed_root).resolve()
try:
resolved.relative_to(root)
except ValueError:
raise ValueError(
f"Directory escapes allowed root: "
f"{resolved} is not under {root}"
)
return str(resolved)
def find_fcpxml_files(directory: str) -> list[str]:
"""Find all FCPXML files in a directory."""
path = Path(directory)
files = list(str(f) for f in path.rglob("*.fcpxml"))
files.extend(str(f) for f in path.rglob("*.fcpxmld"))
return sorted(files)
def format_timecode(tc) -> str:
"""Format a Timecode object to SMPTE string."""
return tc.to_smpte() if tc else "00:00:00:00"
def format_duration(seconds: float) -> str:
"""Format seconds into human-readable duration."""
if seconds < 1:
return f"{seconds*1000:.0f}ms"
elif seconds < 60:
return f"{seconds:.2f}s"
return f"{int(seconds // 60)}m {seconds % 60:.1f}s"
def _format_clip_table(clips: list, header: str) -> str:
"""Render a list of clips as a markdown table with timecodes and durations.
Shared by handlers that filter clips by duration threshold
(find_short_cuts, find_long_clips).
"""
result = f"{header}\n\n| Name | TC | Duration |\n|------|----|---------|\n"
result += "\n".join(
f"| {c.name} | {format_timecode(c.start)} | {format_duration(c.duration_seconds)} |"
for c in clips
)
return result
def _markdown_table(headers: list[str], rows: list[list[str]]) -> str:
"""Build a markdown table from headers and rows.
Returns header row, separator row, and data rows as a single string.
Callers avoid repeating the ``| H1 | H2 |\\n|---|---|`` boilerplate
that appears in 15+ handlers.
"""
header_line = "| " + " | ".join(headers) + " |"
sep_line = "|" + "|".join("------" for _ in headers) + "|"
data_lines = "\n".join(
"| " + " | ".join(str(c) for c in row) + " |" for row in rows
)
return f"{header_line}\n{sep_line}\n{data_lines}"
def _format_batch_result(
title: str,
summary: dict[str, str],
headers: list[str],
rows: list[list[str]],
output_path: str,
) -> str:
"""Build a standard batch-operation result with summary, table, and save footer.
Used by batch fix handlers (flash frames, rapid trim, fill gaps) that all
share the same markdown structure: ``# Title → ## Summary → ## Details table
→ Saved to`` footer.
"""
summary_lines = "\n".join(f"- **{k}**: {v}" for k, v in summary.items())
table = _markdown_table(headers, rows)
return (
f"# {title}\n\n"
f"## Summary\n{summary_lines}\n\n"
f"## Details\n{table}\n\n"
f"Saved to: `{output_path}`"
)
def _fmt_suggestions(suggestions: list[str]) -> str:
"""Format pacing suggestions as markdown list (Python 3.10 compatible)."""
if not suggestions:
return "- Pacing looks good!"
nl = "\n"
return nl.join(f"- {s}" for s in suggestions)
def generate_output_path(input_path: str, suffix: str = "_modified") -> str:
"""Generate output path from input path.
The suffix is sanitized to prevent path-component injection — only
alphanumeric, hyphen, underscore, and dot characters survive.
"""
# Strip anything that could inject path separators or traversal sequences
clean_suffix = re.sub(r'[^a-zA-Z0-9._-]', '', suffix)
if not clean_suffix:
clean_suffix = "_modified"
p = Path(input_path)
return str(p.parent / f"{p.stem}{clean_suffix}{p.suffix}")
def _parse_project(filepath: str):
"""Parse an FCPXML file and return the project with its primary timeline."""
filepath = _validate_filepath(filepath, ('.fcpxml', '.fcpxmld'))
project = FCPXMLParser().parse_file(filepath)
if not project.timelines:
return None, None
return project, project.primary_timeline
def _text_result(text: str) -> list[TextContent]:
"""Wrap a string in the MCP TextContent list that every tool handler returns."""
return [TextContent(type="text", text=text)]
def _no_timeline():
"""Standard response when no timelines are found."""
return _text_result("No timelines found")
def _require_timeline(filepath: str):
"""Parse FCPXML and return (project, timeline), raising if no timeline exists.
Centralises the repeated _parse_project + _no_timeline guard that
appears in every read-only timeline handler. Returns a tuple so
callers can destructure directly::
project, tl = _require_timeline(arguments["filepath"])
"""
project, tl = _parse_project(filepath)
if not tl:
raise _NoTimelineError()
return project, tl
class _NoTimelineError(Exception):
"""Sentinel raised by _require_timeline when no timelines exist."""
def _resolve_io_paths(
arguments: dict,
suffix: str = "_modified",
) -> tuple[str, str]:
"""Validate input filepath and resolve the output path.
Shared foundation for every handler that reads an FCPXML and writes
a derived file. Validates the input, falls back to a suffixed
output name when ``output_path`` is not supplied, and sandbox-checks
the result.
Args:
arguments: Tool arguments dict (must contain ``filepath``; may
contain ``output_path``).
suffix: Default output filename suffix when ``output_path`` is
not provided (e.g. ``"_modified"``, ``"_beats"``).
Returns:
``(filepath, output_path)`` tuple with both paths validated.
"""
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
# Anchor write operations to the input file's directory so LLM-generated
# tool calls cannot write to arbitrary filesystem locations (e.g.
# /etc/cron.d/backdoor). When the explicit sandbox is off, the anchor
# still prevents writes outside the source directory tree.
# `output_dir` is where the caller wants the file written, not merely a
# sandbox boundary: the app's "Pasta do projeto" promises that everything
# generated lands there. Deriving the name from the input but keeping the
# input's directory made every cross-directory call fail its own anchor
# check ("output path escapes allowed directory"), so the setting silently
# only worked when it pointed at the directory the file was already going
# to. An explicit `output_path` still wins, and still has to sit inside
# the anchor.
output_dir = arguments.get("output_dir")
if output_dir:
anchor = _validate_directory(str(output_dir))
default_output = str(Path(anchor) / Path(generate_output_path(filepath, suffix)).name)
else:
anchor = str(Path(filepath).resolve().parent)
default_output = generate_output_path(filepath, suffix)
output_path = _validate_output_path(
arguments.get("output_path") or default_output,
anchor_dir=anchor,
)
return filepath, output_path
def _setup_modifier(
arguments: dict,
suffix: str = "_modified",
) -> tuple[str, str, "FCPXMLModifier"]:
"""Common setup for write handlers: validate paths and create modifier.
Consolidates the repeated validate-filepath → resolve-output-path →
create-modifier boilerplate shared by 18+ write handlers.
Args:
arguments: Tool arguments dict (must contain ``filepath``; may
contain ``output_path``).
suffix: Default output filename suffix when ``output_path`` is
not provided (e.g. ``"_modified"``, ``"_flash_fixed"``).
Returns:
``(filepath, output_path, modifier)`` tuple ready for the
handler's domain-specific operation.
"""
filepath, output_path = _resolve_io_paths(arguments, suffix)
modifier = FCPXMLModifier(filepath)
return filepath, output_path, modifier
def _setup_generator(
arguments: dict,
suffix: str = "_roughcut",
) -> tuple[str, str, "RoughCutGenerator"]:
"""Common setup for generation handlers: validate paths and create generator.
Args:
arguments: Tool arguments dict (must contain ``filepath`` and
``output_path``).
suffix: Default output filename suffix.
Returns:
``(filepath, output_path, generator)`` tuple.
"""
filepath, output_path = _resolve_io_paths(arguments, suffix)
generator = RoughCutGenerator(filepath)
return filepath, output_path, generator
def _parse_timestamp_parts(
parts: list[str], *, frame_rate: float = 24.0
) -> float | None:
"""Convert colon-separated timestamp parts to total seconds.
Handles 2-part (M:SS), 3-part (H:MM:SS / HH:MM:SS.ms), and
4-part (HH:MM:SS:FF SMPTE) formats. Returns ``None`` when the
part count is unrecognised so callers can skip.
Args:
parts: Colon-split timestamp components.
frame_rate: FPS used to convert the frame component of SMPTE
timecodes into fractional seconds (default 24.0).
"""
if len(parts) == 2:
return int(parts[0]) * 60 + float(parts[1])
elif len(parts) == 3:
return int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2])
elif len(parts) == 4:
# SMPTE: HH:MM:SS:FF — convert frames to fractional seconds
base = int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2])
frames = int(parts[3])
return base + (frames / frame_rate) if frame_rate > 0 else base
return None
def _raw_markers_to_batch(
raw_markers: list[dict],
marker_type: str = "chapter",
max_label: int | None = None,
) -> list[dict]:
"""Convert raw {seconds, text} marker dicts to batch_add_markers format.
Shared by import_srt_markers and import_transcript_markers.
"""
batch = []
for m in raw_markers:
label = m["text"]
if max_label and len(label) > max_label:
label = label[:max_label]
batch.append({
"timecode": f"{m['seconds']}s",
"name": label,
"marker_type": marker_type.upper(),
})
return batch
def _extract_subtitle_blocks(text: str, *, strip_vtt_tags: bool = False) -> list[dict]:
"""Extract timestamp/text pairs from subtitle cue blocks (SRT or VTT).
Both SRT and VTT use the same ``start --> end`` cue syntax with
text lines underneath; only header stripping and tag cleaning differ.
"""
markers = []
blocks = re.split(r'\n\s*\n', text.strip())
for block in blocks:
lines = block.strip().split('\n')
if len(lines) < 2:
continue
ts_line = None
text_lines = []
for line in lines:
if '-->' in line:
ts_line = line
elif ts_line is not None:
if strip_vtt_tags:
line = re.sub(r'<[^>]+>', '', line)
cleaned = line.strip()
if cleaned:
text_lines.append(cleaned)
if not ts_line or not text_lines:
continue
start_str = ts_line.split('-->')[0].strip().replace(',', '.')
seconds = _parse_timestamp_parts(start_str.split(':'))
if seconds is not None:
markers.append({'seconds': seconds, 'text': ' '.join(text_lines)})
return markers
def parse_srt(text: str) -> list[dict]:
"""Parse SRT subtitle format into timestamp/text pairs."""
return _extract_subtitle_blocks(text)
def parse_vtt(text: str) -> list[dict]:
"""Parse WebVTT subtitle format into timestamp/text pairs."""
text = re.sub(r'^WEBVTT.*?\n', '', text, flags=re.MULTILINE)
text = re.sub(r'NOTE\n.*?\n\n', '', text, flags=re.DOTALL)
return _extract_subtitle_blocks(text, strip_vtt_tags=True)
def parse_transcript_timestamps(text: str) -> list[dict]:
"""Parse timestamped text (YouTube description format) into markers.
Supports formats like:
0:00 Introduction
00:01:30 Main Topic
1:05:30 Conclusion
00:00:00:00 SMPTE timecode
"""
markers = []
for line in text.strip().split('\n'):
line = line.strip()
if not line:
continue
match = re.match(r'^(\d{1,2}:\d{2}(?::\d{2}){0,2})\s+(.+)$', line)
if match:
seconds = _parse_timestamp_parts(match.group(1).split(':'))
if seconds is not None:
markers.append({'seconds': seconds, 'text': match.group(2).strip()})
return markers
def _detect_flash_frames(
tl: Any, *, critical_threshold: int = 2, warning_threshold: int = 6,
) -> list:
"""Find clips shorter than *warning_threshold* frames.
Returns a list of ``FlashFrame`` objects sorted by severity. Shared by
``handle_detect_flash_frames`` and ``handle_validate_timeline`` so the
detection logic lives in exactly one place.
"""
fps = tl.frame_rate
flash_frames: list[FlashFrame] = []
for clip in tl.clips:
duration_frames = int(clip.duration_seconds * fps)
if duration_frames < warning_threshold:
severity = (
FlashFrameSeverity.CRITICAL
if duration_frames < critical_threshold
else FlashFrameSeverity.WARNING
)
flash_frames.append(FlashFrame(
clip_name=clip.name, clip_id=clip.name,
start=clip.start, duration_frames=duration_frames,
duration_seconds=clip.duration_seconds, severity=severity,
))
return flash_frames
def _detect_gaps(tl: Any, *, min_gap_frames: int = 1) -> list:
"""Find inter-clip gaps of at least *min_gap_frames* length.
Returns a list of ``GapInfo`` objects. Shared by ``handle_detect_gaps``
and ``handle_validate_timeline``.
"""
fps = tl.frame_rate
min_gap_seconds = min_gap_frames / fps
gaps: list[GapInfo] = []
sorted_clips = sorted(tl.clips, key=lambda c: c.start.seconds)
for i in range(len(sorted_clips) - 1):
current_end = sorted_clips[i].end.seconds
next_start = sorted_clips[i + 1].start.seconds
gap_duration = next_start - current_end
if gap_duration >= min_gap_seconds:
gaps.append(GapInfo(
start=Timecode(frames=int(current_end * fps), frame_rate=fps),
duration_frames=int(gap_duration * fps),
duration_seconds=gap_duration,
previous_clip=sorted_clips[i].name,
next_clip=sorted_clips[i + 1].name,
))
return gaps
def _detect_duplicate_groups(tl: Any, *, mode: str = "same_source") -> list:
"""Group clips that share a source media reference.
Returns a list of ``DuplicateGroup`` objects. Shared by
``handle_detect_duplicates`` and ``handle_validate_timeline``.
"""
source_groups: dict[str, list[dict]] = {}
for clip in tl.clips:
source_key = clip.media_path or clip.name
if source_key not in source_groups:
source_groups[source_key] = []
source_groups[source_key].append({
'name': clip.name,
'start': clip.start.seconds,
'duration': clip.duration_seconds,
'source_start': clip.source_start.seconds if clip.source_start else 0,
'source_duration': clip.duration_seconds,
'timecode': format_timecode(clip.start),
})
duplicates: list[DuplicateGroup] = []
for source_key, clips in source_groups.items():
if len(clips) <= 1:
continue
group = DuplicateGroup(
source_ref=source_key,
source_name=source_key.split('/')[-1] if '/' in source_key else source_key,
clips=clips,
)
if mode == "same_source":
duplicates.append(group)
elif mode == "overlapping_ranges" and group.has_overlapping_ranges:
duplicates.append(group)
elif mode == "identical":
seen_ranges: set[tuple] = set()
identical_clips = []
for c in clips:
range_key = (c['source_start'], c['source_duration'])
if range_key in seen_ranges:
identical_clips.append(c)
seen_ranges.add(range_key)
if identical_clips:
group.clips = identical_clips
duplicates.append(group)
return duplicates
AUDIO_MEDIA_EXTENSIONS = (
'.wav', '.aif', '.aiff', '.mp3', '.m4a', '.aac', '.flac', '.mov', '.mp4',
)
_DIARIZATION_INSTALL_HINT = (
"\n\nInstall the optional diarization extra:\n\n"
" pip install 'fcp-mcp-server[diarization]'\n\n"
"and set a HuggingFace token with access to "
"pyannote/speaker-diarization-3.1 (pass hf_token= or persist one via "
"save_hf_token)."
)
_FEATURES_INSTALL_HINT = (
"\n\nInstall the optional media-intelligence extra:\n\n"
" pip install 'fcp-mcp-server[intelligence]'"
)
def _voice_analysis_config_text(config: dict) -> str:
w = config["emphasis_weights"]
text = "# Voice Analysis Settings\n\n"
text += _markdown_table(
["Setting", "Value"],
[
["Energy threshold", f"{config['energy_threshold']:.2f}"],
["Peak selection", f"top {config['peak_percentile']:.1%} of words"],
["Emphasis floor", f"{config['emphasis_floor']:.2f}"],
["Emotion detection", "on" if config["emotion_enabled"] else "off"],
["Emotion sensitivity", f"{config['emotion_sensitivity']:.2f}"],
],
) + "\n\n## Emphasis Weights\n"
text += _markdown_table(
["Factor", "Weight"],
[[k.replace("_", " ").title(), f"{v:.2f}"] for k, v in w.items()],
)
return text
def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str:
"""Apply one non-cut action to the clip that hosts it.
``clip_start`` is where that clip begins on the timeline; the writer
wants times relative to the clip's own head, so the rebase happens here
— the single place that knows about the conversion. The clip *element*
is passed through rather than its name: after a cut the pieces share a
name, and a name lookup would land every edit on the first piece.
"""
rel_start = action.start - clip_start
rel_end = action.end - clip_start
if action.kind == "zoom":
# Only forward an explicit ease — otherwise add_zoom's own default
# (a fast ramp in, instant snap back out) is what should apply.
zoom_args = {}
if action.params.get("ease") is not None:
zoom_args["ease"] = float(action.params["ease"])
if action.params.get("ease_out") is not None:
zoom_args["ease_out"] = float(action.params["ease_out"])
modifier.add_zoom(
clip_id=clip_el,
start=rel_start,
end=rel_end,
scale=float(action.params.get("scale", 1.3)),
**zoom_args,
)
return f"zoom {action.params.get('scale', 1.3):.2f}x"
if action.kind == "text":
modifier.add_text_title(
clip_el,
action.params["content"],
offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(),
)
return f"text \"{action.params['content'][:24]}\""
# marker
modifier.add_marker(
clip_id=clip_el,
timecode=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
name=action.params.get("content") or action.reason or "Voice action",
note=action.reason or None,
)
return "marker"
def _speaker_table(profiles: Sequence[dict]) -> str:
"""Who was detected, ordered by how much of the runtime each holds."""
return _markdown_table(
["ID", "Name", "Share", "Speaking", "Lines", "Avg line"],
[
[
p["id"],
p.get("name", ""),
f"{p['share']:.0%}",
format_duration(p["speaking_seconds"]),
str(p["segment_count"]),
f"{p['avg_segment']:.1f}s",
]
for p in profiles
],
)
TRANSCRIBE_MAX_MEDIA = 10
_TRANSCRIBE_INSTALL_HINT = (
"\n\nInstall the optional transcription extra:\n\n"
" pip install 'fcp-mcp-server[transcribe]'\n\n"
"or run via uvx:\n\n"
" uvx --from \"fcp-mcp-server[transcribe]\" fcp-mcp-server"
)
def _transcript_json_path(media_path: str, output_dir: str | None = None) -> Path:
"""Where the ``_transcript.json`` for ``media_path`` lives.
When ``output_dir`` (the user-selected project folder) is set, the
transcript is saved/read there instead of next to the source media.
"""
p = Path(media_path)
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
return directory / f"{p.stem}_transcript.json"
return p.with_name(p.stem + "_transcript.json")
def _load_or_transcribe(
media_path: str, model: str, language: str | None, output_dir: str | None = None
) -> tuple[dict | None, str]:
"""Load a cached ``_transcript.json`` for a media file, else transcribe and cache it.
Returns ``(transcript, "")`` or ``(None, reason)``. The cache makes
transcription a one-time cost per media file across all transcript tools.
"""
json_path = _transcript_json_path(media_path, output_dir)
if json_path.is_file():
try:
with open(json_path) as f:
data = json.load(f)
if isinstance(data, dict) and isinstance(data.get("words"), list):
return data, ""
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
pass # unreadable cache falls through to re-transcribe
result = transcribe(media_path, model_size=model, language=language)
if result is None:
return None, "untranscribable (faster-whisper not installed or media unreadable)"
anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent)
out_path = _validate_output_path(str(json_path), anchor_dir=anchor)
with open(out_path, "w") as f:
json.dump({"source": Path(media_path).name, **result}, f, indent=2)
return result, ""
def _cut_transcript_spans(modifier, clip_filter, model, language, padding, spans_fn, keep_only=False, output_dir=None):
"""Shared cut engine for transcript-driven editing.
``spans_fn(words) -> [(start, end), ...]`` in source seconds. Spans are
padded, clamped to each clip's used source window, optionally inverted
(keep_only), snapped to the frame grid, and cut with ripple.
"""
to_frame = modifier.snap_seconds_to_frame
cache: dict[str, tuple] = {}
cuts_made: list[tuple[str, int, float]] = []
skipped: list[tuple[str, str]] = []
spine_clips = [el for _, el in modifier._iter_spine_clips()]
for el in spine_clips:
name = el.get("name", "")
if clip_filter and name != clip_filter:
continue
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
media_path = media_src_to_path(src)
if not media_path or not Path(media_path).is_file():
skipped.append((name, "media file missing"))
continue
if media_path not in cache:
if len(cache) >= TRANSCRIBE_MAX_MEDIA:
skipped.append((name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)"))
continue
cache[media_path] = _load_or_transcribe(media_path, model, language, output_dir)
data, reason = cache[media_path]
if data is None:
skipped.append((name, reason))
continue
clip_source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
window_start = clip_source_start
window_end = clip_source_start + clip_duration
spans = spans_fn(data.get("words", []))
padded = merge_ranges([(s - padding, e + padding) for s, e in spans])
clamped = [
(max(s, window_start), min(e, window_end))
for s, e in padded
if min(e, window_end) > max(s, window_start)
]
if keep_only:
if not clamped:
# Never delete a whole clip just because nothing matched in it.
skipped.append((name, "no phrase matches — left untouched (keep_only)"))
continue
cut_source = invert_ranges(clamped, window_start, window_end)
else:
cut_source = clamped
cut_ranges = [
(to_frame(s - clip_source_start), to_frame(e - clip_source_start))
for s, e in cut_source
]
cut_ranges = [(a, b) for a, b in cut_ranges if b > a]
if not cut_ranges:
continue
removed = modifier.cut_clip_ranges(el, cut_ranges)
if removed > TimeValue.zero():
cuts_made.append((name, len(cut_ranges), removed.to_seconds()))
return cuts_made, skipped
def _transcript_cut_report(title, summary_lines, cuts_made, skipped, output_path, footer):
if not cuts_made:
text = f"# {title}\n\nNo cuts to make — file unchanged (nothing saved)."
if skipped:
text += "\n\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
)
if any("faster-whisper" in reason for _, reason in skipped):
text += _TRANSCRIBE_INSTALL_HINT
return _text_result(text)
total_removed = sum(seconds for _, _, seconds in cuts_made)
result = f"# {title}\n\n## Summary\n"
result += "\n".join(summary_lines) + "\n"
result += f"- **Clips Cut**: {len(cuts_made)}\n- **Total Removed**: {format_duration(total_removed)}\n"
result += "\n## Cuts\n"
result += _markdown_table(
["Clip", "Ranges Cut", "Removed"],
[[name, str(count), f"{seconds:.2f}s"] for name, count, seconds in cuts_made],
) + "\n"
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
) + "\n"
result += f"\nSaved to: {output_path}\n\n{footer}"
return _text_result(result)