"""Shared internal helpers used by tool handlers across categories. Extracted from server.py — validation, formatting, and small parsing utilities that more than one server_tools/*.py module needs. """ from __future__ import annotations import json import os import re from pathlib import Path from typing import Any, Sequence from mcp.types import TextContent from fcpxml.media_intel import media_src_to_path from fcpxml.models import ( DuplicateGroup, FlashFrame, FlashFrameSeverity, GapInfo, Timecode, TimeValue, ) from fcpxml.parser import FCPXMLParser from fcpxml.rough_cut import RoughCutGenerator from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe from fcpxml.writer import FCPXMLModifier PROJECTS_DIR = os.environ.get("FCP_PROJECTS_DIR", os.path.expanduser("~/Movies")) _SANDBOX_ENABLED = "FCP_PROJECTS_DIR" in os.environ MAX_FILE_SIZE = 100 * 1024 * 1024 MAX_MEDIA_FILE_SIZE = 32 * 1024 * 1024 * 1024 _MAX_JSON_DEPTH = 50 def _check_json_depth(obj: object, _depth: int = 0) -> None: """Reject JSON structures nested beyond _MAX_JSON_DEPTH. Prevents denial-of-service via deeply nested objects that exhaust the call stack or memory during downstream processing. Called after json.load() since Python's json module has no built-in depth limit. """ if _depth > _MAX_JSON_DEPTH: raise ValueError( f"JSON nesting depth exceeds {_MAX_JSON_DEPTH} — " "file may be malformed or adversarial" ) if isinstance(obj, dict): for v in obj.values(): _check_json_depth(v, _depth + 1) elif isinstance(obj, list): for item in obj: _check_json_depth(item, _depth + 1) def _validate_filepath( filepath: str, allowed_extensions: tuple[str, ...] | None = None, max_size: int = MAX_FILE_SIZE, ) -> str: """Validate a user-provided file path against traversal and size attacks. Resolves symlinks, blocks null bytes, enforces extension whitelist, and checks file size before any parsing takes place. ``max_size`` defaults to the document limit; callers handling source media pass ``MAX_MEDIA_FILE_SIZE``, since media is streamed rather than parsed into memory (see the constant for why). Raises: ValueError: For invalid paths (null bytes, bad extensions, oversized). FileNotFoundError: When the resolved path does not exist. """ if '\x00' in filepath: raise ValueError("Invalid file path: null byte detected") resolved = Path(filepath).resolve() if not resolved.exists(): raise FileNotFoundError(f"File not found: {filepath}") # .fcpxmld bundles are directories (a package wrapping Info.fcpxml plus # sidecar data files for object tracking / Cinematic mode). The size # check applies to the inner Info.fcpxml, which is what gets parsed. if resolved.is_dir(): if resolved.suffix.lower() != '.fcpxmld': raise ValueError(f"Not a regular file: {filepath}") inner = resolved / 'Info.fcpxml' if not inner.is_file(): raise ValueError(f"Invalid bundle (no Info.fcpxml): {filepath}") size_target = inner elif not resolved.is_file(): raise ValueError(f"Not a regular file: {filepath}") else: size_target = resolved if allowed_extensions and resolved.suffix.lower() not in allowed_extensions: raise ValueError( f"Invalid file type '{resolved.suffix}'. " f"Allowed: {', '.join(allowed_extensions)}" ) if size_target.stat().st_size > max_size: size_mb = size_target.stat().st_size / (1024 * 1024) raise ValueError(f"File too large ({size_mb:.1f} MB). Maximum: {max_size // (1024 * 1024)} MB") return str(resolved) def _validate_output_path(output_path: str, *, anchor_dir: str | None = None) -> str: """Validate an output path with optional sandbox enforcement. Resolves traversal, blocks null bytes, ensures parent exists, and — when *anchor_dir* is provided — verifies the resolved output lives under that directory. This prevents LLM-generated tool calls from writing to arbitrary filesystem locations (e.g. ``/etc/cron.d/backdoor``). Args: output_path: The raw output path to validate. anchor_dir: If set, the resolved output must be a child of this directory. Typically the parent directory of the input file so outputs stay co-located with their sources. Raises: ValueError: For null bytes, missing parent, or sandbox escape. """ if '\x00' in output_path: raise ValueError("Invalid output path: null byte detected") resolved = Path(output_path).resolve() if not resolved.parent.exists(): raise ValueError(f"Output directory does not exist: {resolved.parent}") if anchor_dir is not None: anchor = Path(anchor_dir).resolve() try: resolved.relative_to(anchor) except ValueError: raise ValueError( f"Output path escapes allowed directory: " f"{resolved} is not under {anchor}" ) return str(resolved) def _validate_directory(directory: str, *, allowed_root: str | None = None) -> str: """Validate a user-provided directory path against traversal and injection. Resolves symlinks, blocks null bytes, and verifies the path is a real directory. When *allowed_root* is given, the resolved path must be a descendant of (or equal to) that root — preventing filesystem enumeration beyond the project workspace. Raises: ValueError: For invalid paths (null bytes, not a directory, sandbox escape). """ if '\x00' in directory: raise ValueError("Invalid directory path: null byte detected") resolved = Path(directory).resolve() if not resolved.is_dir(): raise ValueError(f"Not a valid directory: {directory}") if allowed_root is not None: root = Path(allowed_root).resolve() try: resolved.relative_to(root) except ValueError: raise ValueError( f"Directory escapes allowed root: " f"{resolved} is not under {root}" ) return str(resolved) def find_fcpxml_files(directory: str) -> list[str]: """Find all FCPXML files in a directory.""" path = Path(directory) files = list(str(f) for f in path.rglob("*.fcpxml")) files.extend(str(f) for f in path.rglob("*.fcpxmld")) return sorted(files) def format_timecode(tc) -> str: """Format a Timecode object to SMPTE string.""" return tc.to_smpte() if tc else "00:00:00:00" def format_duration(seconds: float) -> str: """Format seconds into human-readable duration.""" if seconds < 1: return f"{seconds*1000:.0f}ms" elif seconds < 60: return f"{seconds:.2f}s" return f"{int(seconds // 60)}m {seconds % 60:.1f}s" def _format_clip_table(clips: list, header: str) -> str: """Render a list of clips as a markdown table with timecodes and durations. Shared by handlers that filter clips by duration threshold (find_short_cuts, find_long_clips). """ result = f"{header}\n\n| Name | TC | Duration |\n|------|----|---------|\n" result += "\n".join( f"| {c.name} | {format_timecode(c.start)} | {format_duration(c.duration_seconds)} |" for c in clips ) return result def _markdown_table(headers: list[str], rows: list[list[str]]) -> str: """Build a markdown table from headers and rows. Returns header row, separator row, and data rows as a single string. Callers avoid repeating the ``| H1 | H2 |\\n|---|---|`` boilerplate that appears in 15+ handlers. """ header_line = "| " + " | ".join(headers) + " |" sep_line = "|" + "|".join("------" for _ in headers) + "|" data_lines = "\n".join( "| " + " | ".join(str(c) for c in row) + " |" for row in rows ) return f"{header_line}\n{sep_line}\n{data_lines}" def _format_batch_result( title: str, summary: dict[str, str], headers: list[str], rows: list[list[str]], output_path: str, ) -> str: """Build a standard batch-operation result with summary, table, and save footer. Used by batch fix handlers (flash frames, rapid trim, fill gaps) that all share the same markdown structure: ``# Title → ## Summary → ## Details table → Saved to`` footer. """ summary_lines = "\n".join(f"- **{k}**: {v}" for k, v in summary.items()) table = _markdown_table(headers, rows) return ( f"# {title}\n\n" f"## Summary\n{summary_lines}\n\n" f"## Details\n{table}\n\n" f"Saved to: `{output_path}`" ) def _fmt_suggestions(suggestions: list[str]) -> str: """Format pacing suggestions as markdown list (Python 3.10 compatible).""" if not suggestions: return "- Pacing looks good!" nl = "\n" return nl.join(f"- {s}" for s in suggestions) def generate_output_path(input_path: str, suffix: str = "_modified") -> str: """Generate output path from input path. The suffix is sanitized to prevent path-component injection — only alphanumeric, hyphen, underscore, and dot characters survive. """ # Strip anything that could inject path separators or traversal sequences clean_suffix = re.sub(r'[^a-zA-Z0-9._-]', '', suffix) if not clean_suffix: clean_suffix = "_modified" p = Path(input_path) return str(p.parent / f"{p.stem}{clean_suffix}{p.suffix}") def _parse_project(filepath: str): """Parse an FCPXML file and return the project with its primary timeline.""" filepath = _validate_filepath(filepath, ('.fcpxml', '.fcpxmld')) project = FCPXMLParser().parse_file(filepath) if not project.timelines: return None, None return project, project.primary_timeline def _text_result(text: str) -> list[TextContent]: """Wrap a string in the MCP TextContent list that every tool handler returns.""" return [TextContent(type="text", text=text)] def _no_timeline(): """Standard response when no timelines are found.""" return _text_result("No timelines found") def _require_timeline(filepath: str): """Parse FCPXML and return (project, timeline), raising if no timeline exists. Centralises the repeated _parse_project + _no_timeline guard that appears in every read-only timeline handler. Returns a tuple so callers can destructure directly:: project, tl = _require_timeline(arguments["filepath"]) """ project, tl = _parse_project(filepath) if not tl: raise _NoTimelineError() return project, tl class _NoTimelineError(Exception): """Sentinel raised by _require_timeline when no timelines exist.""" def _resolve_io_paths( arguments: dict, suffix: str = "_modified", ) -> tuple[str, str]: """Validate input filepath and resolve the output path. Shared foundation for every handler that reads an FCPXML and writes a derived file. Validates the input, falls back to a suffixed output name when ``output_path`` is not supplied, and sandbox-checks the result. Args: arguments: Tool arguments dict (must contain ``filepath``; may contain ``output_path``). suffix: Default output filename suffix when ``output_path`` is not provided (e.g. ``"_modified"``, ``"_beats"``). Returns: ``(filepath, output_path)`` tuple with both paths validated. """ filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld')) # Anchor write operations to the input file's directory so LLM-generated # tool calls cannot write to arbitrary filesystem locations (e.g. # /etc/cron.d/backdoor). When the explicit sandbox is off, the anchor # still prevents writes outside the source directory tree. # `output_dir` is where the caller wants the file written, not merely a # sandbox boundary: the app's "Pasta do projeto" promises that everything # generated lands there. Deriving the name from the input but keeping the # input's directory made every cross-directory call fail its own anchor # check ("output path escapes allowed directory"), so the setting silently # only worked when it pointed at the directory the file was already going # to. An explicit `output_path` still wins, and still has to sit inside # the anchor. output_dir = arguments.get("output_dir") if output_dir: anchor = _validate_directory(str(output_dir)) default_output = str(Path(anchor) / Path(generate_output_path(filepath, suffix)).name) else: anchor = str(Path(filepath).resolve().parent) default_output = generate_output_path(filepath, suffix) output_path = _validate_output_path( arguments.get("output_path") or default_output, anchor_dir=anchor, ) return filepath, output_path def _setup_modifier( arguments: dict, suffix: str = "_modified", ) -> tuple[str, str, "FCPXMLModifier"]: """Common setup for write handlers: validate paths and create modifier. Consolidates the repeated validate-filepath → resolve-output-path → create-modifier boilerplate shared by 18+ write handlers. Args: arguments: Tool arguments dict (must contain ``filepath``; may contain ``output_path``). suffix: Default output filename suffix when ``output_path`` is not provided (e.g. ``"_modified"``, ``"_flash_fixed"``). Returns: ``(filepath, output_path, modifier)`` tuple ready for the handler's domain-specific operation. """ filepath, output_path = _resolve_io_paths(arguments, suffix) modifier = FCPXMLModifier(filepath) return filepath, output_path, modifier def _setup_generator( arguments: dict, suffix: str = "_roughcut", ) -> tuple[str, str, "RoughCutGenerator"]: """Common setup for generation handlers: validate paths and create generator. Args: arguments: Tool arguments dict (must contain ``filepath`` and ``output_path``). suffix: Default output filename suffix. Returns: ``(filepath, output_path, generator)`` tuple. """ filepath, output_path = _resolve_io_paths(arguments, suffix) generator = RoughCutGenerator(filepath) return filepath, output_path, generator def _parse_timestamp_parts( parts: list[str], *, frame_rate: float = 24.0 ) -> float | None: """Convert colon-separated timestamp parts to total seconds. Handles 2-part (M:SS), 3-part (H:MM:SS / HH:MM:SS.ms), and 4-part (HH:MM:SS:FF SMPTE) formats. Returns ``None`` when the part count is unrecognised so callers can skip. Args: parts: Colon-split timestamp components. frame_rate: FPS used to convert the frame component of SMPTE timecodes into fractional seconds (default 24.0). """ if len(parts) == 2: return int(parts[0]) * 60 + float(parts[1]) elif len(parts) == 3: return int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2]) elif len(parts) == 4: # SMPTE: HH:MM:SS:FF — convert frames to fractional seconds base = int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2]) frames = int(parts[3]) return base + (frames / frame_rate) if frame_rate > 0 else base return None def _raw_markers_to_batch( raw_markers: list[dict], marker_type: str = "chapter", max_label: int | None = None, ) -> list[dict]: """Convert raw {seconds, text} marker dicts to batch_add_markers format. Shared by import_srt_markers and import_transcript_markers. """ batch = [] for m in raw_markers: label = m["text"] if max_label and len(label) > max_label: label = label[:max_label] batch.append({ "timecode": f"{m['seconds']}s", "name": label, "marker_type": marker_type.upper(), }) return batch def _extract_subtitle_blocks(text: str, *, strip_vtt_tags: bool = False) -> list[dict]: """Extract timestamp/text pairs from subtitle cue blocks (SRT or VTT). Both SRT and VTT use the same ``start --> end`` cue syntax with text lines underneath; only header stripping and tag cleaning differ. """ markers = [] blocks = re.split(r'\n\s*\n', text.strip()) for block in blocks: lines = block.strip().split('\n') if len(lines) < 2: continue ts_line = None text_lines = [] for line in lines: if '-->' in line: ts_line = line elif ts_line is not None: if strip_vtt_tags: line = re.sub(r'<[^>]+>', '', line) cleaned = line.strip() if cleaned: text_lines.append(cleaned) if not ts_line or not text_lines: continue start_str = ts_line.split('-->')[0].strip().replace(',', '.') seconds = _parse_timestamp_parts(start_str.split(':')) if seconds is not None: markers.append({'seconds': seconds, 'text': ' '.join(text_lines)}) return markers def parse_srt(text: str) -> list[dict]: """Parse SRT subtitle format into timestamp/text pairs.""" return _extract_subtitle_blocks(text) def parse_vtt(text: str) -> list[dict]: """Parse WebVTT subtitle format into timestamp/text pairs.""" text = re.sub(r'^WEBVTT.*?\n', '', text, flags=re.MULTILINE) text = re.sub(r'NOTE\n.*?\n\n', '', text, flags=re.DOTALL) return _extract_subtitle_blocks(text, strip_vtt_tags=True) def parse_transcript_timestamps(text: str) -> list[dict]: """Parse timestamped text (YouTube description format) into markers. Supports formats like: 0:00 Introduction 00:01:30 Main Topic 1:05:30 Conclusion 00:00:00:00 SMPTE timecode """ markers = [] for line in text.strip().split('\n'): line = line.strip() if not line: continue match = re.match(r'^(\d{1,2}:\d{2}(?::\d{2}){0,2})\s+(.+)$', line) if match: seconds = _parse_timestamp_parts(match.group(1).split(':')) if seconds is not None: markers.append({'seconds': seconds, 'text': match.group(2).strip()}) return markers def _detect_flash_frames( tl: Any, *, critical_threshold: int = 2, warning_threshold: int = 6, ) -> list: """Find clips shorter than *warning_threshold* frames. Returns a list of ``FlashFrame`` objects sorted by severity. Shared by ``handle_detect_flash_frames`` and ``handle_validate_timeline`` so the detection logic lives in exactly one place. """ fps = tl.frame_rate flash_frames: list[FlashFrame] = [] for clip in tl.clips: duration_frames = int(clip.duration_seconds * fps) if duration_frames < warning_threshold: severity = ( FlashFrameSeverity.CRITICAL if duration_frames < critical_threshold else FlashFrameSeverity.WARNING ) flash_frames.append(FlashFrame( clip_name=clip.name, clip_id=clip.name, start=clip.start, duration_frames=duration_frames, duration_seconds=clip.duration_seconds, severity=severity, )) return flash_frames def _detect_gaps(tl: Any, *, min_gap_frames: int = 1) -> list: """Find inter-clip gaps of at least *min_gap_frames* length. Returns a list of ``GapInfo`` objects. Shared by ``handle_detect_gaps`` and ``handle_validate_timeline``. """ fps = tl.frame_rate min_gap_seconds = min_gap_frames / fps gaps: list[GapInfo] = [] sorted_clips = sorted(tl.clips, key=lambda c: c.start.seconds) for i in range(len(sorted_clips) - 1): current_end = sorted_clips[i].end.seconds next_start = sorted_clips[i + 1].start.seconds gap_duration = next_start - current_end if gap_duration >= min_gap_seconds: gaps.append(GapInfo( start=Timecode(frames=int(current_end * fps), frame_rate=fps), duration_frames=int(gap_duration * fps), duration_seconds=gap_duration, previous_clip=sorted_clips[i].name, next_clip=sorted_clips[i + 1].name, )) return gaps def _detect_duplicate_groups(tl: Any, *, mode: str = "same_source") -> list: """Group clips that share a source media reference. Returns a list of ``DuplicateGroup`` objects. Shared by ``handle_detect_duplicates`` and ``handle_validate_timeline``. """ source_groups: dict[str, list[dict]] = {} for clip in tl.clips: source_key = clip.media_path or clip.name if source_key not in source_groups: source_groups[source_key] = [] source_groups[source_key].append({ 'name': clip.name, 'start': clip.start.seconds, 'duration': clip.duration_seconds, 'source_start': clip.source_start.seconds if clip.source_start else 0, 'source_duration': clip.duration_seconds, 'timecode': format_timecode(clip.start), }) duplicates: list[DuplicateGroup] = [] for source_key, clips in source_groups.items(): if len(clips) <= 1: continue group = DuplicateGroup( source_ref=source_key, source_name=source_key.split('/')[-1] if '/' in source_key else source_key, clips=clips, ) if mode == "same_source": duplicates.append(group) elif mode == "overlapping_ranges" and group.has_overlapping_ranges: duplicates.append(group) elif mode == "identical": seen_ranges: set[tuple] = set() identical_clips = [] for c in clips: range_key = (c['source_start'], c['source_duration']) if range_key in seen_ranges: identical_clips.append(c) seen_ranges.add(range_key) if identical_clips: group.clips = identical_clips duplicates.append(group) return duplicates AUDIO_MEDIA_EXTENSIONS = ( '.wav', '.aif', '.aiff', '.mp3', '.m4a', '.aac', '.flac', '.mov', '.mp4', ) _DIARIZATION_INSTALL_HINT = ( "\n\nInstall the optional diarization extra:\n\n" " pip install 'fcp-mcp-server[diarization]'\n\n" "and set a HuggingFace token with access to " "pyannote/speaker-diarization-3.1 (pass hf_token= or persist one via " "save_hf_token)." ) _FEATURES_INSTALL_HINT = ( "\n\nInstall the optional media-intelligence extra:\n\n" " pip install 'fcp-mcp-server[intelligence]'" ) def _voice_analysis_config_text(config: dict) -> str: w = config["emphasis_weights"] text = "# Voice Analysis Settings\n\n" text += _markdown_table( ["Setting", "Value"], [ ["Energy threshold", f"{config['energy_threshold']:.2f}"], ["Peak selection", f"top {config['peak_percentile']:.1%} of words"], ["Emphasis floor", f"{config['emphasis_floor']:.2f}"], ["Emotion detection", "on" if config["emotion_enabled"] else "off"], ["Emotion sensitivity", f"{config['emotion_sensitivity']:.2f}"], ], ) + "\n\n## Emphasis Weights\n" text += _markdown_table( ["Factor", "Weight"], [[k.replace("_", " ").title(), f"{v:.2f}"] for k, v in w.items()], ) return text def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str: """Apply one non-cut action to the clip that hosts it. ``clip_start`` is where that clip begins on the timeline; the writer wants times relative to the clip's own head, so the rebase happens here — the single place that knows about the conversion. The clip *element* is passed through rather than its name: after a cut the pieces share a name, and a name lookup would land every edit on the first piece. """ rel_start = action.start - clip_start rel_end = action.end - clip_start if action.kind == "zoom": # Only forward an explicit ease — otherwise add_zoom's own default # (a fast ramp in, instant snap back out) is what should apply. zoom_args = {} if action.params.get("ease") is not None: zoom_args["ease"] = float(action.params["ease"]) if action.params.get("ease_out") is not None: zoom_args["ease_out"] = float(action.params["ease_out"]) modifier.add_zoom( clip_id=clip_el, start=rel_start, end=rel_end, scale=float(action.params.get("scale", 1.3)), **zoom_args, ) return f"zoom {action.params.get('scale', 1.3):.2f}x" if action.kind == "text": modifier.add_text_title( clip_el, action.params["content"], offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(), duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(), ) return f"text \"{action.params['content'][:24]}\"" # marker modifier.add_marker( clip_id=clip_el, timecode=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(), name=action.params.get("content") or action.reason or "Voice action", note=action.reason or None, ) return "marker" def _speaker_table(profiles: Sequence[dict]) -> str: """Who was detected, ordered by how much of the runtime each holds.""" return _markdown_table( ["ID", "Name", "Share", "Speaking", "Lines", "Avg line"], [ [ p["id"], p.get("name", ""), f"{p['share']:.0%}", format_duration(p["speaking_seconds"]), str(p["segment_count"]), f"{p['avg_segment']:.1f}s", ] for p in profiles ], ) TRANSCRIBE_MAX_MEDIA = 10 _TRANSCRIBE_INSTALL_HINT = ( "\n\nInstall the optional transcription extra:\n\n" " pip install 'fcp-mcp-server[transcribe]'\n\n" "or run via uvx:\n\n" " uvx --from \"fcp-mcp-server[transcribe]\" fcp-mcp-server" ) def _transcript_json_path(media_path: str, output_dir: str | None = None) -> Path: """Where the ``_transcript.json`` for ``media_path`` lives. When ``output_dir`` (the user-selected project folder) is set, the transcript is saved/read there instead of next to the source media. """ p = Path(media_path) if output_dir: directory = Path(output_dir).expanduser() directory.mkdir(parents=True, exist_ok=True) return directory / f"{p.stem}_transcript.json" return p.with_name(p.stem + "_transcript.json") def _load_or_transcribe( media_path: str, model: str, language: str | None, output_dir: str | None = None ) -> tuple[dict | None, str]: """Load a cached ``_transcript.json`` for a media file, else transcribe and cache it. Returns ``(transcript, "")`` or ``(None, reason)``. The cache makes transcription a one-time cost per media file across all transcript tools. """ json_path = _transcript_json_path(media_path, output_dir) if json_path.is_file(): try: with open(json_path) as f: data = json.load(f) if isinstance(data, dict) and isinstance(data.get("words"), list): return data, "" except (OSError, json.JSONDecodeError, UnicodeDecodeError): pass # unreadable cache falls through to re-transcribe result = transcribe(media_path, model_size=model, language=language) if result is None: return None, "untranscribable (faster-whisper not installed or media unreadable)" anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent) out_path = _validate_output_path(str(json_path), anchor_dir=anchor) with open(out_path, "w") as f: json.dump({"source": Path(media_path).name, **result}, f, indent=2) return result, "" def _cut_transcript_spans(modifier, clip_filter, model, language, padding, spans_fn, keep_only=False, output_dir=None): """Shared cut engine for transcript-driven editing. ``spans_fn(words) -> [(start, end), ...]`` in source seconds. Spans are padded, clamped to each clip's used source window, optionally inverted (keep_only), snapped to the frame grid, and cut with ripple. """ to_frame = modifier.snap_seconds_to_frame cache: dict[str, tuple] = {} cuts_made: list[tuple[str, int, float]] = [] skipped: list[tuple[str, str]] = [] spine_clips = [el for _, el in modifier._iter_spine_clips()] for el in spine_clips: name = el.get("name", "") if clip_filter and name != clip_filter: continue src = modifier.resources.get(el.get("ref", ""), {}).get("src", "") media_path = media_src_to_path(src) if not media_path or not Path(media_path).is_file(): skipped.append((name, "media file missing")) continue if media_path not in cache: if len(cache) >= TRANSCRIBE_MAX_MEDIA: skipped.append((name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)")) continue cache[media_path] = _load_or_transcribe(media_path, model, language, output_dir) data, reason = cache[media_path] if data is None: skipped.append((name, reason)) continue clip_source_start = modifier.source_file_start(el).to_seconds() clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() window_start = clip_source_start window_end = clip_source_start + clip_duration spans = spans_fn(data.get("words", [])) padded = merge_ranges([(s - padding, e + padding) for s, e in spans]) clamped = [ (max(s, window_start), min(e, window_end)) for s, e in padded if min(e, window_end) > max(s, window_start) ] if keep_only: if not clamped: # Never delete a whole clip just because nothing matched in it. skipped.append((name, "no phrase matches — left untouched (keep_only)")) continue cut_source = invert_ranges(clamped, window_start, window_end) else: cut_source = clamped cut_ranges = [ (to_frame(s - clip_source_start), to_frame(e - clip_source_start)) for s, e in cut_source ] cut_ranges = [(a, b) for a, b in cut_ranges if b > a] if not cut_ranges: continue removed = modifier.cut_clip_ranges(el, cut_ranges) if removed > TimeValue.zero(): cuts_made.append((name, len(cut_ranges), removed.to_seconds())) return cuts_made, skipped def _transcript_cut_report(title, summary_lines, cuts_made, skipped, output_path, footer): if not cuts_made: text = f"# {title}\n\nNo cuts to make — file unchanged (nothing saved)." if skipped: text += "\n\n## Skipped Clips\n" + _markdown_table( ["Clip", "Reason"], [[name, reason] for name, reason in skipped] ) if any("faster-whisper" in reason for _, reason in skipped): text += _TRANSCRIBE_INSTALL_HINT return _text_result(text) total_removed = sum(seconds for _, _, seconds in cuts_made) result = f"# {title}\n\n## Summary\n" result += "\n".join(summary_lines) + "\n" result += f"- **Clips Cut**: {len(cuts_made)}\n- **Total Removed**: {format_duration(total_removed)}\n" result += "\n## Cuts\n" result += _markdown_table( ["Clip", "Ranges Cut", "Removed"], [[name, str(count), f"{seconds:.2f}s"] for name, count, seconds in cuts_made], ) + "\n" if skipped: result += "\n## Skipped Clips\n" + _markdown_table( ["Clip", "Reason"], [[name, reason] for name, reason in skipped] ) + "\n" result += f"\nSaved to: {output_path}\n\n{footer}" return _text_result(result)