From ffaebb3f72d8be0d658ff537dad2d07d9d4dc1b7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jo=C3=A3o=20Henrique?= Date: Wed, 19 Aug 2026 22:35:02 -0400 Subject: [PATCH] =?UTF-8?q?refactor:=20=5Fshared.py=20vira=20subpacote,=20?= =?UTF-8?q?um=20m=C3=B3dulo=20por=20papel?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Eram 882 linhas de seis papéis sem relação, sob um nome que só dizia "compartilhado" — o depósito onde tudo que servia a mais de um handler acabava caindo. media 316 transcrição em cache, corte por fala, relatório paths 206 sandbox, limites, caminho de saída project 116 abrir projeto, preparar modifier/generator captions 112 SRT, VTT, listas com timestamp detection 99 flash frames, buracos, duplicados formatting 86 tabelas e relatórios dos handlers O __init__ reexporta os 46 nomes, então os treze pontos que importam daqui não mudaram. _transcript_cut_report saiu de formatting para media: ele precisa do hint de instalação e do _text_result, ou seja, é relatório de transcrição e não formatação genérica — mover foi mais honesto que cruzar imports entre os dois módulos. Quatro testes patchavam `server_tools._shared.transcribe`; o nome agora é ligado por _shared/media.py, então o patch passou a apontar para lá — mesmo padrão da experiência #23. Lint zerado, 1454 testes passando. Co-Authored-By: Claude Opus 5 --- code/server_tools/_shared.py | 882 ------------------ code/server_tools/_shared/__init__.py | 124 +++ code/server_tools/_shared/captions.py | 117 +++ code/server_tools/_shared/detection.py | 115 +++ code/server_tools/_shared/formatting.py | 114 +++ code/server_tools/_shared/media.py | 274 ++++++ code/server_tools/_shared/paths.py | 226 +++++ code/server_tools/_shared/project.py | 89 ++ code/tests/test_diarize_media_tool.py | 4 +- code/tests/test_refine_voice_timeline_tool.py | 2 +- code/tests/test_voice_features_tool.py | 8 +- code/tests/test_voice_timeline_tool.py | 4 +- 12 files changed, 1068 insertions(+), 891 deletions(-) delete mode 100644 code/server_tools/_shared.py create mode 100644 code/server_tools/_shared/__init__.py create mode 100644 code/server_tools/_shared/captions.py create mode 100644 code/server_tools/_shared/detection.py create mode 100644 code/server_tools/_shared/formatting.py create mode 100644 code/server_tools/_shared/media.py create mode 100644 code/server_tools/_shared/paths.py create mode 100644 code/server_tools/_shared/project.py diff --git a/code/server_tools/_shared.py b/code/server_tools/_shared.py deleted file mode 100644 index fcb6354..0000000 --- a/code/server_tools/_shared.py +++ /dev/null @@ -1,882 +0,0 @@ -"""Shared internal helpers used by tool handlers across categories. - -Extracted from server.py — validation, formatting, and small parsing utilities -that more than one server_tools/*.py module needs. -""" - -from __future__ import annotations - -import json -import os -import re -from pathlib import Path -from typing import Any, Sequence - -from mcp.types import TextContent - -from fcpxml.media_intel import media_src_to_path -from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config -from fcpxml.models import ( - DuplicateGroup, - FlashFrame, - FlashFrameSeverity, - GapInfo, - Timecode, - TimeValue, -) -from fcpxml.parser import FCPXMLParser -from fcpxml.rough_cut import RoughCutGenerator -from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text -from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe -from fcpxml.writer import FCPXMLModifier - -PROJECTS_DIR = os.environ.get("FCP_PROJECTS_DIR", os.path.expanduser("~/Movies")) - -_SANDBOX_ENABLED = "FCP_PROJECTS_DIR" in os.environ - -MAX_FILE_SIZE = 100 * 1024 * 1024 - -MAX_MEDIA_FILE_SIZE = 32 * 1024 * 1024 * 1024 - -_MAX_JSON_DEPTH = 50 - -def _check_json_depth(obj: object, _depth: int = 0) -> None: - """Reject JSON structures nested beyond _MAX_JSON_DEPTH. - - Prevents denial-of-service via deeply nested objects that exhaust the - call stack or memory during downstream processing. Called after - json.load() since Python's json module has no built-in depth limit. - """ - if _depth > _MAX_JSON_DEPTH: - raise ValueError( - f"JSON nesting depth exceeds {_MAX_JSON_DEPTH} — " - "file may be malformed or adversarial" - ) - if isinstance(obj, dict): - for v in obj.values(): - _check_json_depth(v, _depth + 1) - elif isinstance(obj, list): - for item in obj: - _check_json_depth(item, _depth + 1) - -def _validate_filepath( - filepath: str, - allowed_extensions: tuple[str, ...] | None = None, - max_size: int = MAX_FILE_SIZE, -) -> str: - """Validate a user-provided file path against traversal and size attacks. - - Resolves symlinks, blocks null bytes, enforces extension whitelist, and - checks file size before any parsing takes place. - - ``max_size`` defaults to the document limit; callers handling source - media pass ``MAX_MEDIA_FILE_SIZE``, since media is streamed rather than - parsed into memory (see the constant for why). - - Raises: - ValueError: For invalid paths (null bytes, bad extensions, oversized). - FileNotFoundError: When the resolved path does not exist. - """ - if '\x00' in filepath: - raise ValueError("Invalid file path: null byte detected") - - resolved = Path(filepath).resolve() - - if not resolved.exists(): - raise FileNotFoundError(f"File not found: {filepath}") - - # .fcpxmld bundles are directories (a package wrapping Info.fcpxml plus - # sidecar data files for object tracking / Cinematic mode). The size - # check applies to the inner Info.fcpxml, which is what gets parsed. - if resolved.is_dir(): - if resolved.suffix.lower() != '.fcpxmld': - raise ValueError(f"Not a regular file: {filepath}") - inner = resolved / 'Info.fcpxml' - if not inner.is_file(): - raise ValueError(f"Invalid bundle (no Info.fcpxml): {filepath}") - size_target = inner - elif not resolved.is_file(): - raise ValueError(f"Not a regular file: {filepath}") - else: - size_target = resolved - - if allowed_extensions and resolved.suffix.lower() not in allowed_extensions: - raise ValueError( - f"Invalid file type '{resolved.suffix}'. " - f"Allowed: {', '.join(allowed_extensions)}" - ) - - if size_target.stat().st_size > max_size: - size_mb = size_target.stat().st_size / (1024 * 1024) - raise ValueError(f"File too large ({size_mb:.1f} MB). Maximum: {max_size // (1024 * 1024)} MB") - - return str(resolved) - -def _validate_output_path(output_path: str, *, anchor_dir: str | None = None) -> str: - """Validate an output path with optional sandbox enforcement. - - Resolves traversal, blocks null bytes, ensures parent exists, and — when - *anchor_dir* is provided — verifies the resolved output lives under that - directory. This prevents LLM-generated tool calls from writing to - arbitrary filesystem locations (e.g. ``/etc/cron.d/backdoor``). - - Args: - output_path: The raw output path to validate. - anchor_dir: If set, the resolved output must be a child of this - directory. Typically the parent directory of the input file so - outputs stay co-located with their sources. - - Raises: - ValueError: For null bytes, missing parent, or sandbox escape. - """ - if '\x00' in output_path: - raise ValueError("Invalid output path: null byte detected") - - resolved = Path(output_path).resolve() - - if not resolved.parent.exists(): - raise ValueError(f"Output directory does not exist: {resolved.parent}") - - if anchor_dir is not None: - anchor = Path(anchor_dir).resolve() - try: - resolved.relative_to(anchor) - except ValueError: - raise ValueError( - f"Output path escapes allowed directory: " - f"{resolved} is not under {anchor}" - ) - - return str(resolved) - -def _validate_directory(directory: str, *, allowed_root: str | None = None) -> str: - """Validate a user-provided directory path against traversal and injection. - - Resolves symlinks, blocks null bytes, and verifies the path is a real - directory. When *allowed_root* is given, the resolved path must be a - descendant of (or equal to) that root — preventing filesystem enumeration - beyond the project workspace. - - Raises: - ValueError: For invalid paths (null bytes, not a directory, sandbox escape). - """ - if '\x00' in directory: - raise ValueError("Invalid directory path: null byte detected") - - resolved = Path(directory).resolve() - - if not resolved.is_dir(): - raise ValueError(f"Not a valid directory: {directory}") - - if allowed_root is not None: - root = Path(allowed_root).resolve() - try: - resolved.relative_to(root) - except ValueError: - raise ValueError( - f"Directory escapes allowed root: " - f"{resolved} is not under {root}" - ) - - return str(resolved) - -def find_fcpxml_files(directory: str) -> list[str]: - """Find all FCPXML files in a directory.""" - path = Path(directory) - files = list(str(f) for f in path.rglob("*.fcpxml")) - files.extend(str(f) for f in path.rglob("*.fcpxmld")) - return sorted(files) - -def format_timecode(tc) -> str: - """Format a Timecode object to SMPTE string.""" - return tc.to_smpte() if tc else "00:00:00:00" - -def format_duration(seconds: float) -> str: - """Format seconds into human-readable duration.""" - if seconds < 1: - return f"{seconds*1000:.0f}ms" - elif seconds < 60: - return f"{seconds:.2f}s" - return f"{int(seconds // 60)}m {seconds % 60:.1f}s" - -def _format_clip_table(clips: list, header: str) -> str: - """Render a list of clips as a markdown table with timecodes and durations. - - Shared by handlers that filter clips by duration threshold - (find_short_cuts, find_long_clips). - """ - result = f"{header}\n\n| Name | TC | Duration |\n|------|----|---------|\n" - result += "\n".join( - f"| {c.name} | {format_timecode(c.start)} | {format_duration(c.duration_seconds)} |" - for c in clips - ) - return result - -def _markdown_table(headers: list[str], rows: list[list[str]]) -> str: - """Build a markdown table from headers and rows. - - Returns header row, separator row, and data rows as a single string. - Callers avoid repeating the ``| H1 | H2 |\\n|---|---|`` boilerplate - that appears in 15+ handlers. - """ - header_line = "| " + " | ".join(headers) + " |" - sep_line = "|" + "|".join("------" for _ in headers) + "|" - data_lines = "\n".join( - "| " + " | ".join(str(c) for c in row) + " |" for row in rows - ) - return f"{header_line}\n{sep_line}\n{data_lines}" - -def _format_batch_result( - title: str, - summary: dict[str, str], - headers: list[str], - rows: list[list[str]], - output_path: str, -) -> str: - """Build a standard batch-operation result with summary, table, and save footer. - - Used by batch fix handlers (flash frames, rapid trim, fill gaps) that all - share the same markdown structure: ``# Title → ## Summary → ## Details table - → Saved to`` footer. - """ - summary_lines = "\n".join(f"- **{k}**: {v}" for k, v in summary.items()) - table = _markdown_table(headers, rows) - return ( - f"# {title}\n\n" - f"## Summary\n{summary_lines}\n\n" - f"## Details\n{table}\n\n" - f"Saved to: `{output_path}`" - ) - -def _fmt_suggestions(suggestions: list[str]) -> str: - """Format pacing suggestions as markdown list (Python 3.10 compatible).""" - if not suggestions: - return "- Pacing looks good!" - nl = "\n" - return nl.join(f"- {s}" for s in suggestions) - -def generate_output_path(input_path: str, suffix: str = "_modified") -> str: - """Generate output path from input path. - - The suffix is sanitized to prevent path-component injection — only - alphanumeric, hyphen, underscore, and dot characters survive. - """ - # Strip anything that could inject path separators or traversal sequences - clean_suffix = re.sub(r'[^a-zA-Z0-9._-]', '', suffix) - if not clean_suffix: - clean_suffix = "_modified" - p = Path(input_path) - return str(p.parent / f"{p.stem}{clean_suffix}{p.suffix}") - -def _parse_project(filepath: str): - """Parse an FCPXML file and return the project with its primary timeline.""" - filepath = _validate_filepath(filepath, ('.fcpxml', '.fcpxmld')) - project = FCPXMLParser().parse_file(filepath) - if not project.timelines: - return None, None - return project, project.primary_timeline - -def _text_result(text: str) -> list[TextContent]: - """Wrap a string in the MCP TextContent list that every tool handler returns.""" - return [TextContent(type="text", text=text)] - -def _no_timeline(): - """Standard response when no timelines are found.""" - return _text_result("No timelines found") - -def _require_timeline(filepath: str): - """Parse FCPXML and return (project, timeline), raising if no timeline exists. - - Centralises the repeated _parse_project + _no_timeline guard that - appears in every read-only timeline handler. Returns a tuple so - callers can destructure directly:: - - project, tl = _require_timeline(arguments["filepath"]) - """ - project, tl = _parse_project(filepath) - if not tl: - raise _NoTimelineError() - return project, tl - -class _NoTimelineError(Exception): - """Sentinel raised by _require_timeline when no timelines exist.""" - -def _resolve_io_paths( - arguments: dict, - suffix: str = "_modified", -) -> tuple[str, str]: - """Validate input filepath and resolve the output path. - - Shared foundation for every handler that reads an FCPXML and writes - a derived file. Validates the input, falls back to a suffixed - output name when ``output_path`` is not supplied, and sandbox-checks - the result. - - Args: - arguments: Tool arguments dict (must contain ``filepath``; may - contain ``output_path``). - suffix: Default output filename suffix when ``output_path`` is - not provided (e.g. ``"_modified"``, ``"_beats"``). - - Returns: - ``(filepath, output_path)`` tuple with both paths validated. - """ - filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld')) - # Anchor write operations to the input file's directory so LLM-generated - # tool calls cannot write to arbitrary filesystem locations (e.g. - # /etc/cron.d/backdoor). When the explicit sandbox is off, the anchor - # still prevents writes outside the source directory tree. - # `output_dir` is where the caller wants the file written, not merely a - # sandbox boundary: the app's "Pasta do projeto" promises that everything - # generated lands there. Deriving the name from the input but keeping the - # input's directory made every cross-directory call fail its own anchor - # check ("output path escapes allowed directory"), so the setting silently - # only worked when it pointed at the directory the file was already going - # to. An explicit `output_path` still wins, and still has to sit inside - # the anchor. - output_dir = arguments.get("output_dir") - if output_dir: - anchor = _validate_directory(str(output_dir)) - default_output = str(Path(anchor) / Path(generate_output_path(filepath, suffix)).name) - else: - anchor = str(Path(filepath).resolve().parent) - default_output = generate_output_path(filepath, suffix) - output_path = _validate_output_path( - arguments.get("output_path") or default_output, - anchor_dir=anchor, - ) - return filepath, output_path - -def _setup_modifier( - arguments: dict, - suffix: str = "_modified", -) -> tuple[str, str, "FCPXMLModifier"]: - """Common setup for write handlers: validate paths and create modifier. - - Consolidates the repeated validate-filepath → resolve-output-path → - create-modifier boilerplate shared by 18+ write handlers. - - Args: - arguments: Tool arguments dict (must contain ``filepath``; may - contain ``output_path``). - suffix: Default output filename suffix when ``output_path`` is - not provided (e.g. ``"_modified"``, ``"_flash_fixed"``). - - Returns: - ``(filepath, output_path, modifier)`` tuple ready for the - handler's domain-specific operation. - """ - filepath, output_path = _resolve_io_paths(arguments, suffix) - modifier = FCPXMLModifier(filepath) - return filepath, output_path, modifier - -def _setup_generator( - arguments: dict, - suffix: str = "_roughcut", -) -> tuple[str, str, "RoughCutGenerator"]: - """Common setup for generation handlers: validate paths and create generator. - - Args: - arguments: Tool arguments dict (must contain ``filepath`` and - ``output_path``). - suffix: Default output filename suffix. - - Returns: - ``(filepath, output_path, generator)`` tuple. - """ - filepath, output_path = _resolve_io_paths(arguments, suffix) - generator = RoughCutGenerator(filepath) - return filepath, output_path, generator - -def _parse_timestamp_parts( - parts: list[str], *, frame_rate: float = 24.0 -) -> float | None: - """Convert colon-separated timestamp parts to total seconds. - - Handles 2-part (M:SS), 3-part (H:MM:SS / HH:MM:SS.ms), and - 4-part (HH:MM:SS:FF SMPTE) formats. Returns ``None`` when the - part count is unrecognised so callers can skip. - - Args: - parts: Colon-split timestamp components. - frame_rate: FPS used to convert the frame component of SMPTE - timecodes into fractional seconds (default 24.0). - """ - if len(parts) == 2: - return int(parts[0]) * 60 + float(parts[1]) - elif len(parts) == 3: - return int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2]) - elif len(parts) == 4: - # SMPTE: HH:MM:SS:FF — convert frames to fractional seconds - base = int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2]) - frames = int(parts[3]) - return base + (frames / frame_rate) if frame_rate > 0 else base - return None - -def _raw_markers_to_batch( - raw_markers: list[dict], - marker_type: str = "chapter", - max_label: int | None = None, -) -> list[dict]: - """Convert raw {seconds, text} marker dicts to batch_add_markers format. - - Shared by import_srt_markers and import_transcript_markers. - """ - batch = [] - for m in raw_markers: - label = m["text"] - if max_label and len(label) > max_label: - label = label[:max_label] - batch.append({ - "timecode": f"{m['seconds']}s", - "name": label, - "marker_type": marker_type.upper(), - }) - return batch - -def _extract_subtitle_blocks(text: str, *, strip_vtt_tags: bool = False) -> list[dict]: - """Extract timestamp/text pairs from subtitle cue blocks (SRT or VTT). - - Both SRT and VTT use the same ``start --> end`` cue syntax with - text lines underneath; only header stripping and tag cleaning differ. - """ - markers = [] - blocks = re.split(r'\n\s*\n', text.strip()) - for block in blocks: - lines = block.strip().split('\n') - if len(lines) < 2: - continue - ts_line = None - text_lines = [] - for line in lines: - if '-->' in line: - ts_line = line - elif ts_line is not None: - if strip_vtt_tags: - line = re.sub(r'<[^>]+>', '', line) - cleaned = line.strip() - if cleaned: - text_lines.append(cleaned) - if not ts_line or not text_lines: - continue - start_str = ts_line.split('-->')[0].strip().replace(',', '.') - seconds = _parse_timestamp_parts(start_str.split(':')) - if seconds is not None: - markers.append({'seconds': seconds, 'text': ' '.join(text_lines)}) - return markers - -def parse_srt(text: str) -> list[dict]: - """Parse SRT subtitle format into timestamp/text pairs.""" - return _extract_subtitle_blocks(text) - -def parse_vtt(text: str) -> list[dict]: - """Parse WebVTT subtitle format into timestamp/text pairs.""" - text = re.sub(r'^WEBVTT.*?\n', '', text, flags=re.MULTILINE) - text = re.sub(r'NOTE\n.*?\n\n', '', text, flags=re.DOTALL) - return _extract_subtitle_blocks(text, strip_vtt_tags=True) - -def parse_transcript_timestamps(text: str) -> list[dict]: - """Parse timestamped text (YouTube description format) into markers. - - Supports formats like: - 0:00 Introduction - 00:01:30 Main Topic - 1:05:30 Conclusion - 00:00:00:00 SMPTE timecode - """ - markers = [] - for line in text.strip().split('\n'): - line = line.strip() - if not line: - continue - match = re.match(r'^(\d{1,2}:\d{2}(?::\d{2}){0,2})\s+(.+)$', line) - if match: - seconds = _parse_timestamp_parts(match.group(1).split(':')) - if seconds is not None: - markers.append({'seconds': seconds, 'text': match.group(2).strip()}) - return markers - -def _detect_flash_frames( - tl: Any, *, critical_threshold: int = 2, warning_threshold: int = 6, -) -> list: - """Find clips shorter than *warning_threshold* frames. - - Returns a list of ``FlashFrame`` objects sorted by severity. Shared by - ``handle_detect_flash_frames`` and ``handle_validate_timeline`` so the - detection logic lives in exactly one place. - """ - fps = tl.frame_rate - flash_frames: list[FlashFrame] = [] - for clip in tl.clips: - duration_frames = int(clip.duration_seconds * fps) - if duration_frames < warning_threshold: - severity = ( - FlashFrameSeverity.CRITICAL - if duration_frames < critical_threshold - else FlashFrameSeverity.WARNING - ) - flash_frames.append(FlashFrame( - clip_name=clip.name, clip_id=clip.name, - start=clip.start, duration_frames=duration_frames, - duration_seconds=clip.duration_seconds, severity=severity, - )) - return flash_frames - -def _detect_gaps(tl: Any, *, min_gap_frames: int = 1) -> list: - """Find inter-clip gaps of at least *min_gap_frames* length. - - Returns a list of ``GapInfo`` objects. Shared by ``handle_detect_gaps`` - and ``handle_validate_timeline``. - """ - fps = tl.frame_rate - min_gap_seconds = min_gap_frames / fps - gaps: list[GapInfo] = [] - sorted_clips = sorted(tl.clips, key=lambda c: c.start.seconds) - for i in range(len(sorted_clips) - 1): - current_end = sorted_clips[i].end.seconds - next_start = sorted_clips[i + 1].start.seconds - gap_duration = next_start - current_end - if gap_duration >= min_gap_seconds: - gaps.append(GapInfo( - start=Timecode(frames=int(current_end * fps), frame_rate=fps), - duration_frames=int(gap_duration * fps), - duration_seconds=gap_duration, - previous_clip=sorted_clips[i].name, - next_clip=sorted_clips[i + 1].name, - )) - return gaps - -def _detect_duplicate_groups(tl: Any, *, mode: str = "same_source") -> list: - """Group clips that share a source media reference. - - Returns a list of ``DuplicateGroup`` objects. Shared by - ``handle_detect_duplicates`` and ``handle_validate_timeline``. - """ - source_groups: dict[str, list[dict]] = {} - for clip in tl.clips: - source_key = clip.media_path or clip.name - if source_key not in source_groups: - source_groups[source_key] = [] - source_groups[source_key].append({ - 'name': clip.name, - 'start': clip.start.seconds, - 'duration': clip.duration_seconds, - 'source_start': clip.source_start.seconds if clip.source_start else 0, - 'source_duration': clip.duration_seconds, - 'timecode': format_timecode(clip.start), - }) - - duplicates: list[DuplicateGroup] = [] - for source_key, clips in source_groups.items(): - if len(clips) <= 1: - continue - group = DuplicateGroup( - source_ref=source_key, - source_name=source_key.split('/')[-1] if '/' in source_key else source_key, - clips=clips, - ) - if mode == "same_source": - duplicates.append(group) - elif mode == "overlapping_ranges" and group.has_overlapping_ranges: - duplicates.append(group) - elif mode == "identical": - seen_ranges: set[tuple] = set() - identical_clips = [] - for c in clips: - range_key = (c['source_start'], c['source_duration']) - if range_key in seen_ranges: - identical_clips.append(c) - seen_ranges.add(range_key) - if identical_clips: - group.clips = identical_clips - duplicates.append(group) - return duplicates - -AUDIO_MEDIA_EXTENSIONS = ( - '.wav', '.aif', '.aiff', '.mp3', '.m4a', '.aac', '.flac', '.mov', '.mp4', -) - -_DIARIZATION_INSTALL_HINT = ( - "\n\nInstall the optional diarization extra:\n\n" - " pip install 'fcp-mcp-server[diarization]'\n\n" - "and set a HuggingFace token with access to " - "pyannote/speaker-diarization-3.1 (pass hf_token= or persist one via " - "save_hf_token)." -) - -_FEATURES_INSTALL_HINT = ( - "\n\nInstall the optional media-intelligence extra:\n\n" - " pip install 'fcp-mcp-server[intelligence]'" -) - -def _voice_analysis_config_text(config: dict) -> str: - w = config["emphasis_weights"] - text = "# Voice Analysis Settings\n\n" - text += _markdown_table( - ["Setting", "Value"], - [ - ["Energy threshold", f"{config['energy_threshold']:.2f}"], - ["Peak selection", f"top {config['peak_percentile']:.1%} of words"], - ["Emphasis floor", f"{config['emphasis_floor']:.2f}"], - ["Emotion detection", "on" if config["emotion_enabled"] else "off"], - ["Emotion sensitivity", f"{config['emotion_sensitivity']:.2f}"], - ], - ) + "\n\n## Emphasis Weights\n" - text += _markdown_table( - ["Factor", "Weight"], - [[k.replace("_", " ").title(), f"{v:.2f}"] for k, v in w.items()], - ) - return text - -def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str: - """Apply one non-cut action to the clip that hosts it. - - ``clip_start`` is where that clip begins on the timeline; the writer - wants times relative to the clip's own head, so the rebase happens here - — the single place that knows about the conversion. The clip *element* - is passed through rather than its name: after a cut the pieces share a - name, and a name lookup would land every edit on the first piece. - """ - rel_start = action.start - clip_start - rel_end = action.end - clip_start - - if action.kind == "zoom": - config = load_voice_analysis_config() - # Only forward an explicit ease — otherwise add_zoom's own default - # (a fast ramp in, instant snap back out) is what should apply. - zoom_args = { - "ease": float(action.params.get("ease", config["zoom_ease_in"])), - "ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])), - } - mode = str(action.params.get("mode", config["zoom_mode"])) - if mode == "in": - zoom_args["hold_at_end"] = True - zoom_args["start_at_peak"] = False - elif mode == "out": - zoom_args["hold_at_end"] = False - zoom_args["start_at_peak"] = True - elif mode == "in_out": - zoom_args["hold_at_end"] = False - zoom_args["start_at_peak"] = False - modifier.add_zoom( - clip_id=clip_el, - start=rel_start, - end=rel_end, - scale=float(action.params.get("scale", config["zoom_scale"])), - **zoom_args, - ) - return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x" - - if action.kind == "text": - # Default to the "Legendas Dinâmicas" emphasis style (the font used - # to highlight a word in the captions) rather than a hardcoded - # Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of - # the video's on-screen text instead of looking like a stray default - # title. Any of these the action itself specifies still wins. - subtitle_cfg = load_dynamic_subtitle_config() - font = action.params.get("font", subtitle_cfg["emphasis_font"]) - face = action.params.get("face", subtitle_cfg["emphasis_face"]) - font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0) - requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"])) - requested_kerning = float(action.params.get("kerning", 0.0) or 0.0) - - # Voice-action callouts are not part of the dynamic subtitle block. - # When omitted, put them above the subtitle band and shrink wide - # phrases to the title-safe width. The previous default (Position 0 0, - # full emphasis size) made long callouts like "PRÓTESES DE SILICONE" - # collide with captions and run off both sides of a vertical frame. - emitted_size = requested_size * font_scale - emitted_kerning = requested_kerning * font_scale - safe_width = modifier.frame_width() * 0.90 - width = measure_text( - action.params["content"], - emitted_size, - bold=bool(action.params.get("bold", False)), - kerning=emitted_kerning, - font=font, - face=face, - ) - font_size = requested_size - if width > safe_width and width > 0: - font_size = max(32, int(requested_size * safe_width / width)) - position = action.params.get("position") - if not position: - position = f"0 {modifier.frame_height() * 0.23:g}" - - modifier.add_text_title( - clip_el, - action.params["content"], - offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(), - duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(), - position=position, - font=font, - font_size=font_size, - font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]), - face=face, - bold=action.params.get("bold", False), - ) - return f"text \"{action.params['content'][:24]}\"" - - # marker - modifier.add_marker( - clip_id=clip_el, - timecode=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(), - name=action.params.get("content") or action.reason or "Voice action", - note=action.reason or None, - ) - return "marker" - -def _speaker_table(profiles: Sequence[dict]) -> str: - """Who was detected, ordered by how much of the runtime each holds.""" - return _markdown_table( - ["ID", "Name", "Share", "Speaking", "Lines", "Avg line"], - [ - [ - p["id"], - p.get("name", ""), - f"{p['share']:.0%}", - format_duration(p["speaking_seconds"]), - str(p["segment_count"]), - f"{p['avg_segment']:.1f}s", - ] - for p in profiles - ], - ) - -TRANSCRIBE_MAX_MEDIA = 10 - -_TRANSCRIBE_INSTALL_HINT = ( - "\n\nInstall the optional transcription extra:\n\n" - " pip install 'fcp-mcp-server[transcribe]'\n\n" - "or run via uvx:\n\n" - " uvx --from \"fcp-mcp-server[transcribe]\" fcp-mcp-server" -) - -def _transcript_json_path(media_path: str, output_dir: str | None = None) -> Path: - """Where the ``_transcript.json`` for ``media_path`` lives. - - When ``output_dir`` (the user-selected project folder) is set, the - transcript is saved/read there instead of next to the source media. - """ - p = Path(media_path) - if output_dir: - directory = Path(output_dir).expanduser() - directory.mkdir(parents=True, exist_ok=True) - return directory / f"{p.stem}_transcript.json" - return p.with_name(p.stem + "_transcript.json") - -def _load_or_transcribe( - media_path: str, model: str, language: str | None, output_dir: str | None = None -) -> tuple[dict | None, str]: - """Load a cached ``_transcript.json`` for a media file, else transcribe and cache it. - - Returns ``(transcript, "")`` or ``(None, reason)``. The cache makes - transcription a one-time cost per media file across all transcript tools. - """ - json_path = _transcript_json_path(media_path, output_dir) - if json_path.is_file(): - try: - with open(json_path) as f: - data = json.load(f) - if isinstance(data, dict) and isinstance(data.get("words"), list): - return data, "" - except (OSError, json.JSONDecodeError, UnicodeDecodeError): - pass # unreadable cache falls through to re-transcribe - result = transcribe(media_path, model_size=model, language=language) - if result is None: - return None, "untranscribable (faster-whisper not installed or media unreadable)" - anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent) - out_path = _validate_output_path(str(json_path), anchor_dir=anchor) - with open(out_path, "w") as f: - json.dump({"source": Path(media_path).name, **result}, f, indent=2) - return result, "" - -def _cut_transcript_spans(modifier, clip_filter, model, language, padding, spans_fn, keep_only=False, output_dir=None): - """Shared cut engine for transcript-driven editing. - - ``spans_fn(words) -> [(start, end), ...]`` in source seconds. Spans are - padded, clamped to each clip's used source window, optionally inverted - (keep_only), snapped to the frame grid, and cut with ripple. - """ - to_frame = modifier.snap_seconds_to_frame - - cache: dict[str, tuple] = {} - cuts_made: list[tuple[str, int, float]] = [] - skipped: list[tuple[str, str]] = [] - spine_clips = [el for _, el in modifier._iter_spine_clips()] - for el in spine_clips: - name = el.get("name", "") - if clip_filter and name != clip_filter: - continue - src = modifier.resources.get(el.get("ref", ""), {}).get("src", "") - media_path = media_src_to_path(src) - if not media_path or not Path(media_path).is_file(): - skipped.append((name, "media file missing")) - continue - if media_path not in cache: - if len(cache) >= TRANSCRIBE_MAX_MEDIA: - skipped.append((name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)")) - continue - cache[media_path] = _load_or_transcribe(media_path, model, language, output_dir) - data, reason = cache[media_path] - if data is None: - skipped.append((name, reason)) - continue - - clip_source_start = modifier.source_file_start(el).to_seconds() - clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() - window_start = clip_source_start - window_end = clip_source_start + clip_duration - - spans = spans_fn(data.get("words", [])) - padded = merge_ranges([(s - padding, e + padding) for s, e in spans]) - clamped = [ - (max(s, window_start), min(e, window_end)) - for s, e in padded - if min(e, window_end) > max(s, window_start) - ] - if keep_only: - if not clamped: - # Never delete a whole clip just because nothing matched in it. - skipped.append((name, "no phrase matches — left untouched (keep_only)")) - continue - cut_source = invert_ranges(clamped, window_start, window_end) - else: - cut_source = clamped - cut_ranges = [ - (to_frame(s - clip_source_start), to_frame(e - clip_source_start)) - for s, e in cut_source - ] - cut_ranges = [(a, b) for a, b in cut_ranges if b > a] - if not cut_ranges: - continue - removed = modifier.cut_clip_ranges(el, cut_ranges) - if removed > TimeValue.zero(): - cuts_made.append((name, len(cut_ranges), removed.to_seconds())) - return cuts_made, skipped - -def _transcript_cut_report(title, summary_lines, cuts_made, skipped, output_path, footer): - if not cuts_made: - text = f"# {title}\n\nNo cuts to make — file unchanged (nothing saved)." - if skipped: - text += "\n\n## Skipped Clips\n" + _markdown_table( - ["Clip", "Reason"], [[name, reason] for name, reason in skipped] - ) - if any("faster-whisper" in reason for _, reason in skipped): - text += _TRANSCRIBE_INSTALL_HINT - return _text_result(text) - total_removed = sum(seconds for _, _, seconds in cuts_made) - result = f"# {title}\n\n## Summary\n" - result += "\n".join(summary_lines) + "\n" - result += f"- **Clips Cut**: {len(cuts_made)}\n- **Total Removed**: {format_duration(total_removed)}\n" - result += "\n## Cuts\n" - result += _markdown_table( - ["Clip", "Ranges Cut", "Removed"], - [[name, str(count), f"{seconds:.2f}s"] for name, count, seconds in cuts_made], - ) + "\n" - if skipped: - result += "\n## Skipped Clips\n" + _markdown_table( - ["Clip", "Reason"], [[name, reason] for name, reason in skipped] - ) + "\n" - result += f"\nSaved to: {output_path}\n\n{footer}" - return _text_result(result) diff --git a/code/server_tools/_shared/__init__.py b/code/server_tools/_shared/__init__.py new file mode 100644 index 0000000..762b7f2 --- /dev/null +++ b/code/server_tools/_shared/__init__.py @@ -0,0 +1,124 @@ +"""Shared internal helpers used by tool handlers across categories. + +Extracted from server.py — validation, formatting, and small parsing utilities +that more than one server_tools/*.py module needs. + +Eram 882 linhas de seis papéis diferentes sob um nome que só dizia +"compartilhado". Cada papel virou um módulo; este pacote reexporta tudo, então +os treze pontos que importam daqui seguem iguais. + + paths validação contra a sandbox, limites, caminho de saída + formatting tabelas e relatórios devolvidos pelos handlers + project abrir projeto, preparar modifier/generator + captions SRT, VTT e listas com timestamp + detection flash frames, buracos, duplicados + media transcrição em cache, corte por fala, ações posicionadas +""" + +from .captions import ( + _extract_subtitle_blocks, + _parse_timestamp_parts, + _raw_markers_to_batch, + parse_srt, + parse_transcript_timestamps, + parse_vtt, +) +from .detection import ( + _detect_duplicate_groups, + _detect_flash_frames, + _detect_gaps, +) +from .formatting import ( + _fmt_suggestions, + _format_batch_result, + _format_clip_table, + _markdown_table, + _speaker_table, + _voice_analysis_config_text, + format_duration, + format_timecode, +) +from .media import ( + _DIARIZATION_INSTALL_HINT, + _FEATURES_INSTALL_HINT, + _TRANSCRIBE_INSTALL_HINT, + AUDIO_MEDIA_EXTENSIONS, + TRANSCRIBE_MAX_MEDIA, + _apply_placed_action, + _cut_transcript_spans, + _load_or_transcribe, + _transcript_cut_report, + _transcript_json_path, +) +from .paths import ( + _MAX_JSON_DEPTH, + _SANDBOX_ENABLED, + MAX_FILE_SIZE, + MAX_MEDIA_FILE_SIZE, + PROJECTS_DIR, + _check_json_depth, + _resolve_io_paths, + _validate_directory, + _validate_filepath, + _validate_output_path, + find_fcpxml_files, + generate_output_path, +) +from .project import ( + _no_timeline, + _NoTimelineError, + _parse_project, + _require_timeline, + _setup_generator, + _setup_modifier, + _text_result, +) + +__all__ = [ + "AUDIO_MEDIA_EXTENSIONS", + "MAX_FILE_SIZE", + "MAX_MEDIA_FILE_SIZE", + "PROJECTS_DIR", + "TRANSCRIBE_MAX_MEDIA", + "_DIARIZATION_INSTALL_HINT", + "_FEATURES_INSTALL_HINT", + "_MAX_JSON_DEPTH", + "_NoTimelineError", + "_SANDBOX_ENABLED", + "_TRANSCRIBE_INSTALL_HINT", + "_apply_placed_action", + "_check_json_depth", + "_cut_transcript_spans", + "_detect_duplicate_groups", + "_detect_flash_frames", + "_detect_gaps", + "_extract_subtitle_blocks", + "_fmt_suggestions", + "_format_batch_result", + "_format_clip_table", + "_load_or_transcribe", + "_markdown_table", + "_no_timeline", + "_parse_project", + "_parse_timestamp_parts", + "_raw_markers_to_batch", + "_require_timeline", + "_resolve_io_paths", + "_setup_generator", + "_setup_modifier", + "_speaker_table", + "_text_result", + "_transcript_cut_report", + "_transcript_json_path", + "_validate_directory", + "_validate_filepath", + "_validate_output_path", + "_voice_analysis_config_text", + "find_fcpxml_files", + "format_duration", + "format_timecode", + "generate_output_path", + "parse_srt", + "parse_transcript_timestamps", + "parse_vtt", +] diff --git a/code/server_tools/_shared/captions.py b/code/server_tools/_shared/captions.py new file mode 100644 index 0000000..d489955 --- /dev/null +++ b/code/server_tools/_shared/captions.py @@ -0,0 +1,117 @@ +"""Leitura de legendas e listas com timestamp (SRT, VTT, texto colado). + +Extraído de _shared.py — ver server_tools/_shared/__init__.py. +""" + +from __future__ import annotations + +import re + + +def _parse_timestamp_parts( + parts: list[str], *, frame_rate: float = 24.0 +) -> float | None: + """Convert colon-separated timestamp parts to total seconds. + + Handles 2-part (M:SS), 3-part (H:MM:SS / HH:MM:SS.ms), and + 4-part (HH:MM:SS:FF SMPTE) formats. Returns ``None`` when the + part count is unrecognised so callers can skip. + + Args: + parts: Colon-split timestamp components. + frame_rate: FPS used to convert the frame component of SMPTE + timecodes into fractional seconds (default 24.0). + """ + if len(parts) == 2: + return int(parts[0]) * 60 + float(parts[1]) + elif len(parts) == 3: + return int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2]) + elif len(parts) == 4: + # SMPTE: HH:MM:SS:FF — convert frames to fractional seconds + base = int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2]) + frames = int(parts[3]) + return base + (frames / frame_rate) if frame_rate > 0 else base + return None + +def _raw_markers_to_batch( + raw_markers: list[dict], + marker_type: str = "chapter", + max_label: int | None = None, +) -> list[dict]: + """Convert raw {seconds, text} marker dicts to batch_add_markers format. + + Shared by import_srt_markers and import_transcript_markers. + """ + batch = [] + for m in raw_markers: + label = m["text"] + if max_label and len(label) > max_label: + label = label[:max_label] + batch.append({ + "timecode": f"{m['seconds']}s", + "name": label, + "marker_type": marker_type.upper(), + }) + return batch + +def _extract_subtitle_blocks(text: str, *, strip_vtt_tags: bool = False) -> list[dict]: + """Extract timestamp/text pairs from subtitle cue blocks (SRT or VTT). + + Both SRT and VTT use the same ``start --> end`` cue syntax with + text lines underneath; only header stripping and tag cleaning differ. + """ + markers = [] + blocks = re.split(r'\n\s*\n', text.strip()) + for block in blocks: + lines = block.strip().split('\n') + if len(lines) < 2: + continue + ts_line = None + text_lines = [] + for line in lines: + if '-->' in line: + ts_line = line + elif ts_line is not None: + if strip_vtt_tags: + line = re.sub(r'<[^>]+>', '', line) + cleaned = line.strip() + if cleaned: + text_lines.append(cleaned) + if not ts_line or not text_lines: + continue + start_str = ts_line.split('-->')[0].strip().replace(',', '.') + seconds = _parse_timestamp_parts(start_str.split(':')) + if seconds is not None: + markers.append({'seconds': seconds, 'text': ' '.join(text_lines)}) + return markers + +def parse_srt(text: str) -> list[dict]: + """Parse SRT subtitle format into timestamp/text pairs.""" + return _extract_subtitle_blocks(text) + +def parse_vtt(text: str) -> list[dict]: + """Parse WebVTT subtitle format into timestamp/text pairs.""" + text = re.sub(r'^WEBVTT.*?\n', '', text, flags=re.MULTILINE) + text = re.sub(r'NOTE\n.*?\n\n', '', text, flags=re.DOTALL) + return _extract_subtitle_blocks(text, strip_vtt_tags=True) + +def parse_transcript_timestamps(text: str) -> list[dict]: + """Parse timestamped text (YouTube description format) into markers. + + Supports formats like: + 0:00 Introduction + 00:01:30 Main Topic + 1:05:30 Conclusion + 00:00:00:00 SMPTE timecode + """ + markers = [] + for line in text.strip().split('\n'): + line = line.strip() + if not line: + continue + match = re.match(r'^(\d{1,2}:\d{2}(?::\d{2}){0,2})\s+(.+)$', line) + if match: + seconds = _parse_timestamp_parts(match.group(1).split(':')) + if seconds is not None: + markers.append({'seconds': seconds, 'text': match.group(2).strip()}) + return markers diff --git a/code/server_tools/_shared/detection.py b/code/server_tools/_shared/detection.py new file mode 100644 index 0000000..b52dfa3 --- /dev/null +++ b/code/server_tools/_shared/detection.py @@ -0,0 +1,115 @@ +"""Detecção para QC: flash frames, buracos e clipes duplicados. + +Extraído de _shared.py — ver server_tools/_shared/__init__.py. +""" + +from __future__ import annotations + +from typing import Any + +from fcpxml.models import ( + DuplicateGroup, + FlashFrame, + FlashFrameSeverity, + GapInfo, + Timecode, +) + +from .formatting import format_timecode + + +def _detect_flash_frames( + tl: Any, *, critical_threshold: int = 2, warning_threshold: int = 6, +) -> list: + """Find clips shorter than *warning_threshold* frames. + + Returns a list of ``FlashFrame`` objects sorted by severity. Shared by + ``handle_detect_flash_frames`` and ``handle_validate_timeline`` so the + detection logic lives in exactly one place. + """ + fps = tl.frame_rate + flash_frames: list[FlashFrame] = [] + for clip in tl.clips: + duration_frames = int(clip.duration_seconds * fps) + if duration_frames < warning_threshold: + severity = ( + FlashFrameSeverity.CRITICAL + if duration_frames < critical_threshold + else FlashFrameSeverity.WARNING + ) + flash_frames.append(FlashFrame( + clip_name=clip.name, clip_id=clip.name, + start=clip.start, duration_frames=duration_frames, + duration_seconds=clip.duration_seconds, severity=severity, + )) + return flash_frames + +def _detect_gaps(tl: Any, *, min_gap_frames: int = 1) -> list: + """Find inter-clip gaps of at least *min_gap_frames* length. + + Returns a list of ``GapInfo`` objects. Shared by ``handle_detect_gaps`` + and ``handle_validate_timeline``. + """ + fps = tl.frame_rate + min_gap_seconds = min_gap_frames / fps + gaps: list[GapInfo] = [] + sorted_clips = sorted(tl.clips, key=lambda c: c.start.seconds) + for i in range(len(sorted_clips) - 1): + current_end = sorted_clips[i].end.seconds + next_start = sorted_clips[i + 1].start.seconds + gap_duration = next_start - current_end + if gap_duration >= min_gap_seconds: + gaps.append(GapInfo( + start=Timecode(frames=int(current_end * fps), frame_rate=fps), + duration_frames=int(gap_duration * fps), + duration_seconds=gap_duration, + previous_clip=sorted_clips[i].name, + next_clip=sorted_clips[i + 1].name, + )) + return gaps + +def _detect_duplicate_groups(tl: Any, *, mode: str = "same_source") -> list: + """Group clips that share a source media reference. + + Returns a list of ``DuplicateGroup`` objects. Shared by + ``handle_detect_duplicates`` and ``handle_validate_timeline``. + """ + source_groups: dict[str, list[dict]] = {} + for clip in tl.clips: + source_key = clip.media_path or clip.name + if source_key not in source_groups: + source_groups[source_key] = [] + source_groups[source_key].append({ + 'name': clip.name, + 'start': clip.start.seconds, + 'duration': clip.duration_seconds, + 'source_start': clip.source_start.seconds if clip.source_start else 0, + 'source_duration': clip.duration_seconds, + 'timecode': format_timecode(clip.start), + }) + + duplicates: list[DuplicateGroup] = [] + for source_key, clips in source_groups.items(): + if len(clips) <= 1: + continue + group = DuplicateGroup( + source_ref=source_key, + source_name=source_key.split('/')[-1] if '/' in source_key else source_key, + clips=clips, + ) + if mode == "same_source": + duplicates.append(group) + elif mode == "overlapping_ranges" and group.has_overlapping_ranges: + duplicates.append(group) + elif mode == "identical": + seen_ranges: set[tuple] = set() + identical_clips = [] + for c in clips: + range_key = (c['source_start'], c['source_duration']) + if range_key in seen_ranges: + identical_clips.append(c) + seen_ranges.add(range_key) + if identical_clips: + group.clips = identical_clips + duplicates.append(group) + return duplicates diff --git a/code/server_tools/_shared/formatting.py b/code/server_tools/_shared/formatting.py new file mode 100644 index 0000000..e92092b --- /dev/null +++ b/code/server_tools/_shared/formatting.py @@ -0,0 +1,114 @@ +"""Formatação do texto que os handlers devolvem — tabelas e relatórios. + +Extraído de _shared.py — ver server_tools/_shared/__init__.py. +""" + +from __future__ import annotations + +from typing import Sequence + + +def format_timecode(tc) -> str: + """Format a Timecode object to SMPTE string.""" + return tc.to_smpte() if tc else "00:00:00:00" + +def format_duration(seconds: float) -> str: + """Format seconds into human-readable duration.""" + if seconds < 1: + return f"{seconds*1000:.0f}ms" + elif seconds < 60: + return f"{seconds:.2f}s" + return f"{int(seconds // 60)}m {seconds % 60:.1f}s" + +def _format_clip_table(clips: list, header: str) -> str: + """Render a list of clips as a markdown table with timecodes and durations. + + Shared by handlers that filter clips by duration threshold + (find_short_cuts, find_long_clips). + """ + result = f"{header}\n\n| Name | TC | Duration |\n|------|----|---------|\n" + result += "\n".join( + f"| {c.name} | {format_timecode(c.start)} | {format_duration(c.duration_seconds)} |" + for c in clips + ) + return result + +def _markdown_table(headers: list[str], rows: list[list[str]]) -> str: + """Build a markdown table from headers and rows. + + Returns header row, separator row, and data rows as a single string. + Callers avoid repeating the ``| H1 | H2 |\\n|---|---|`` boilerplate + that appears in 15+ handlers. + """ + header_line = "| " + " | ".join(headers) + " |" + sep_line = "|" + "|".join("------" for _ in headers) + "|" + data_lines = "\n".join( + "| " + " | ".join(str(c) for c in row) + " |" for row in rows + ) + return f"{header_line}\n{sep_line}\n{data_lines}" + +def _format_batch_result( + title: str, + summary: dict[str, str], + headers: list[str], + rows: list[list[str]], + output_path: str, +) -> str: + """Build a standard batch-operation result with summary, table, and save footer. + + Used by batch fix handlers (flash frames, rapid trim, fill gaps) that all + share the same markdown structure: ``# Title → ## Summary → ## Details table + → Saved to`` footer. + """ + summary_lines = "\n".join(f"- **{k}**: {v}" for k, v in summary.items()) + table = _markdown_table(headers, rows) + return ( + f"# {title}\n\n" + f"## Summary\n{summary_lines}\n\n" + f"## Details\n{table}\n\n" + f"Saved to: `{output_path}`" + ) + +def _fmt_suggestions(suggestions: list[str]) -> str: + """Format pacing suggestions as markdown list (Python 3.10 compatible).""" + if not suggestions: + return "- Pacing looks good!" + nl = "\n" + return nl.join(f"- {s}" for s in suggestions) + +def _voice_analysis_config_text(config: dict) -> str: + w = config["emphasis_weights"] + text = "# Voice Analysis Settings\n\n" + text += _markdown_table( + ["Setting", "Value"], + [ + ["Energy threshold", f"{config['energy_threshold']:.2f}"], + ["Peak selection", f"top {config['peak_percentile']:.1%} of words"], + ["Emphasis floor", f"{config['emphasis_floor']:.2f}"], + ["Emotion detection", "on" if config["emotion_enabled"] else "off"], + ["Emotion sensitivity", f"{config['emotion_sensitivity']:.2f}"], + ], + ) + "\n\n## Emphasis Weights\n" + text += _markdown_table( + ["Factor", "Weight"], + [[k.replace("_", " ").title(), f"{v:.2f}"] for k, v in w.items()], + ) + return text + +def _speaker_table(profiles: Sequence[dict]) -> str: + """Who was detected, ordered by how much of the runtime each holds.""" + return _markdown_table( + ["ID", "Name", "Share", "Speaking", "Lines", "Avg line"], + [ + [ + p["id"], + p.get("name", ""), + f"{p['share']:.0%}", + format_duration(p["speaking_seconds"]), + str(p["segment_count"]), + f"{p['avg_segment']:.1f}s", + ] + for p in profiles + ], + ) + diff --git a/code/server_tools/_shared/media.py b/code/server_tools/_shared/media.py new file mode 100644 index 0000000..2db5a19 --- /dev/null +++ b/code/server_tools/_shared/media.py @@ -0,0 +1,274 @@ +"""Mídia e transcrição: cache, corte por trecho falado e ações posicionadas. + +Extraído de _shared.py — ver server_tools/_shared/__init__.py. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +from fcpxml.media_intel import media_src_to_path +from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config +from fcpxml.models import ( + TimeValue, +) +from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text +from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe + +from .formatting import _markdown_table, format_duration +from .paths import _validate_output_path +from .project import _text_result + +AUDIO_MEDIA_EXTENSIONS = ( + '.wav', '.aif', '.aiff', '.mp3', '.m4a', '.aac', '.flac', '.mov', '.mp4', +) + +_DIARIZATION_INSTALL_HINT = ( + "\n\nInstall the optional diarization extra:\n\n" + " pip install 'fcp-mcp-server[diarization]'\n\n" + "and set a HuggingFace token with access to " + "pyannote/speaker-diarization-3.1 (pass hf_token= or persist one via " + "save_hf_token)." +) + +_FEATURES_INSTALL_HINT = ( + "\n\nInstall the optional media-intelligence extra:\n\n" + " pip install 'fcp-mcp-server[intelligence]'" +) + +def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str: + """Apply one non-cut action to the clip that hosts it. + + ``clip_start`` is where that clip begins on the timeline; the writer + wants times relative to the clip's own head, so the rebase happens here + — the single place that knows about the conversion. The clip *element* + is passed through rather than its name: after a cut the pieces share a + name, and a name lookup would land every edit on the first piece. + """ + rel_start = action.start - clip_start + rel_end = action.end - clip_start + + if action.kind == "zoom": + config = load_voice_analysis_config() + # Only forward an explicit ease — otherwise add_zoom's own default + # (a fast ramp in, instant snap back out) is what should apply. + zoom_args = { + "ease": float(action.params.get("ease", config["zoom_ease_in"])), + "ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])), + } + mode = str(action.params.get("mode", config["zoom_mode"])) + if mode == "in": + zoom_args["hold_at_end"] = True + zoom_args["start_at_peak"] = False + elif mode == "out": + zoom_args["hold_at_end"] = False + zoom_args["start_at_peak"] = True + elif mode == "in_out": + zoom_args["hold_at_end"] = False + zoom_args["start_at_peak"] = False + modifier.add_zoom( + clip_id=clip_el, + start=rel_start, + end=rel_end, + scale=float(action.params.get("scale", config["zoom_scale"])), + **zoom_args, + ) + return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x" + + if action.kind == "text": + # Default to the "Legendas Dinâmicas" emphasis style (the font used + # to highlight a word in the captions) rather than a hardcoded + # Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of + # the video's on-screen text instead of looking like a stray default + # title. Any of these the action itself specifies still wins. + subtitle_cfg = load_dynamic_subtitle_config() + font = action.params.get("font", subtitle_cfg["emphasis_font"]) + face = action.params.get("face", subtitle_cfg["emphasis_face"]) + font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0) + requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"])) + requested_kerning = float(action.params.get("kerning", 0.0) or 0.0) + + # Voice-action callouts are not part of the dynamic subtitle block. + # When omitted, put them above the subtitle band and shrink wide + # phrases to the title-safe width. The previous default (Position 0 0, + # full emphasis size) made long callouts like "PRÓTESES DE SILICONE" + # collide with captions and run off both sides of a vertical frame. + emitted_size = requested_size * font_scale + emitted_kerning = requested_kerning * font_scale + safe_width = modifier.frame_width() * 0.90 + width = measure_text( + action.params["content"], + emitted_size, + bold=bool(action.params.get("bold", False)), + kerning=emitted_kerning, + font=font, + face=face, + ) + font_size = requested_size + if width > safe_width and width > 0: + font_size = max(32, int(requested_size * safe_width / width)) + position = action.params.get("position") + if not position: + position = f"0 {modifier.frame_height() * 0.23:g}" + + modifier.add_text_title( + clip_el, + action.params["content"], + offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(), + duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(), + position=position, + font=font, + font_size=font_size, + font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]), + face=face, + bold=action.params.get("bold", False), + ) + return f"text \"{action.params['content'][:24]}\"" + + # marker + modifier.add_marker( + clip_id=clip_el, + timecode=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(), + name=action.params.get("content") or action.reason or "Voice action", + note=action.reason or None, + ) + return "marker" + +TRANSCRIBE_MAX_MEDIA = 10 + +_TRANSCRIBE_INSTALL_HINT = ( + "\n\nInstall the optional transcription extra:\n\n" + " pip install 'fcp-mcp-server[transcribe]'\n\n" + "or run via uvx:\n\n" + " uvx --from \"fcp-mcp-server[transcribe]\" fcp-mcp-server" +) + +def _transcript_json_path(media_path: str, output_dir: str | None = None) -> Path: + """Where the ``_transcript.json`` for ``media_path`` lives. + + When ``output_dir`` (the user-selected project folder) is set, the + transcript is saved/read there instead of next to the source media. + """ + p = Path(media_path) + if output_dir: + directory = Path(output_dir).expanduser() + directory.mkdir(parents=True, exist_ok=True) + return directory / f"{p.stem}_transcript.json" + return p.with_name(p.stem + "_transcript.json") + +def _load_or_transcribe( + media_path: str, model: str, language: str | None, output_dir: str | None = None +) -> tuple[dict | None, str]: + """Load a cached ``_transcript.json`` for a media file, else transcribe and cache it. + + Returns ``(transcript, "")`` or ``(None, reason)``. The cache makes + transcription a one-time cost per media file across all transcript tools. + """ + json_path = _transcript_json_path(media_path, output_dir) + if json_path.is_file(): + try: + with open(json_path) as f: + data = json.load(f) + if isinstance(data, dict) and isinstance(data.get("words"), list): + return data, "" + except (OSError, json.JSONDecodeError, UnicodeDecodeError): + pass # unreadable cache falls through to re-transcribe + result = transcribe(media_path, model_size=model, language=language) + if result is None: + return None, "untranscribable (faster-whisper not installed or media unreadable)" + anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent) + out_path = _validate_output_path(str(json_path), anchor_dir=anchor) + with open(out_path, "w") as f: + json.dump({"source": Path(media_path).name, **result}, f, indent=2) + return result, "" + +def _cut_transcript_spans(modifier, clip_filter, model, language, padding, spans_fn, keep_only=False, output_dir=None): + """Shared cut engine for transcript-driven editing. + + ``spans_fn(words) -> [(start, end), ...]`` in source seconds. Spans are + padded, clamped to each clip's used source window, optionally inverted + (keep_only), snapped to the frame grid, and cut with ripple. + """ + to_frame = modifier.snap_seconds_to_frame + + cache: dict[str, tuple] = {} + cuts_made: list[tuple[str, int, float]] = [] + skipped: list[tuple[str, str]] = [] + spine_clips = [el for _, el in modifier._iter_spine_clips()] + for el in spine_clips: + name = el.get("name", "") + if clip_filter and name != clip_filter: + continue + src = modifier.resources.get(el.get("ref", ""), {}).get("src", "") + media_path = media_src_to_path(src) + if not media_path or not Path(media_path).is_file(): + skipped.append((name, "media file missing")) + continue + if media_path not in cache: + if len(cache) >= TRANSCRIBE_MAX_MEDIA: + skipped.append((name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)")) + continue + cache[media_path] = _load_or_transcribe(media_path, model, language, output_dir) + data, reason = cache[media_path] + if data is None: + skipped.append((name, reason)) + continue + + clip_source_start = modifier.source_file_start(el).to_seconds() + clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() + window_start = clip_source_start + window_end = clip_source_start + clip_duration + + spans = spans_fn(data.get("words", [])) + padded = merge_ranges([(s - padding, e + padding) for s, e in spans]) + clamped = [ + (max(s, window_start), min(e, window_end)) + for s, e in padded + if min(e, window_end) > max(s, window_start) + ] + if keep_only: + if not clamped: + # Never delete a whole clip just because nothing matched in it. + skipped.append((name, "no phrase matches — left untouched (keep_only)")) + continue + cut_source = invert_ranges(clamped, window_start, window_end) + else: + cut_source = clamped + cut_ranges = [ + (to_frame(s - clip_source_start), to_frame(e - clip_source_start)) + for s, e in cut_source + ] + cut_ranges = [(a, b) for a, b in cut_ranges if b > a] + if not cut_ranges: + continue + removed = modifier.cut_clip_ranges(el, cut_ranges) + if removed > TimeValue.zero(): + cuts_made.append((name, len(cut_ranges), removed.to_seconds())) + return cuts_made, skipped + +def _transcript_cut_report(title, summary_lines, cuts_made, skipped, output_path, footer): + if not cuts_made: + text = f"# {title}\n\nNo cuts to make — file unchanged (nothing saved)." + if skipped: + text += "\n\n## Skipped Clips\n" + _markdown_table( + ["Clip", "Reason"], [[name, reason] for name, reason in skipped] + ) + if any("faster-whisper" in reason for _, reason in skipped): + text += _TRANSCRIBE_INSTALL_HINT + return _text_result(text) + total_removed = sum(seconds for _, _, seconds in cuts_made) + result = f"# {title}\n\n## Summary\n" + result += "\n".join(summary_lines) + "\n" + result += f"- **Clips Cut**: {len(cuts_made)}\n- **Total Removed**: {format_duration(total_removed)}\n" + result += "\n## Cuts\n" + result += _markdown_table( + ["Clip", "Ranges Cut", "Removed"], + [[name, str(count), f"{seconds:.2f}s"] for name, count, seconds in cuts_made], + ) + "\n" + if skipped: + result += "\n## Skipped Clips\n" + _markdown_table( + ["Clip", "Reason"], [[name, reason] for name, reason in skipped] + ) + "\n" + result += f"\nSaved to: {output_path}\n\n{footer}" + return _text_result(result) diff --git a/code/server_tools/_shared/paths.py b/code/server_tools/_shared/paths.py new file mode 100644 index 0000000..e741c15 --- /dev/null +++ b/code/server_tools/_shared/paths.py @@ -0,0 +1,226 @@ +"""Caminhos: validação contra a sandbox, limites de tamanho, saída derivada. + +Extraído de _shared.py — ver server_tools/_shared/__init__.py. +""" + +from __future__ import annotations + +import os +import re +from pathlib import Path + +PROJECTS_DIR = os.environ.get("FCP_PROJECTS_DIR", os.path.expanduser("~/Movies")) + +_SANDBOX_ENABLED = "FCP_PROJECTS_DIR" in os.environ + +MAX_FILE_SIZE = 100 * 1024 * 1024 + +MAX_MEDIA_FILE_SIZE = 32 * 1024 * 1024 * 1024 + +_MAX_JSON_DEPTH = 50 + +def _check_json_depth(obj: object, _depth: int = 0) -> None: + """Reject JSON structures nested beyond _MAX_JSON_DEPTH. + + Prevents denial-of-service via deeply nested objects that exhaust the + call stack or memory during downstream processing. Called after + json.load() since Python's json module has no built-in depth limit. + """ + if _depth > _MAX_JSON_DEPTH: + raise ValueError( + f"JSON nesting depth exceeds {_MAX_JSON_DEPTH} — " + "file may be malformed or adversarial" + ) + if isinstance(obj, dict): + for v in obj.values(): + _check_json_depth(v, _depth + 1) + elif isinstance(obj, list): + for item in obj: + _check_json_depth(item, _depth + 1) + +def _validate_filepath( + filepath: str, + allowed_extensions: tuple[str, ...] | None = None, + max_size: int = MAX_FILE_SIZE, +) -> str: + """Validate a user-provided file path against traversal and size attacks. + + Resolves symlinks, blocks null bytes, enforces extension whitelist, and + checks file size before any parsing takes place. + + ``max_size`` defaults to the document limit; callers handling source + media pass ``MAX_MEDIA_FILE_SIZE``, since media is streamed rather than + parsed into memory (see the constant for why). + + Raises: + ValueError: For invalid paths (null bytes, bad extensions, oversized). + FileNotFoundError: When the resolved path does not exist. + """ + if '\x00' in filepath: + raise ValueError("Invalid file path: null byte detected") + + resolved = Path(filepath).resolve() + + if not resolved.exists(): + raise FileNotFoundError(f"File not found: {filepath}") + + # .fcpxmld bundles are directories (a package wrapping Info.fcpxml plus + # sidecar data files for object tracking / Cinematic mode). The size + # check applies to the inner Info.fcpxml, which is what gets parsed. + if resolved.is_dir(): + if resolved.suffix.lower() != '.fcpxmld': + raise ValueError(f"Not a regular file: {filepath}") + inner = resolved / 'Info.fcpxml' + if not inner.is_file(): + raise ValueError(f"Invalid bundle (no Info.fcpxml): {filepath}") + size_target = inner + elif not resolved.is_file(): + raise ValueError(f"Not a regular file: {filepath}") + else: + size_target = resolved + + if allowed_extensions and resolved.suffix.lower() not in allowed_extensions: + raise ValueError( + f"Invalid file type '{resolved.suffix}'. " + f"Allowed: {', '.join(allowed_extensions)}" + ) + + if size_target.stat().st_size > max_size: + size_mb = size_target.stat().st_size / (1024 * 1024) + raise ValueError(f"File too large ({size_mb:.1f} MB). Maximum: {max_size // (1024 * 1024)} MB") + + return str(resolved) + +def _validate_output_path(output_path: str, *, anchor_dir: str | None = None) -> str: + """Validate an output path with optional sandbox enforcement. + + Resolves traversal, blocks null bytes, ensures parent exists, and — when + *anchor_dir* is provided — verifies the resolved output lives under that + directory. This prevents LLM-generated tool calls from writing to + arbitrary filesystem locations (e.g. ``/etc/cron.d/backdoor``). + + Args: + output_path: The raw output path to validate. + anchor_dir: If set, the resolved output must be a child of this + directory. Typically the parent directory of the input file so + outputs stay co-located with their sources. + + Raises: + ValueError: For null bytes, missing parent, or sandbox escape. + """ + if '\x00' in output_path: + raise ValueError("Invalid output path: null byte detected") + + resolved = Path(output_path).resolve() + + if not resolved.parent.exists(): + raise ValueError(f"Output directory does not exist: {resolved.parent}") + + if anchor_dir is not None: + anchor = Path(anchor_dir).resolve() + try: + resolved.relative_to(anchor) + except ValueError: + raise ValueError( + f"Output path escapes allowed directory: " + f"{resolved} is not under {anchor}" + ) + + return str(resolved) + +def _validate_directory(directory: str, *, allowed_root: str | None = None) -> str: + """Validate a user-provided directory path against traversal and injection. + + Resolves symlinks, blocks null bytes, and verifies the path is a real + directory. When *allowed_root* is given, the resolved path must be a + descendant of (or equal to) that root — preventing filesystem enumeration + beyond the project workspace. + + Raises: + ValueError: For invalid paths (null bytes, not a directory, sandbox escape). + """ + if '\x00' in directory: + raise ValueError("Invalid directory path: null byte detected") + + resolved = Path(directory).resolve() + + if not resolved.is_dir(): + raise ValueError(f"Not a valid directory: {directory}") + + if allowed_root is not None: + root = Path(allowed_root).resolve() + try: + resolved.relative_to(root) + except ValueError: + raise ValueError( + f"Directory escapes allowed root: " + f"{resolved} is not under {root}" + ) + + return str(resolved) + +def find_fcpxml_files(directory: str) -> list[str]: + """Find all FCPXML files in a directory.""" + path = Path(directory) + files = list(str(f) for f in path.rglob("*.fcpxml")) + files.extend(str(f) for f in path.rglob("*.fcpxmld")) + return sorted(files) + +def generate_output_path(input_path: str, suffix: str = "_modified") -> str: + """Generate output path from input path. + + The suffix is sanitized to prevent path-component injection — only + alphanumeric, hyphen, underscore, and dot characters survive. + """ + # Strip anything that could inject path separators or traversal sequences + clean_suffix = re.sub(r'[^a-zA-Z0-9._-]', '', suffix) + if not clean_suffix: + clean_suffix = "_modified" + p = Path(input_path) + return str(p.parent / f"{p.stem}{clean_suffix}{p.suffix}") + +def _resolve_io_paths( + arguments: dict, + suffix: str = "_modified", +) -> tuple[str, str]: + """Validate input filepath and resolve the output path. + + Shared foundation for every handler that reads an FCPXML and writes + a derived file. Validates the input, falls back to a suffixed + output name when ``output_path`` is not supplied, and sandbox-checks + the result. + + Args: + arguments: Tool arguments dict (must contain ``filepath``; may + contain ``output_path``). + suffix: Default output filename suffix when ``output_path`` is + not provided (e.g. ``"_modified"``, ``"_beats"``). + + Returns: + ``(filepath, output_path)`` tuple with both paths validated. + """ + filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld')) + # Anchor write operations to the input file's directory so LLM-generated + # tool calls cannot write to arbitrary filesystem locations (e.g. + # /etc/cron.d/backdoor). When the explicit sandbox is off, the anchor + # still prevents writes outside the source directory tree. + # `output_dir` is where the caller wants the file written, not merely a + # sandbox boundary: the app's "Pasta do projeto" promises that everything + # generated lands there. Deriving the name from the input but keeping the + # input's directory made every cross-directory call fail its own anchor + # check ("output path escapes allowed directory"), so the setting silently + # only worked when it pointed at the directory the file was already going + # to. An explicit `output_path` still wins, and still has to sit inside + # the anchor. + output_dir = arguments.get("output_dir") + if output_dir: + anchor = _validate_directory(str(output_dir)) + default_output = str(Path(anchor) / Path(generate_output_path(filepath, suffix)).name) + else: + anchor = str(Path(filepath).resolve().parent) + default_output = generate_output_path(filepath, suffix) + output_path = _validate_output_path( + arguments.get("output_path") or default_output, + anchor_dir=anchor, + ) + return filepath, output_path diff --git a/code/server_tools/_shared/project.py b/code/server_tools/_shared/project.py new file mode 100644 index 0000000..9f967fa --- /dev/null +++ b/code/server_tools/_shared/project.py @@ -0,0 +1,89 @@ +"""Abrir um projeto e preparar modifier/generator para editá-lo. + +Extraído de _shared.py — ver server_tools/_shared/__init__.py. +""" + +from __future__ import annotations + +from mcp.types import TextContent + +from fcpxml.parser import FCPXMLParser +from fcpxml.rough_cut import RoughCutGenerator +from fcpxml.writer import FCPXMLModifier + +from .paths import _resolve_io_paths, _validate_filepath + + +def _parse_project(filepath: str): + """Parse an FCPXML file and return the project with its primary timeline.""" + filepath = _validate_filepath(filepath, ('.fcpxml', '.fcpxmld')) + project = FCPXMLParser().parse_file(filepath) + if not project.timelines: + return None, None + return project, project.primary_timeline + +def _text_result(text: str) -> list[TextContent]: + """Wrap a string in the MCP TextContent list that every tool handler returns.""" + return [TextContent(type="text", text=text)] + +def _no_timeline(): + """Standard response when no timelines are found.""" + return _text_result("No timelines found") + +def _require_timeline(filepath: str): + """Parse FCPXML and return (project, timeline), raising if no timeline exists. + + Centralises the repeated _parse_project + _no_timeline guard that + appears in every read-only timeline handler. Returns a tuple so + callers can destructure directly:: + + project, tl = _require_timeline(arguments["filepath"]) + """ + project, tl = _parse_project(filepath) + if not tl: + raise _NoTimelineError() + return project, tl + +class _NoTimelineError(Exception): + """Sentinel raised by _require_timeline when no timelines exist.""" + +def _setup_modifier( + arguments: dict, + suffix: str = "_modified", +) -> tuple[str, str, "FCPXMLModifier"]: + """Common setup for write handlers: validate paths and create modifier. + + Consolidates the repeated validate-filepath → resolve-output-path → + create-modifier boilerplate shared by 18+ write handlers. + + Args: + arguments: Tool arguments dict (must contain ``filepath``; may + contain ``output_path``). + suffix: Default output filename suffix when ``output_path`` is + not provided (e.g. ``"_modified"``, ``"_flash_fixed"``). + + Returns: + ``(filepath, output_path, modifier)`` tuple ready for the + handler's domain-specific operation. + """ + filepath, output_path = _resolve_io_paths(arguments, suffix) + modifier = FCPXMLModifier(filepath) + return filepath, output_path, modifier + +def _setup_generator( + arguments: dict, + suffix: str = "_roughcut", +) -> tuple[str, str, "RoughCutGenerator"]: + """Common setup for generation handlers: validate paths and create generator. + + Args: + arguments: Tool arguments dict (must contain ``filepath`` and + ``output_path``). + suffix: Default output filename suffix. + + Returns: + ``(filepath, output_path, generator)`` tuple. + """ + filepath, output_path = _resolve_io_paths(arguments, suffix) + generator = RoughCutGenerator(filepath) + return filepath, output_path, generator diff --git a/code/tests/test_diarize_media_tool.py b/code/tests/test_diarize_media_tool.py index cf5c584..bb373bc 100644 --- a/code/tests/test_diarize_media_tool.py +++ b/code/tests/test_diarize_media_tool.py @@ -46,7 +46,7 @@ class TestDiarizeMediaHandler: await handle_diarize_media({"media_path": str(bad)}) async def test_writes_diarization_json_and_reports(self, tmp_path, monkeypatch): - import server_tools._shared as _shared_mod + import server_tools._shared.media as _shared_mod import server_tools.voice as server_mod from server import handle_diarize_media @@ -86,7 +86,7 @@ class TestDiarizeMediaHandler: assert data["words"][1]["speaker_id"] == "SPEAKER_01" async def test_reports_when_diarization_fails(self, tmp_path, monkeypatch): - import server_tools._shared as _shared_mod + import server_tools._shared.media as _shared_mod import server_tools.voice as server_mod from server import handle_diarize_media diff --git a/code/tests/test_refine_voice_timeline_tool.py b/code/tests/test_refine_voice_timeline_tool.py index 1330bbf..c4baafd 100644 --- a/code/tests/test_refine_voice_timeline_tool.py +++ b/code/tests/test_refine_voice_timeline_tool.py @@ -32,7 +32,7 @@ def two_candidates(monkeypatch): than one candidate to cap. """ import fcpxml.voice_timeline as vt - import server_tools._shared as _shared_mod + import server_tools._shared.media as _shared_mod transcript = { "language": "pt", diff --git a/code/tests/test_voice_features_tool.py b/code/tests/test_voice_features_tool.py index ffb5b6f..4286bf6 100644 --- a/code/tests/test_voice_features_tool.py +++ b/code/tests/test_voice_features_tool.py @@ -43,7 +43,7 @@ def wav(tmp_path): @pytest.fixture def patched_analysis(monkeypatch): """Make the tool's transcription + librosa extractors deterministic.""" - import server_tools._shared as _shared_mod + import server_tools._shared.media as _shared_mod import server_tools.voice as server_mod monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT) @@ -111,7 +111,7 @@ class TestAnalyzeVoiceFeaturesHandler: assert "emphasis_weights" in data["config"] async def test_empty_transcript_reports_instead_of_crashing(self, wav, monkeypatch): - import server_tools._shared as _shared_mod + import server_tools._shared.media as _shared_mod import server_tools.voice as server_mod from server import handle_analyze_voice_features @@ -123,7 +123,7 @@ class TestAnalyzeVoiceFeaturesHandler: assert "no words" in result[0].text.lower() async def test_untranscribable_media_reports_install_hint(self, wav, monkeypatch): - import server_tools._shared as _shared_mod + import server_tools._shared.media as _shared_mod import server_tools.voice as server_mod from server import handle_analyze_voice_features @@ -133,7 +133,7 @@ class TestAnalyzeVoiceFeaturesHandler: assert "faster-whisper" in result[0].text async def test_missing_pitch_track_degrades_without_crashing(self, wav, monkeypatch): - import server_tools._shared as _shared_mod + import server_tools._shared.media as _shared_mod import server_tools.voice as server_mod from server import handle_analyze_voice_features diff --git a/code/tests/test_voice_timeline_tool.py b/code/tests/test_voice_timeline_tool.py index 42f5acd..9ca40c7 100644 --- a/code/tests/test_voice_timeline_tool.py +++ b/code/tests/test_voice_timeline_tool.py @@ -34,7 +34,7 @@ def wav(tmp_path): def patched(monkeypatch): """Deterministic transcription + acoustics, no optional extras needed.""" import fcpxml.voice_timeline as vt - import server_tools._shared as _shared_mod + import server_tools._shared.media as _shared_mod monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _TRANSCRIPT) monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: [(2.4, 260.0), (0.2, 120.0)]) @@ -71,7 +71,7 @@ class TestBuildVoiceTimelineHandler: await handle_build_voice_timeline({"media_path": str(bad)}) async def test_untranscribable_media_reports_hint(self, wav, monkeypatch): - import server_tools._shared as _shared_mod + import server_tools._shared.media as _shared_mod from server import handle_build_voice_timeline monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)