refactor: _shared.py vira subpacote, um módulo por papel
Eram 882 linhas de seis papéis sem relação, sob um nome que só dizia
"compartilhado" — o depósito onde tudo que servia a mais de um handler
acabava caindo.
media 316 transcrição em cache, corte por fala, relatório
paths 206 sandbox, limites, caminho de saída
project 116 abrir projeto, preparar modifier/generator
captions 112 SRT, VTT, listas com timestamp
detection 99 flash frames, buracos, duplicados
formatting 86 tabelas e relatórios dos handlers
O __init__ reexporta os 46 nomes, então os treze pontos que importam daqui
não mudaram.
_transcript_cut_report saiu de formatting para media: ele precisa do hint de
instalação e do _text_result, ou seja, é relatório de transcrição e não
formatação genérica — mover foi mais honesto que cruzar imports entre os
dois módulos.
Quatro testes patchavam `server_tools._shared.transcribe`; o nome agora é
ligado por _shared/media.py, então o patch passou a apontar para lá — mesmo
padrão da experiência #23.
Lint zerado, 1454 testes passando.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
368bb62706
commit
ffaebb3f72
@@ -1,882 +0,0 @@
|
|||||||
"""Shared internal helpers used by tool handlers across categories.
|
|
||||||
|
|
||||||
Extracted from server.py — validation, formatting, and small parsing utilities
|
|
||||||
that more than one server_tools/*.py module needs.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
|
||||||
import os
|
|
||||||
import re
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Any, Sequence
|
|
||||||
|
|
||||||
from mcp.types import TextContent
|
|
||||||
|
|
||||||
from fcpxml.media_intel import media_src_to_path
|
|
||||||
from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config
|
|
||||||
from fcpxml.models import (
|
|
||||||
DuplicateGroup,
|
|
||||||
FlashFrame,
|
|
||||||
FlashFrameSeverity,
|
|
||||||
GapInfo,
|
|
||||||
Timecode,
|
|
||||||
TimeValue,
|
|
||||||
)
|
|
||||||
from fcpxml.parser import FCPXMLParser
|
|
||||||
from fcpxml.rough_cut import RoughCutGenerator
|
|
||||||
from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text
|
|
||||||
from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe
|
|
||||||
from fcpxml.writer import FCPXMLModifier
|
|
||||||
|
|
||||||
PROJECTS_DIR = os.environ.get("FCP_PROJECTS_DIR", os.path.expanduser("~/Movies"))
|
|
||||||
|
|
||||||
_SANDBOX_ENABLED = "FCP_PROJECTS_DIR" in os.environ
|
|
||||||
|
|
||||||
MAX_FILE_SIZE = 100 * 1024 * 1024
|
|
||||||
|
|
||||||
MAX_MEDIA_FILE_SIZE = 32 * 1024 * 1024 * 1024
|
|
||||||
|
|
||||||
_MAX_JSON_DEPTH = 50
|
|
||||||
|
|
||||||
def _check_json_depth(obj: object, _depth: int = 0) -> None:
|
|
||||||
"""Reject JSON structures nested beyond _MAX_JSON_DEPTH.
|
|
||||||
|
|
||||||
Prevents denial-of-service via deeply nested objects that exhaust the
|
|
||||||
call stack or memory during downstream processing. Called after
|
|
||||||
json.load() since Python's json module has no built-in depth limit.
|
|
||||||
"""
|
|
||||||
if _depth > _MAX_JSON_DEPTH:
|
|
||||||
raise ValueError(
|
|
||||||
f"JSON nesting depth exceeds {_MAX_JSON_DEPTH} — "
|
|
||||||
"file may be malformed or adversarial"
|
|
||||||
)
|
|
||||||
if isinstance(obj, dict):
|
|
||||||
for v in obj.values():
|
|
||||||
_check_json_depth(v, _depth + 1)
|
|
||||||
elif isinstance(obj, list):
|
|
||||||
for item in obj:
|
|
||||||
_check_json_depth(item, _depth + 1)
|
|
||||||
|
|
||||||
def _validate_filepath(
|
|
||||||
filepath: str,
|
|
||||||
allowed_extensions: tuple[str, ...] | None = None,
|
|
||||||
max_size: int = MAX_FILE_SIZE,
|
|
||||||
) -> str:
|
|
||||||
"""Validate a user-provided file path against traversal and size attacks.
|
|
||||||
|
|
||||||
Resolves symlinks, blocks null bytes, enforces extension whitelist, and
|
|
||||||
checks file size before any parsing takes place.
|
|
||||||
|
|
||||||
``max_size`` defaults to the document limit; callers handling source
|
|
||||||
media pass ``MAX_MEDIA_FILE_SIZE``, since media is streamed rather than
|
|
||||||
parsed into memory (see the constant for why).
|
|
||||||
|
|
||||||
Raises:
|
|
||||||
ValueError: For invalid paths (null bytes, bad extensions, oversized).
|
|
||||||
FileNotFoundError: When the resolved path does not exist.
|
|
||||||
"""
|
|
||||||
if '\x00' in filepath:
|
|
||||||
raise ValueError("Invalid file path: null byte detected")
|
|
||||||
|
|
||||||
resolved = Path(filepath).resolve()
|
|
||||||
|
|
||||||
if not resolved.exists():
|
|
||||||
raise FileNotFoundError(f"File not found: {filepath}")
|
|
||||||
|
|
||||||
# .fcpxmld bundles are directories (a package wrapping Info.fcpxml plus
|
|
||||||
# sidecar data files for object tracking / Cinematic mode). The size
|
|
||||||
# check applies to the inner Info.fcpxml, which is what gets parsed.
|
|
||||||
if resolved.is_dir():
|
|
||||||
if resolved.suffix.lower() != '.fcpxmld':
|
|
||||||
raise ValueError(f"Not a regular file: {filepath}")
|
|
||||||
inner = resolved / 'Info.fcpxml'
|
|
||||||
if not inner.is_file():
|
|
||||||
raise ValueError(f"Invalid bundle (no Info.fcpxml): {filepath}")
|
|
||||||
size_target = inner
|
|
||||||
elif not resolved.is_file():
|
|
||||||
raise ValueError(f"Not a regular file: {filepath}")
|
|
||||||
else:
|
|
||||||
size_target = resolved
|
|
||||||
|
|
||||||
if allowed_extensions and resolved.suffix.lower() not in allowed_extensions:
|
|
||||||
raise ValueError(
|
|
||||||
f"Invalid file type '{resolved.suffix}'. "
|
|
||||||
f"Allowed: {', '.join(allowed_extensions)}"
|
|
||||||
)
|
|
||||||
|
|
||||||
if size_target.stat().st_size > max_size:
|
|
||||||
size_mb = size_target.stat().st_size / (1024 * 1024)
|
|
||||||
raise ValueError(f"File too large ({size_mb:.1f} MB). Maximum: {max_size // (1024 * 1024)} MB")
|
|
||||||
|
|
||||||
return str(resolved)
|
|
||||||
|
|
||||||
def _validate_output_path(output_path: str, *, anchor_dir: str | None = None) -> str:
|
|
||||||
"""Validate an output path with optional sandbox enforcement.
|
|
||||||
|
|
||||||
Resolves traversal, blocks null bytes, ensures parent exists, and — when
|
|
||||||
*anchor_dir* is provided — verifies the resolved output lives under that
|
|
||||||
directory. This prevents LLM-generated tool calls from writing to
|
|
||||||
arbitrary filesystem locations (e.g. ``/etc/cron.d/backdoor``).
|
|
||||||
|
|
||||||
Args:
|
|
||||||
output_path: The raw output path to validate.
|
|
||||||
anchor_dir: If set, the resolved output must be a child of this
|
|
||||||
directory. Typically the parent directory of the input file so
|
|
||||||
outputs stay co-located with their sources.
|
|
||||||
|
|
||||||
Raises:
|
|
||||||
ValueError: For null bytes, missing parent, or sandbox escape.
|
|
||||||
"""
|
|
||||||
if '\x00' in output_path:
|
|
||||||
raise ValueError("Invalid output path: null byte detected")
|
|
||||||
|
|
||||||
resolved = Path(output_path).resolve()
|
|
||||||
|
|
||||||
if not resolved.parent.exists():
|
|
||||||
raise ValueError(f"Output directory does not exist: {resolved.parent}")
|
|
||||||
|
|
||||||
if anchor_dir is not None:
|
|
||||||
anchor = Path(anchor_dir).resolve()
|
|
||||||
try:
|
|
||||||
resolved.relative_to(anchor)
|
|
||||||
except ValueError:
|
|
||||||
raise ValueError(
|
|
||||||
f"Output path escapes allowed directory: "
|
|
||||||
f"{resolved} is not under {anchor}"
|
|
||||||
)
|
|
||||||
|
|
||||||
return str(resolved)
|
|
||||||
|
|
||||||
def _validate_directory(directory: str, *, allowed_root: str | None = None) -> str:
|
|
||||||
"""Validate a user-provided directory path against traversal and injection.
|
|
||||||
|
|
||||||
Resolves symlinks, blocks null bytes, and verifies the path is a real
|
|
||||||
directory. When *allowed_root* is given, the resolved path must be a
|
|
||||||
descendant of (or equal to) that root — preventing filesystem enumeration
|
|
||||||
beyond the project workspace.
|
|
||||||
|
|
||||||
Raises:
|
|
||||||
ValueError: For invalid paths (null bytes, not a directory, sandbox escape).
|
|
||||||
"""
|
|
||||||
if '\x00' in directory:
|
|
||||||
raise ValueError("Invalid directory path: null byte detected")
|
|
||||||
|
|
||||||
resolved = Path(directory).resolve()
|
|
||||||
|
|
||||||
if not resolved.is_dir():
|
|
||||||
raise ValueError(f"Not a valid directory: {directory}")
|
|
||||||
|
|
||||||
if allowed_root is not None:
|
|
||||||
root = Path(allowed_root).resolve()
|
|
||||||
try:
|
|
||||||
resolved.relative_to(root)
|
|
||||||
except ValueError:
|
|
||||||
raise ValueError(
|
|
||||||
f"Directory escapes allowed root: "
|
|
||||||
f"{resolved} is not under {root}"
|
|
||||||
)
|
|
||||||
|
|
||||||
return str(resolved)
|
|
||||||
|
|
||||||
def find_fcpxml_files(directory: str) -> list[str]:
|
|
||||||
"""Find all FCPXML files in a directory."""
|
|
||||||
path = Path(directory)
|
|
||||||
files = list(str(f) for f in path.rglob("*.fcpxml"))
|
|
||||||
files.extend(str(f) for f in path.rglob("*.fcpxmld"))
|
|
||||||
return sorted(files)
|
|
||||||
|
|
||||||
def format_timecode(tc) -> str:
|
|
||||||
"""Format a Timecode object to SMPTE string."""
|
|
||||||
return tc.to_smpte() if tc else "00:00:00:00"
|
|
||||||
|
|
||||||
def format_duration(seconds: float) -> str:
|
|
||||||
"""Format seconds into human-readable duration."""
|
|
||||||
if seconds < 1:
|
|
||||||
return f"{seconds*1000:.0f}ms"
|
|
||||||
elif seconds < 60:
|
|
||||||
return f"{seconds:.2f}s"
|
|
||||||
return f"{int(seconds // 60)}m {seconds % 60:.1f}s"
|
|
||||||
|
|
||||||
def _format_clip_table(clips: list, header: str) -> str:
|
|
||||||
"""Render a list of clips as a markdown table with timecodes and durations.
|
|
||||||
|
|
||||||
Shared by handlers that filter clips by duration threshold
|
|
||||||
(find_short_cuts, find_long_clips).
|
|
||||||
"""
|
|
||||||
result = f"{header}\n\n| Name | TC | Duration |\n|------|----|---------|\n"
|
|
||||||
result += "\n".join(
|
|
||||||
f"| {c.name} | {format_timecode(c.start)} | {format_duration(c.duration_seconds)} |"
|
|
||||||
for c in clips
|
|
||||||
)
|
|
||||||
return result
|
|
||||||
|
|
||||||
def _markdown_table(headers: list[str], rows: list[list[str]]) -> str:
|
|
||||||
"""Build a markdown table from headers and rows.
|
|
||||||
|
|
||||||
Returns header row, separator row, and data rows as a single string.
|
|
||||||
Callers avoid repeating the ``| H1 | H2 |\\n|---|---|`` boilerplate
|
|
||||||
that appears in 15+ handlers.
|
|
||||||
"""
|
|
||||||
header_line = "| " + " | ".join(headers) + " |"
|
|
||||||
sep_line = "|" + "|".join("------" for _ in headers) + "|"
|
|
||||||
data_lines = "\n".join(
|
|
||||||
"| " + " | ".join(str(c) for c in row) + " |" for row in rows
|
|
||||||
)
|
|
||||||
return f"{header_line}\n{sep_line}\n{data_lines}"
|
|
||||||
|
|
||||||
def _format_batch_result(
|
|
||||||
title: str,
|
|
||||||
summary: dict[str, str],
|
|
||||||
headers: list[str],
|
|
||||||
rows: list[list[str]],
|
|
||||||
output_path: str,
|
|
||||||
) -> str:
|
|
||||||
"""Build a standard batch-operation result with summary, table, and save footer.
|
|
||||||
|
|
||||||
Used by batch fix handlers (flash frames, rapid trim, fill gaps) that all
|
|
||||||
share the same markdown structure: ``# Title → ## Summary → ## Details table
|
|
||||||
→ Saved to`` footer.
|
|
||||||
"""
|
|
||||||
summary_lines = "\n".join(f"- **{k}**: {v}" for k, v in summary.items())
|
|
||||||
table = _markdown_table(headers, rows)
|
|
||||||
return (
|
|
||||||
f"# {title}\n\n"
|
|
||||||
f"## Summary\n{summary_lines}\n\n"
|
|
||||||
f"## Details\n{table}\n\n"
|
|
||||||
f"Saved to: `{output_path}`"
|
|
||||||
)
|
|
||||||
|
|
||||||
def _fmt_suggestions(suggestions: list[str]) -> str:
|
|
||||||
"""Format pacing suggestions as markdown list (Python 3.10 compatible)."""
|
|
||||||
if not suggestions:
|
|
||||||
return "- Pacing looks good!"
|
|
||||||
nl = "\n"
|
|
||||||
return nl.join(f"- {s}" for s in suggestions)
|
|
||||||
|
|
||||||
def generate_output_path(input_path: str, suffix: str = "_modified") -> str:
|
|
||||||
"""Generate output path from input path.
|
|
||||||
|
|
||||||
The suffix is sanitized to prevent path-component injection — only
|
|
||||||
alphanumeric, hyphen, underscore, and dot characters survive.
|
|
||||||
"""
|
|
||||||
# Strip anything that could inject path separators or traversal sequences
|
|
||||||
clean_suffix = re.sub(r'[^a-zA-Z0-9._-]', '', suffix)
|
|
||||||
if not clean_suffix:
|
|
||||||
clean_suffix = "_modified"
|
|
||||||
p = Path(input_path)
|
|
||||||
return str(p.parent / f"{p.stem}{clean_suffix}{p.suffix}")
|
|
||||||
|
|
||||||
def _parse_project(filepath: str):
|
|
||||||
"""Parse an FCPXML file and return the project with its primary timeline."""
|
|
||||||
filepath = _validate_filepath(filepath, ('.fcpxml', '.fcpxmld'))
|
|
||||||
project = FCPXMLParser().parse_file(filepath)
|
|
||||||
if not project.timelines:
|
|
||||||
return None, None
|
|
||||||
return project, project.primary_timeline
|
|
||||||
|
|
||||||
def _text_result(text: str) -> list[TextContent]:
|
|
||||||
"""Wrap a string in the MCP TextContent list that every tool handler returns."""
|
|
||||||
return [TextContent(type="text", text=text)]
|
|
||||||
|
|
||||||
def _no_timeline():
|
|
||||||
"""Standard response when no timelines are found."""
|
|
||||||
return _text_result("No timelines found")
|
|
||||||
|
|
||||||
def _require_timeline(filepath: str):
|
|
||||||
"""Parse FCPXML and return (project, timeline), raising if no timeline exists.
|
|
||||||
|
|
||||||
Centralises the repeated _parse_project + _no_timeline guard that
|
|
||||||
appears in every read-only timeline handler. Returns a tuple so
|
|
||||||
callers can destructure directly::
|
|
||||||
|
|
||||||
project, tl = _require_timeline(arguments["filepath"])
|
|
||||||
"""
|
|
||||||
project, tl = _parse_project(filepath)
|
|
||||||
if not tl:
|
|
||||||
raise _NoTimelineError()
|
|
||||||
return project, tl
|
|
||||||
|
|
||||||
class _NoTimelineError(Exception):
|
|
||||||
"""Sentinel raised by _require_timeline when no timelines exist."""
|
|
||||||
|
|
||||||
def _resolve_io_paths(
|
|
||||||
arguments: dict,
|
|
||||||
suffix: str = "_modified",
|
|
||||||
) -> tuple[str, str]:
|
|
||||||
"""Validate input filepath and resolve the output path.
|
|
||||||
|
|
||||||
Shared foundation for every handler that reads an FCPXML and writes
|
|
||||||
a derived file. Validates the input, falls back to a suffixed
|
|
||||||
output name when ``output_path`` is not supplied, and sandbox-checks
|
|
||||||
the result.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
arguments: Tool arguments dict (must contain ``filepath``; may
|
|
||||||
contain ``output_path``).
|
|
||||||
suffix: Default output filename suffix when ``output_path`` is
|
|
||||||
not provided (e.g. ``"_modified"``, ``"_beats"``).
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
``(filepath, output_path)`` tuple with both paths validated.
|
|
||||||
"""
|
|
||||||
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
|
|
||||||
# Anchor write operations to the input file's directory so LLM-generated
|
|
||||||
# tool calls cannot write to arbitrary filesystem locations (e.g.
|
|
||||||
# /etc/cron.d/backdoor). When the explicit sandbox is off, the anchor
|
|
||||||
# still prevents writes outside the source directory tree.
|
|
||||||
# `output_dir` is where the caller wants the file written, not merely a
|
|
||||||
# sandbox boundary: the app's "Pasta do projeto" promises that everything
|
|
||||||
# generated lands there. Deriving the name from the input but keeping the
|
|
||||||
# input's directory made every cross-directory call fail its own anchor
|
|
||||||
# check ("output path escapes allowed directory"), so the setting silently
|
|
||||||
# only worked when it pointed at the directory the file was already going
|
|
||||||
# to. An explicit `output_path` still wins, and still has to sit inside
|
|
||||||
# the anchor.
|
|
||||||
output_dir = arguments.get("output_dir")
|
|
||||||
if output_dir:
|
|
||||||
anchor = _validate_directory(str(output_dir))
|
|
||||||
default_output = str(Path(anchor) / Path(generate_output_path(filepath, suffix)).name)
|
|
||||||
else:
|
|
||||||
anchor = str(Path(filepath).resolve().parent)
|
|
||||||
default_output = generate_output_path(filepath, suffix)
|
|
||||||
output_path = _validate_output_path(
|
|
||||||
arguments.get("output_path") or default_output,
|
|
||||||
anchor_dir=anchor,
|
|
||||||
)
|
|
||||||
return filepath, output_path
|
|
||||||
|
|
||||||
def _setup_modifier(
|
|
||||||
arguments: dict,
|
|
||||||
suffix: str = "_modified",
|
|
||||||
) -> tuple[str, str, "FCPXMLModifier"]:
|
|
||||||
"""Common setup for write handlers: validate paths and create modifier.
|
|
||||||
|
|
||||||
Consolidates the repeated validate-filepath → resolve-output-path →
|
|
||||||
create-modifier boilerplate shared by 18+ write handlers.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
arguments: Tool arguments dict (must contain ``filepath``; may
|
|
||||||
contain ``output_path``).
|
|
||||||
suffix: Default output filename suffix when ``output_path`` is
|
|
||||||
not provided (e.g. ``"_modified"``, ``"_flash_fixed"``).
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
``(filepath, output_path, modifier)`` tuple ready for the
|
|
||||||
handler's domain-specific operation.
|
|
||||||
"""
|
|
||||||
filepath, output_path = _resolve_io_paths(arguments, suffix)
|
|
||||||
modifier = FCPXMLModifier(filepath)
|
|
||||||
return filepath, output_path, modifier
|
|
||||||
|
|
||||||
def _setup_generator(
|
|
||||||
arguments: dict,
|
|
||||||
suffix: str = "_roughcut",
|
|
||||||
) -> tuple[str, str, "RoughCutGenerator"]:
|
|
||||||
"""Common setup for generation handlers: validate paths and create generator.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
arguments: Tool arguments dict (must contain ``filepath`` and
|
|
||||||
``output_path``).
|
|
||||||
suffix: Default output filename suffix.
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
``(filepath, output_path, generator)`` tuple.
|
|
||||||
"""
|
|
||||||
filepath, output_path = _resolve_io_paths(arguments, suffix)
|
|
||||||
generator = RoughCutGenerator(filepath)
|
|
||||||
return filepath, output_path, generator
|
|
||||||
|
|
||||||
def _parse_timestamp_parts(
|
|
||||||
parts: list[str], *, frame_rate: float = 24.0
|
|
||||||
) -> float | None:
|
|
||||||
"""Convert colon-separated timestamp parts to total seconds.
|
|
||||||
|
|
||||||
Handles 2-part (M:SS), 3-part (H:MM:SS / HH:MM:SS.ms), and
|
|
||||||
4-part (HH:MM:SS:FF SMPTE) formats. Returns ``None`` when the
|
|
||||||
part count is unrecognised so callers can skip.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
parts: Colon-split timestamp components.
|
|
||||||
frame_rate: FPS used to convert the frame component of SMPTE
|
|
||||||
timecodes into fractional seconds (default 24.0).
|
|
||||||
"""
|
|
||||||
if len(parts) == 2:
|
|
||||||
return int(parts[0]) * 60 + float(parts[1])
|
|
||||||
elif len(parts) == 3:
|
|
||||||
return int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2])
|
|
||||||
elif len(parts) == 4:
|
|
||||||
# SMPTE: HH:MM:SS:FF — convert frames to fractional seconds
|
|
||||||
base = int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2])
|
|
||||||
frames = int(parts[3])
|
|
||||||
return base + (frames / frame_rate) if frame_rate > 0 else base
|
|
||||||
return None
|
|
||||||
|
|
||||||
def _raw_markers_to_batch(
|
|
||||||
raw_markers: list[dict],
|
|
||||||
marker_type: str = "chapter",
|
|
||||||
max_label: int | None = None,
|
|
||||||
) -> list[dict]:
|
|
||||||
"""Convert raw {seconds, text} marker dicts to batch_add_markers format.
|
|
||||||
|
|
||||||
Shared by import_srt_markers and import_transcript_markers.
|
|
||||||
"""
|
|
||||||
batch = []
|
|
||||||
for m in raw_markers:
|
|
||||||
label = m["text"]
|
|
||||||
if max_label and len(label) > max_label:
|
|
||||||
label = label[:max_label]
|
|
||||||
batch.append({
|
|
||||||
"timecode": f"{m['seconds']}s",
|
|
||||||
"name": label,
|
|
||||||
"marker_type": marker_type.upper(),
|
|
||||||
})
|
|
||||||
return batch
|
|
||||||
|
|
||||||
def _extract_subtitle_blocks(text: str, *, strip_vtt_tags: bool = False) -> list[dict]:
|
|
||||||
"""Extract timestamp/text pairs from subtitle cue blocks (SRT or VTT).
|
|
||||||
|
|
||||||
Both SRT and VTT use the same ``start --> end`` cue syntax with
|
|
||||||
text lines underneath; only header stripping and tag cleaning differ.
|
|
||||||
"""
|
|
||||||
markers = []
|
|
||||||
blocks = re.split(r'\n\s*\n', text.strip())
|
|
||||||
for block in blocks:
|
|
||||||
lines = block.strip().split('\n')
|
|
||||||
if len(lines) < 2:
|
|
||||||
continue
|
|
||||||
ts_line = None
|
|
||||||
text_lines = []
|
|
||||||
for line in lines:
|
|
||||||
if '-->' in line:
|
|
||||||
ts_line = line
|
|
||||||
elif ts_line is not None:
|
|
||||||
if strip_vtt_tags:
|
|
||||||
line = re.sub(r'<[^>]+>', '', line)
|
|
||||||
cleaned = line.strip()
|
|
||||||
if cleaned:
|
|
||||||
text_lines.append(cleaned)
|
|
||||||
if not ts_line or not text_lines:
|
|
||||||
continue
|
|
||||||
start_str = ts_line.split('-->')[0].strip().replace(',', '.')
|
|
||||||
seconds = _parse_timestamp_parts(start_str.split(':'))
|
|
||||||
if seconds is not None:
|
|
||||||
markers.append({'seconds': seconds, 'text': ' '.join(text_lines)})
|
|
||||||
return markers
|
|
||||||
|
|
||||||
def parse_srt(text: str) -> list[dict]:
|
|
||||||
"""Parse SRT subtitle format into timestamp/text pairs."""
|
|
||||||
return _extract_subtitle_blocks(text)
|
|
||||||
|
|
||||||
def parse_vtt(text: str) -> list[dict]:
|
|
||||||
"""Parse WebVTT subtitle format into timestamp/text pairs."""
|
|
||||||
text = re.sub(r'^WEBVTT.*?\n', '', text, flags=re.MULTILINE)
|
|
||||||
text = re.sub(r'NOTE\n.*?\n\n', '', text, flags=re.DOTALL)
|
|
||||||
return _extract_subtitle_blocks(text, strip_vtt_tags=True)
|
|
||||||
|
|
||||||
def parse_transcript_timestamps(text: str) -> list[dict]:
|
|
||||||
"""Parse timestamped text (YouTube description format) into markers.
|
|
||||||
|
|
||||||
Supports formats like:
|
|
||||||
0:00 Introduction
|
|
||||||
00:01:30 Main Topic
|
|
||||||
1:05:30 Conclusion
|
|
||||||
00:00:00:00 SMPTE timecode
|
|
||||||
"""
|
|
||||||
markers = []
|
|
||||||
for line in text.strip().split('\n'):
|
|
||||||
line = line.strip()
|
|
||||||
if not line:
|
|
||||||
continue
|
|
||||||
match = re.match(r'^(\d{1,2}:\d{2}(?::\d{2}){0,2})\s+(.+)$', line)
|
|
||||||
if match:
|
|
||||||
seconds = _parse_timestamp_parts(match.group(1).split(':'))
|
|
||||||
if seconds is not None:
|
|
||||||
markers.append({'seconds': seconds, 'text': match.group(2).strip()})
|
|
||||||
return markers
|
|
||||||
|
|
||||||
def _detect_flash_frames(
|
|
||||||
tl: Any, *, critical_threshold: int = 2, warning_threshold: int = 6,
|
|
||||||
) -> list:
|
|
||||||
"""Find clips shorter than *warning_threshold* frames.
|
|
||||||
|
|
||||||
Returns a list of ``FlashFrame`` objects sorted by severity. Shared by
|
|
||||||
``handle_detect_flash_frames`` and ``handle_validate_timeline`` so the
|
|
||||||
detection logic lives in exactly one place.
|
|
||||||
"""
|
|
||||||
fps = tl.frame_rate
|
|
||||||
flash_frames: list[FlashFrame] = []
|
|
||||||
for clip in tl.clips:
|
|
||||||
duration_frames = int(clip.duration_seconds * fps)
|
|
||||||
if duration_frames < warning_threshold:
|
|
||||||
severity = (
|
|
||||||
FlashFrameSeverity.CRITICAL
|
|
||||||
if duration_frames < critical_threshold
|
|
||||||
else FlashFrameSeverity.WARNING
|
|
||||||
)
|
|
||||||
flash_frames.append(FlashFrame(
|
|
||||||
clip_name=clip.name, clip_id=clip.name,
|
|
||||||
start=clip.start, duration_frames=duration_frames,
|
|
||||||
duration_seconds=clip.duration_seconds, severity=severity,
|
|
||||||
))
|
|
||||||
return flash_frames
|
|
||||||
|
|
||||||
def _detect_gaps(tl: Any, *, min_gap_frames: int = 1) -> list:
|
|
||||||
"""Find inter-clip gaps of at least *min_gap_frames* length.
|
|
||||||
|
|
||||||
Returns a list of ``GapInfo`` objects. Shared by ``handle_detect_gaps``
|
|
||||||
and ``handle_validate_timeline``.
|
|
||||||
"""
|
|
||||||
fps = tl.frame_rate
|
|
||||||
min_gap_seconds = min_gap_frames / fps
|
|
||||||
gaps: list[GapInfo] = []
|
|
||||||
sorted_clips = sorted(tl.clips, key=lambda c: c.start.seconds)
|
|
||||||
for i in range(len(sorted_clips) - 1):
|
|
||||||
current_end = sorted_clips[i].end.seconds
|
|
||||||
next_start = sorted_clips[i + 1].start.seconds
|
|
||||||
gap_duration = next_start - current_end
|
|
||||||
if gap_duration >= min_gap_seconds:
|
|
||||||
gaps.append(GapInfo(
|
|
||||||
start=Timecode(frames=int(current_end * fps), frame_rate=fps),
|
|
||||||
duration_frames=int(gap_duration * fps),
|
|
||||||
duration_seconds=gap_duration,
|
|
||||||
previous_clip=sorted_clips[i].name,
|
|
||||||
next_clip=sorted_clips[i + 1].name,
|
|
||||||
))
|
|
||||||
return gaps
|
|
||||||
|
|
||||||
def _detect_duplicate_groups(tl: Any, *, mode: str = "same_source") -> list:
|
|
||||||
"""Group clips that share a source media reference.
|
|
||||||
|
|
||||||
Returns a list of ``DuplicateGroup`` objects. Shared by
|
|
||||||
``handle_detect_duplicates`` and ``handle_validate_timeline``.
|
|
||||||
"""
|
|
||||||
source_groups: dict[str, list[dict]] = {}
|
|
||||||
for clip in tl.clips:
|
|
||||||
source_key = clip.media_path or clip.name
|
|
||||||
if source_key not in source_groups:
|
|
||||||
source_groups[source_key] = []
|
|
||||||
source_groups[source_key].append({
|
|
||||||
'name': clip.name,
|
|
||||||
'start': clip.start.seconds,
|
|
||||||
'duration': clip.duration_seconds,
|
|
||||||
'source_start': clip.source_start.seconds if clip.source_start else 0,
|
|
||||||
'source_duration': clip.duration_seconds,
|
|
||||||
'timecode': format_timecode(clip.start),
|
|
||||||
})
|
|
||||||
|
|
||||||
duplicates: list[DuplicateGroup] = []
|
|
||||||
for source_key, clips in source_groups.items():
|
|
||||||
if len(clips) <= 1:
|
|
||||||
continue
|
|
||||||
group = DuplicateGroup(
|
|
||||||
source_ref=source_key,
|
|
||||||
source_name=source_key.split('/')[-1] if '/' in source_key else source_key,
|
|
||||||
clips=clips,
|
|
||||||
)
|
|
||||||
if mode == "same_source":
|
|
||||||
duplicates.append(group)
|
|
||||||
elif mode == "overlapping_ranges" and group.has_overlapping_ranges:
|
|
||||||
duplicates.append(group)
|
|
||||||
elif mode == "identical":
|
|
||||||
seen_ranges: set[tuple] = set()
|
|
||||||
identical_clips = []
|
|
||||||
for c in clips:
|
|
||||||
range_key = (c['source_start'], c['source_duration'])
|
|
||||||
if range_key in seen_ranges:
|
|
||||||
identical_clips.append(c)
|
|
||||||
seen_ranges.add(range_key)
|
|
||||||
if identical_clips:
|
|
||||||
group.clips = identical_clips
|
|
||||||
duplicates.append(group)
|
|
||||||
return duplicates
|
|
||||||
|
|
||||||
AUDIO_MEDIA_EXTENSIONS = (
|
|
||||||
'.wav', '.aif', '.aiff', '.mp3', '.m4a', '.aac', '.flac', '.mov', '.mp4',
|
|
||||||
)
|
|
||||||
|
|
||||||
_DIARIZATION_INSTALL_HINT = (
|
|
||||||
"\n\nInstall the optional diarization extra:\n\n"
|
|
||||||
" pip install 'fcp-mcp-server[diarization]'\n\n"
|
|
||||||
"and set a HuggingFace token with access to "
|
|
||||||
"pyannote/speaker-diarization-3.1 (pass hf_token= or persist one via "
|
|
||||||
"save_hf_token)."
|
|
||||||
)
|
|
||||||
|
|
||||||
_FEATURES_INSTALL_HINT = (
|
|
||||||
"\n\nInstall the optional media-intelligence extra:\n\n"
|
|
||||||
" pip install 'fcp-mcp-server[intelligence]'"
|
|
||||||
)
|
|
||||||
|
|
||||||
def _voice_analysis_config_text(config: dict) -> str:
|
|
||||||
w = config["emphasis_weights"]
|
|
||||||
text = "# Voice Analysis Settings\n\n"
|
|
||||||
text += _markdown_table(
|
|
||||||
["Setting", "Value"],
|
|
||||||
[
|
|
||||||
["Energy threshold", f"{config['energy_threshold']:.2f}"],
|
|
||||||
["Peak selection", f"top {config['peak_percentile']:.1%} of words"],
|
|
||||||
["Emphasis floor", f"{config['emphasis_floor']:.2f}"],
|
|
||||||
["Emotion detection", "on" if config["emotion_enabled"] else "off"],
|
|
||||||
["Emotion sensitivity", f"{config['emotion_sensitivity']:.2f}"],
|
|
||||||
],
|
|
||||||
) + "\n\n## Emphasis Weights\n"
|
|
||||||
text += _markdown_table(
|
|
||||||
["Factor", "Weight"],
|
|
||||||
[[k.replace("_", " ").title(), f"{v:.2f}"] for k, v in w.items()],
|
|
||||||
)
|
|
||||||
return text
|
|
||||||
|
|
||||||
def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str:
|
|
||||||
"""Apply one non-cut action to the clip that hosts it.
|
|
||||||
|
|
||||||
``clip_start`` is where that clip begins on the timeline; the writer
|
|
||||||
wants times relative to the clip's own head, so the rebase happens here
|
|
||||||
— the single place that knows about the conversion. The clip *element*
|
|
||||||
is passed through rather than its name: after a cut the pieces share a
|
|
||||||
name, and a name lookup would land every edit on the first piece.
|
|
||||||
"""
|
|
||||||
rel_start = action.start - clip_start
|
|
||||||
rel_end = action.end - clip_start
|
|
||||||
|
|
||||||
if action.kind == "zoom":
|
|
||||||
config = load_voice_analysis_config()
|
|
||||||
# Only forward an explicit ease — otherwise add_zoom's own default
|
|
||||||
# (a fast ramp in, instant snap back out) is what should apply.
|
|
||||||
zoom_args = {
|
|
||||||
"ease": float(action.params.get("ease", config["zoom_ease_in"])),
|
|
||||||
"ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])),
|
|
||||||
}
|
|
||||||
mode = str(action.params.get("mode", config["zoom_mode"]))
|
|
||||||
if mode == "in":
|
|
||||||
zoom_args["hold_at_end"] = True
|
|
||||||
zoom_args["start_at_peak"] = False
|
|
||||||
elif mode == "out":
|
|
||||||
zoom_args["hold_at_end"] = False
|
|
||||||
zoom_args["start_at_peak"] = True
|
|
||||||
elif mode == "in_out":
|
|
||||||
zoom_args["hold_at_end"] = False
|
|
||||||
zoom_args["start_at_peak"] = False
|
|
||||||
modifier.add_zoom(
|
|
||||||
clip_id=clip_el,
|
|
||||||
start=rel_start,
|
|
||||||
end=rel_end,
|
|
||||||
scale=float(action.params.get("scale", config["zoom_scale"])),
|
|
||||||
**zoom_args,
|
|
||||||
)
|
|
||||||
return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x"
|
|
||||||
|
|
||||||
if action.kind == "text":
|
|
||||||
# Default to the "Legendas Dinâmicas" emphasis style (the font used
|
|
||||||
# to highlight a word in the captions) rather than a hardcoded
|
|
||||||
# Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of
|
|
||||||
# the video's on-screen text instead of looking like a stray default
|
|
||||||
# title. Any of these the action itself specifies still wins.
|
|
||||||
subtitle_cfg = load_dynamic_subtitle_config()
|
|
||||||
font = action.params.get("font", subtitle_cfg["emphasis_font"])
|
|
||||||
face = action.params.get("face", subtitle_cfg["emphasis_face"])
|
|
||||||
font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0)
|
|
||||||
requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"]))
|
|
||||||
requested_kerning = float(action.params.get("kerning", 0.0) or 0.0)
|
|
||||||
|
|
||||||
# Voice-action callouts are not part of the dynamic subtitle block.
|
|
||||||
# When omitted, put them above the subtitle band and shrink wide
|
|
||||||
# phrases to the title-safe width. The previous default (Position 0 0,
|
|
||||||
# full emphasis size) made long callouts like "PRÓTESES DE SILICONE"
|
|
||||||
# collide with captions and run off both sides of a vertical frame.
|
|
||||||
emitted_size = requested_size * font_scale
|
|
||||||
emitted_kerning = requested_kerning * font_scale
|
|
||||||
safe_width = modifier.frame_width() * 0.90
|
|
||||||
width = measure_text(
|
|
||||||
action.params["content"],
|
|
||||||
emitted_size,
|
|
||||||
bold=bool(action.params.get("bold", False)),
|
|
||||||
kerning=emitted_kerning,
|
|
||||||
font=font,
|
|
||||||
face=face,
|
|
||||||
)
|
|
||||||
font_size = requested_size
|
|
||||||
if width > safe_width and width > 0:
|
|
||||||
font_size = max(32, int(requested_size * safe_width / width))
|
|
||||||
position = action.params.get("position")
|
|
||||||
if not position:
|
|
||||||
position = f"0 {modifier.frame_height() * 0.23:g}"
|
|
||||||
|
|
||||||
modifier.add_text_title(
|
|
||||||
clip_el,
|
|
||||||
action.params["content"],
|
|
||||||
offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
|
|
||||||
duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(),
|
|
||||||
position=position,
|
|
||||||
font=font,
|
|
||||||
font_size=font_size,
|
|
||||||
font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]),
|
|
||||||
face=face,
|
|
||||||
bold=action.params.get("bold", False),
|
|
||||||
)
|
|
||||||
return f"text \"{action.params['content'][:24]}\""
|
|
||||||
|
|
||||||
# marker
|
|
||||||
modifier.add_marker(
|
|
||||||
clip_id=clip_el,
|
|
||||||
timecode=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
|
|
||||||
name=action.params.get("content") or action.reason or "Voice action",
|
|
||||||
note=action.reason or None,
|
|
||||||
)
|
|
||||||
return "marker"
|
|
||||||
|
|
||||||
def _speaker_table(profiles: Sequence[dict]) -> str:
|
|
||||||
"""Who was detected, ordered by how much of the runtime each holds."""
|
|
||||||
return _markdown_table(
|
|
||||||
["ID", "Name", "Share", "Speaking", "Lines", "Avg line"],
|
|
||||||
[
|
|
||||||
[
|
|
||||||
p["id"],
|
|
||||||
p.get("name", ""),
|
|
||||||
f"{p['share']:.0%}",
|
|
||||||
format_duration(p["speaking_seconds"]),
|
|
||||||
str(p["segment_count"]),
|
|
||||||
f"{p['avg_segment']:.1f}s",
|
|
||||||
]
|
|
||||||
for p in profiles
|
|
||||||
],
|
|
||||||
)
|
|
||||||
|
|
||||||
TRANSCRIBE_MAX_MEDIA = 10
|
|
||||||
|
|
||||||
_TRANSCRIBE_INSTALL_HINT = (
|
|
||||||
"\n\nInstall the optional transcription extra:\n\n"
|
|
||||||
" pip install 'fcp-mcp-server[transcribe]'\n\n"
|
|
||||||
"or run via uvx:\n\n"
|
|
||||||
" uvx --from \"fcp-mcp-server[transcribe]\" fcp-mcp-server"
|
|
||||||
)
|
|
||||||
|
|
||||||
def _transcript_json_path(media_path: str, output_dir: str | None = None) -> Path:
|
|
||||||
"""Where the ``_transcript.json`` for ``media_path`` lives.
|
|
||||||
|
|
||||||
When ``output_dir`` (the user-selected project folder) is set, the
|
|
||||||
transcript is saved/read there instead of next to the source media.
|
|
||||||
"""
|
|
||||||
p = Path(media_path)
|
|
||||||
if output_dir:
|
|
||||||
directory = Path(output_dir).expanduser()
|
|
||||||
directory.mkdir(parents=True, exist_ok=True)
|
|
||||||
return directory / f"{p.stem}_transcript.json"
|
|
||||||
return p.with_name(p.stem + "_transcript.json")
|
|
||||||
|
|
||||||
def _load_or_transcribe(
|
|
||||||
media_path: str, model: str, language: str | None, output_dir: str | None = None
|
|
||||||
) -> tuple[dict | None, str]:
|
|
||||||
"""Load a cached ``_transcript.json`` for a media file, else transcribe and cache it.
|
|
||||||
|
|
||||||
Returns ``(transcript, "")`` or ``(None, reason)``. The cache makes
|
|
||||||
transcription a one-time cost per media file across all transcript tools.
|
|
||||||
"""
|
|
||||||
json_path = _transcript_json_path(media_path, output_dir)
|
|
||||||
if json_path.is_file():
|
|
||||||
try:
|
|
||||||
with open(json_path) as f:
|
|
||||||
data = json.load(f)
|
|
||||||
if isinstance(data, dict) and isinstance(data.get("words"), list):
|
|
||||||
return data, ""
|
|
||||||
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
|
||||||
pass # unreadable cache falls through to re-transcribe
|
|
||||||
result = transcribe(media_path, model_size=model, language=language)
|
|
||||||
if result is None:
|
|
||||||
return None, "untranscribable (faster-whisper not installed or media unreadable)"
|
|
||||||
anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent)
|
|
||||||
out_path = _validate_output_path(str(json_path), anchor_dir=anchor)
|
|
||||||
with open(out_path, "w") as f:
|
|
||||||
json.dump({"source": Path(media_path).name, **result}, f, indent=2)
|
|
||||||
return result, ""
|
|
||||||
|
|
||||||
def _cut_transcript_spans(modifier, clip_filter, model, language, padding, spans_fn, keep_only=False, output_dir=None):
|
|
||||||
"""Shared cut engine for transcript-driven editing.
|
|
||||||
|
|
||||||
``spans_fn(words) -> [(start, end), ...]`` in source seconds. Spans are
|
|
||||||
padded, clamped to each clip's used source window, optionally inverted
|
|
||||||
(keep_only), snapped to the frame grid, and cut with ripple.
|
|
||||||
"""
|
|
||||||
to_frame = modifier.snap_seconds_to_frame
|
|
||||||
|
|
||||||
cache: dict[str, tuple] = {}
|
|
||||||
cuts_made: list[tuple[str, int, float]] = []
|
|
||||||
skipped: list[tuple[str, str]] = []
|
|
||||||
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
|
||||||
for el in spine_clips:
|
|
||||||
name = el.get("name", "")
|
|
||||||
if clip_filter and name != clip_filter:
|
|
||||||
continue
|
|
||||||
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
|
||||||
media_path = media_src_to_path(src)
|
|
||||||
if not media_path or not Path(media_path).is_file():
|
|
||||||
skipped.append((name, "media file missing"))
|
|
||||||
continue
|
|
||||||
if media_path not in cache:
|
|
||||||
if len(cache) >= TRANSCRIBE_MAX_MEDIA:
|
|
||||||
skipped.append((name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)"))
|
|
||||||
continue
|
|
||||||
cache[media_path] = _load_or_transcribe(media_path, model, language, output_dir)
|
|
||||||
data, reason = cache[media_path]
|
|
||||||
if data is None:
|
|
||||||
skipped.append((name, reason))
|
|
||||||
continue
|
|
||||||
|
|
||||||
clip_source_start = modifier.source_file_start(el).to_seconds()
|
|
||||||
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
|
||||||
window_start = clip_source_start
|
|
||||||
window_end = clip_source_start + clip_duration
|
|
||||||
|
|
||||||
spans = spans_fn(data.get("words", []))
|
|
||||||
padded = merge_ranges([(s - padding, e + padding) for s, e in spans])
|
|
||||||
clamped = [
|
|
||||||
(max(s, window_start), min(e, window_end))
|
|
||||||
for s, e in padded
|
|
||||||
if min(e, window_end) > max(s, window_start)
|
|
||||||
]
|
|
||||||
if keep_only:
|
|
||||||
if not clamped:
|
|
||||||
# Never delete a whole clip just because nothing matched in it.
|
|
||||||
skipped.append((name, "no phrase matches — left untouched (keep_only)"))
|
|
||||||
continue
|
|
||||||
cut_source = invert_ranges(clamped, window_start, window_end)
|
|
||||||
else:
|
|
||||||
cut_source = clamped
|
|
||||||
cut_ranges = [
|
|
||||||
(to_frame(s - clip_source_start), to_frame(e - clip_source_start))
|
|
||||||
for s, e in cut_source
|
|
||||||
]
|
|
||||||
cut_ranges = [(a, b) for a, b in cut_ranges if b > a]
|
|
||||||
if not cut_ranges:
|
|
||||||
continue
|
|
||||||
removed = modifier.cut_clip_ranges(el, cut_ranges)
|
|
||||||
if removed > TimeValue.zero():
|
|
||||||
cuts_made.append((name, len(cut_ranges), removed.to_seconds()))
|
|
||||||
return cuts_made, skipped
|
|
||||||
|
|
||||||
def _transcript_cut_report(title, summary_lines, cuts_made, skipped, output_path, footer):
|
|
||||||
if not cuts_made:
|
|
||||||
text = f"# {title}\n\nNo cuts to make — file unchanged (nothing saved)."
|
|
||||||
if skipped:
|
|
||||||
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
|
||||||
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
|
|
||||||
)
|
|
||||||
if any("faster-whisper" in reason for _, reason in skipped):
|
|
||||||
text += _TRANSCRIBE_INSTALL_HINT
|
|
||||||
return _text_result(text)
|
|
||||||
total_removed = sum(seconds for _, _, seconds in cuts_made)
|
|
||||||
result = f"# {title}\n\n## Summary\n"
|
|
||||||
result += "\n".join(summary_lines) + "\n"
|
|
||||||
result += f"- **Clips Cut**: {len(cuts_made)}\n- **Total Removed**: {format_duration(total_removed)}\n"
|
|
||||||
result += "\n## Cuts\n"
|
|
||||||
result += _markdown_table(
|
|
||||||
["Clip", "Ranges Cut", "Removed"],
|
|
||||||
[[name, str(count), f"{seconds:.2f}s"] for name, count, seconds in cuts_made],
|
|
||||||
) + "\n"
|
|
||||||
if skipped:
|
|
||||||
result += "\n## Skipped Clips\n" + _markdown_table(
|
|
||||||
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
|
|
||||||
) + "\n"
|
|
||||||
result += f"\nSaved to: {output_path}\n\n{footer}"
|
|
||||||
return _text_result(result)
|
|
||||||
@@ -0,0 +1,124 @@
|
|||||||
|
"""Shared internal helpers used by tool handlers across categories.
|
||||||
|
|
||||||
|
Extracted from server.py — validation, formatting, and small parsing utilities
|
||||||
|
that more than one server_tools/*.py module needs.
|
||||||
|
|
||||||
|
Eram 882 linhas de seis papéis diferentes sob um nome que só dizia
|
||||||
|
"compartilhado". Cada papel virou um módulo; este pacote reexporta tudo, então
|
||||||
|
os treze pontos que importam daqui seguem iguais.
|
||||||
|
|
||||||
|
paths validação contra a sandbox, limites, caminho de saída
|
||||||
|
formatting tabelas e relatórios devolvidos pelos handlers
|
||||||
|
project abrir projeto, preparar modifier/generator
|
||||||
|
captions SRT, VTT e listas com timestamp
|
||||||
|
detection flash frames, buracos, duplicados
|
||||||
|
media transcrição em cache, corte por fala, ações posicionadas
|
||||||
|
"""
|
||||||
|
|
||||||
|
from .captions import (
|
||||||
|
_extract_subtitle_blocks,
|
||||||
|
_parse_timestamp_parts,
|
||||||
|
_raw_markers_to_batch,
|
||||||
|
parse_srt,
|
||||||
|
parse_transcript_timestamps,
|
||||||
|
parse_vtt,
|
||||||
|
)
|
||||||
|
from .detection import (
|
||||||
|
_detect_duplicate_groups,
|
||||||
|
_detect_flash_frames,
|
||||||
|
_detect_gaps,
|
||||||
|
)
|
||||||
|
from .formatting import (
|
||||||
|
_fmt_suggestions,
|
||||||
|
_format_batch_result,
|
||||||
|
_format_clip_table,
|
||||||
|
_markdown_table,
|
||||||
|
_speaker_table,
|
||||||
|
_voice_analysis_config_text,
|
||||||
|
format_duration,
|
||||||
|
format_timecode,
|
||||||
|
)
|
||||||
|
from .media import (
|
||||||
|
_DIARIZATION_INSTALL_HINT,
|
||||||
|
_FEATURES_INSTALL_HINT,
|
||||||
|
_TRANSCRIBE_INSTALL_HINT,
|
||||||
|
AUDIO_MEDIA_EXTENSIONS,
|
||||||
|
TRANSCRIBE_MAX_MEDIA,
|
||||||
|
_apply_placed_action,
|
||||||
|
_cut_transcript_spans,
|
||||||
|
_load_or_transcribe,
|
||||||
|
_transcript_cut_report,
|
||||||
|
_transcript_json_path,
|
||||||
|
)
|
||||||
|
from .paths import (
|
||||||
|
_MAX_JSON_DEPTH,
|
||||||
|
_SANDBOX_ENABLED,
|
||||||
|
MAX_FILE_SIZE,
|
||||||
|
MAX_MEDIA_FILE_SIZE,
|
||||||
|
PROJECTS_DIR,
|
||||||
|
_check_json_depth,
|
||||||
|
_resolve_io_paths,
|
||||||
|
_validate_directory,
|
||||||
|
_validate_filepath,
|
||||||
|
_validate_output_path,
|
||||||
|
find_fcpxml_files,
|
||||||
|
generate_output_path,
|
||||||
|
)
|
||||||
|
from .project import (
|
||||||
|
_no_timeline,
|
||||||
|
_NoTimelineError,
|
||||||
|
_parse_project,
|
||||||
|
_require_timeline,
|
||||||
|
_setup_generator,
|
||||||
|
_setup_modifier,
|
||||||
|
_text_result,
|
||||||
|
)
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"AUDIO_MEDIA_EXTENSIONS",
|
||||||
|
"MAX_FILE_SIZE",
|
||||||
|
"MAX_MEDIA_FILE_SIZE",
|
||||||
|
"PROJECTS_DIR",
|
||||||
|
"TRANSCRIBE_MAX_MEDIA",
|
||||||
|
"_DIARIZATION_INSTALL_HINT",
|
||||||
|
"_FEATURES_INSTALL_HINT",
|
||||||
|
"_MAX_JSON_DEPTH",
|
||||||
|
"_NoTimelineError",
|
||||||
|
"_SANDBOX_ENABLED",
|
||||||
|
"_TRANSCRIBE_INSTALL_HINT",
|
||||||
|
"_apply_placed_action",
|
||||||
|
"_check_json_depth",
|
||||||
|
"_cut_transcript_spans",
|
||||||
|
"_detect_duplicate_groups",
|
||||||
|
"_detect_flash_frames",
|
||||||
|
"_detect_gaps",
|
||||||
|
"_extract_subtitle_blocks",
|
||||||
|
"_fmt_suggestions",
|
||||||
|
"_format_batch_result",
|
||||||
|
"_format_clip_table",
|
||||||
|
"_load_or_transcribe",
|
||||||
|
"_markdown_table",
|
||||||
|
"_no_timeline",
|
||||||
|
"_parse_project",
|
||||||
|
"_parse_timestamp_parts",
|
||||||
|
"_raw_markers_to_batch",
|
||||||
|
"_require_timeline",
|
||||||
|
"_resolve_io_paths",
|
||||||
|
"_setup_generator",
|
||||||
|
"_setup_modifier",
|
||||||
|
"_speaker_table",
|
||||||
|
"_text_result",
|
||||||
|
"_transcript_cut_report",
|
||||||
|
"_transcript_json_path",
|
||||||
|
"_validate_directory",
|
||||||
|
"_validate_filepath",
|
||||||
|
"_validate_output_path",
|
||||||
|
"_voice_analysis_config_text",
|
||||||
|
"find_fcpxml_files",
|
||||||
|
"format_duration",
|
||||||
|
"format_timecode",
|
||||||
|
"generate_output_path",
|
||||||
|
"parse_srt",
|
||||||
|
"parse_transcript_timestamps",
|
||||||
|
"parse_vtt",
|
||||||
|
]
|
||||||
@@ -0,0 +1,117 @@
|
|||||||
|
"""Leitura de legendas e listas com timestamp (SRT, VTT, texto colado).
|
||||||
|
|
||||||
|
Extraído de _shared.py — ver server_tools/_shared/__init__.py.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_timestamp_parts(
|
||||||
|
parts: list[str], *, frame_rate: float = 24.0
|
||||||
|
) -> float | None:
|
||||||
|
"""Convert colon-separated timestamp parts to total seconds.
|
||||||
|
|
||||||
|
Handles 2-part (M:SS), 3-part (H:MM:SS / HH:MM:SS.ms), and
|
||||||
|
4-part (HH:MM:SS:FF SMPTE) formats. Returns ``None`` when the
|
||||||
|
part count is unrecognised so callers can skip.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
parts: Colon-split timestamp components.
|
||||||
|
frame_rate: FPS used to convert the frame component of SMPTE
|
||||||
|
timecodes into fractional seconds (default 24.0).
|
||||||
|
"""
|
||||||
|
if len(parts) == 2:
|
||||||
|
return int(parts[0]) * 60 + float(parts[1])
|
||||||
|
elif len(parts) == 3:
|
||||||
|
return int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2])
|
||||||
|
elif len(parts) == 4:
|
||||||
|
# SMPTE: HH:MM:SS:FF — convert frames to fractional seconds
|
||||||
|
base = int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2])
|
||||||
|
frames = int(parts[3])
|
||||||
|
return base + (frames / frame_rate) if frame_rate > 0 else base
|
||||||
|
return None
|
||||||
|
|
||||||
|
def _raw_markers_to_batch(
|
||||||
|
raw_markers: list[dict],
|
||||||
|
marker_type: str = "chapter",
|
||||||
|
max_label: int | None = None,
|
||||||
|
) -> list[dict]:
|
||||||
|
"""Convert raw {seconds, text} marker dicts to batch_add_markers format.
|
||||||
|
|
||||||
|
Shared by import_srt_markers and import_transcript_markers.
|
||||||
|
"""
|
||||||
|
batch = []
|
||||||
|
for m in raw_markers:
|
||||||
|
label = m["text"]
|
||||||
|
if max_label and len(label) > max_label:
|
||||||
|
label = label[:max_label]
|
||||||
|
batch.append({
|
||||||
|
"timecode": f"{m['seconds']}s",
|
||||||
|
"name": label,
|
||||||
|
"marker_type": marker_type.upper(),
|
||||||
|
})
|
||||||
|
return batch
|
||||||
|
|
||||||
|
def _extract_subtitle_blocks(text: str, *, strip_vtt_tags: bool = False) -> list[dict]:
|
||||||
|
"""Extract timestamp/text pairs from subtitle cue blocks (SRT or VTT).
|
||||||
|
|
||||||
|
Both SRT and VTT use the same ``start --> end`` cue syntax with
|
||||||
|
text lines underneath; only header stripping and tag cleaning differ.
|
||||||
|
"""
|
||||||
|
markers = []
|
||||||
|
blocks = re.split(r'\n\s*\n', text.strip())
|
||||||
|
for block in blocks:
|
||||||
|
lines = block.strip().split('\n')
|
||||||
|
if len(lines) < 2:
|
||||||
|
continue
|
||||||
|
ts_line = None
|
||||||
|
text_lines = []
|
||||||
|
for line in lines:
|
||||||
|
if '-->' in line:
|
||||||
|
ts_line = line
|
||||||
|
elif ts_line is not None:
|
||||||
|
if strip_vtt_tags:
|
||||||
|
line = re.sub(r'<[^>]+>', '', line)
|
||||||
|
cleaned = line.strip()
|
||||||
|
if cleaned:
|
||||||
|
text_lines.append(cleaned)
|
||||||
|
if not ts_line or not text_lines:
|
||||||
|
continue
|
||||||
|
start_str = ts_line.split('-->')[0].strip().replace(',', '.')
|
||||||
|
seconds = _parse_timestamp_parts(start_str.split(':'))
|
||||||
|
if seconds is not None:
|
||||||
|
markers.append({'seconds': seconds, 'text': ' '.join(text_lines)})
|
||||||
|
return markers
|
||||||
|
|
||||||
|
def parse_srt(text: str) -> list[dict]:
|
||||||
|
"""Parse SRT subtitle format into timestamp/text pairs."""
|
||||||
|
return _extract_subtitle_blocks(text)
|
||||||
|
|
||||||
|
def parse_vtt(text: str) -> list[dict]:
|
||||||
|
"""Parse WebVTT subtitle format into timestamp/text pairs."""
|
||||||
|
text = re.sub(r'^WEBVTT.*?\n', '', text, flags=re.MULTILINE)
|
||||||
|
text = re.sub(r'NOTE\n.*?\n\n', '', text, flags=re.DOTALL)
|
||||||
|
return _extract_subtitle_blocks(text, strip_vtt_tags=True)
|
||||||
|
|
||||||
|
def parse_transcript_timestamps(text: str) -> list[dict]:
|
||||||
|
"""Parse timestamped text (YouTube description format) into markers.
|
||||||
|
|
||||||
|
Supports formats like:
|
||||||
|
0:00 Introduction
|
||||||
|
00:01:30 Main Topic
|
||||||
|
1:05:30 Conclusion
|
||||||
|
00:00:00:00 SMPTE timecode
|
||||||
|
"""
|
||||||
|
markers = []
|
||||||
|
for line in text.strip().split('\n'):
|
||||||
|
line = line.strip()
|
||||||
|
if not line:
|
||||||
|
continue
|
||||||
|
match = re.match(r'^(\d{1,2}:\d{2}(?::\d{2}){0,2})\s+(.+)$', line)
|
||||||
|
if match:
|
||||||
|
seconds = _parse_timestamp_parts(match.group(1).split(':'))
|
||||||
|
if seconds is not None:
|
||||||
|
markers.append({'seconds': seconds, 'text': match.group(2).strip()})
|
||||||
|
return markers
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
"""Detecção para QC: flash frames, buracos e clipes duplicados.
|
||||||
|
|
||||||
|
Extraído de _shared.py — ver server_tools/_shared/__init__.py.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from fcpxml.models import (
|
||||||
|
DuplicateGroup,
|
||||||
|
FlashFrame,
|
||||||
|
FlashFrameSeverity,
|
||||||
|
GapInfo,
|
||||||
|
Timecode,
|
||||||
|
)
|
||||||
|
|
||||||
|
from .formatting import format_timecode
|
||||||
|
|
||||||
|
|
||||||
|
def _detect_flash_frames(
|
||||||
|
tl: Any, *, critical_threshold: int = 2, warning_threshold: int = 6,
|
||||||
|
) -> list:
|
||||||
|
"""Find clips shorter than *warning_threshold* frames.
|
||||||
|
|
||||||
|
Returns a list of ``FlashFrame`` objects sorted by severity. Shared by
|
||||||
|
``handle_detect_flash_frames`` and ``handle_validate_timeline`` so the
|
||||||
|
detection logic lives in exactly one place.
|
||||||
|
"""
|
||||||
|
fps = tl.frame_rate
|
||||||
|
flash_frames: list[FlashFrame] = []
|
||||||
|
for clip in tl.clips:
|
||||||
|
duration_frames = int(clip.duration_seconds * fps)
|
||||||
|
if duration_frames < warning_threshold:
|
||||||
|
severity = (
|
||||||
|
FlashFrameSeverity.CRITICAL
|
||||||
|
if duration_frames < critical_threshold
|
||||||
|
else FlashFrameSeverity.WARNING
|
||||||
|
)
|
||||||
|
flash_frames.append(FlashFrame(
|
||||||
|
clip_name=clip.name, clip_id=clip.name,
|
||||||
|
start=clip.start, duration_frames=duration_frames,
|
||||||
|
duration_seconds=clip.duration_seconds, severity=severity,
|
||||||
|
))
|
||||||
|
return flash_frames
|
||||||
|
|
||||||
|
def _detect_gaps(tl: Any, *, min_gap_frames: int = 1) -> list:
|
||||||
|
"""Find inter-clip gaps of at least *min_gap_frames* length.
|
||||||
|
|
||||||
|
Returns a list of ``GapInfo`` objects. Shared by ``handle_detect_gaps``
|
||||||
|
and ``handle_validate_timeline``.
|
||||||
|
"""
|
||||||
|
fps = tl.frame_rate
|
||||||
|
min_gap_seconds = min_gap_frames / fps
|
||||||
|
gaps: list[GapInfo] = []
|
||||||
|
sorted_clips = sorted(tl.clips, key=lambda c: c.start.seconds)
|
||||||
|
for i in range(len(sorted_clips) - 1):
|
||||||
|
current_end = sorted_clips[i].end.seconds
|
||||||
|
next_start = sorted_clips[i + 1].start.seconds
|
||||||
|
gap_duration = next_start - current_end
|
||||||
|
if gap_duration >= min_gap_seconds:
|
||||||
|
gaps.append(GapInfo(
|
||||||
|
start=Timecode(frames=int(current_end * fps), frame_rate=fps),
|
||||||
|
duration_frames=int(gap_duration * fps),
|
||||||
|
duration_seconds=gap_duration,
|
||||||
|
previous_clip=sorted_clips[i].name,
|
||||||
|
next_clip=sorted_clips[i + 1].name,
|
||||||
|
))
|
||||||
|
return gaps
|
||||||
|
|
||||||
|
def _detect_duplicate_groups(tl: Any, *, mode: str = "same_source") -> list:
|
||||||
|
"""Group clips that share a source media reference.
|
||||||
|
|
||||||
|
Returns a list of ``DuplicateGroup`` objects. Shared by
|
||||||
|
``handle_detect_duplicates`` and ``handle_validate_timeline``.
|
||||||
|
"""
|
||||||
|
source_groups: dict[str, list[dict]] = {}
|
||||||
|
for clip in tl.clips:
|
||||||
|
source_key = clip.media_path or clip.name
|
||||||
|
if source_key not in source_groups:
|
||||||
|
source_groups[source_key] = []
|
||||||
|
source_groups[source_key].append({
|
||||||
|
'name': clip.name,
|
||||||
|
'start': clip.start.seconds,
|
||||||
|
'duration': clip.duration_seconds,
|
||||||
|
'source_start': clip.source_start.seconds if clip.source_start else 0,
|
||||||
|
'source_duration': clip.duration_seconds,
|
||||||
|
'timecode': format_timecode(clip.start),
|
||||||
|
})
|
||||||
|
|
||||||
|
duplicates: list[DuplicateGroup] = []
|
||||||
|
for source_key, clips in source_groups.items():
|
||||||
|
if len(clips) <= 1:
|
||||||
|
continue
|
||||||
|
group = DuplicateGroup(
|
||||||
|
source_ref=source_key,
|
||||||
|
source_name=source_key.split('/')[-1] if '/' in source_key else source_key,
|
||||||
|
clips=clips,
|
||||||
|
)
|
||||||
|
if mode == "same_source":
|
||||||
|
duplicates.append(group)
|
||||||
|
elif mode == "overlapping_ranges" and group.has_overlapping_ranges:
|
||||||
|
duplicates.append(group)
|
||||||
|
elif mode == "identical":
|
||||||
|
seen_ranges: set[tuple] = set()
|
||||||
|
identical_clips = []
|
||||||
|
for c in clips:
|
||||||
|
range_key = (c['source_start'], c['source_duration'])
|
||||||
|
if range_key in seen_ranges:
|
||||||
|
identical_clips.append(c)
|
||||||
|
seen_ranges.add(range_key)
|
||||||
|
if identical_clips:
|
||||||
|
group.clips = identical_clips
|
||||||
|
duplicates.append(group)
|
||||||
|
return duplicates
|
||||||
@@ -0,0 +1,114 @@
|
|||||||
|
"""Formatação do texto que os handlers devolvem — tabelas e relatórios.
|
||||||
|
|
||||||
|
Extraído de _shared.py — ver server_tools/_shared/__init__.py.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import Sequence
|
||||||
|
|
||||||
|
|
||||||
|
def format_timecode(tc) -> str:
|
||||||
|
"""Format a Timecode object to SMPTE string."""
|
||||||
|
return tc.to_smpte() if tc else "00:00:00:00"
|
||||||
|
|
||||||
|
def format_duration(seconds: float) -> str:
|
||||||
|
"""Format seconds into human-readable duration."""
|
||||||
|
if seconds < 1:
|
||||||
|
return f"{seconds*1000:.0f}ms"
|
||||||
|
elif seconds < 60:
|
||||||
|
return f"{seconds:.2f}s"
|
||||||
|
return f"{int(seconds // 60)}m {seconds % 60:.1f}s"
|
||||||
|
|
||||||
|
def _format_clip_table(clips: list, header: str) -> str:
|
||||||
|
"""Render a list of clips as a markdown table with timecodes and durations.
|
||||||
|
|
||||||
|
Shared by handlers that filter clips by duration threshold
|
||||||
|
(find_short_cuts, find_long_clips).
|
||||||
|
"""
|
||||||
|
result = f"{header}\n\n| Name | TC | Duration |\n|------|----|---------|\n"
|
||||||
|
result += "\n".join(
|
||||||
|
f"| {c.name} | {format_timecode(c.start)} | {format_duration(c.duration_seconds)} |"
|
||||||
|
for c in clips
|
||||||
|
)
|
||||||
|
return result
|
||||||
|
|
||||||
|
def _markdown_table(headers: list[str], rows: list[list[str]]) -> str:
|
||||||
|
"""Build a markdown table from headers and rows.
|
||||||
|
|
||||||
|
Returns header row, separator row, and data rows as a single string.
|
||||||
|
Callers avoid repeating the ``| H1 | H2 |\\n|---|---|`` boilerplate
|
||||||
|
that appears in 15+ handlers.
|
||||||
|
"""
|
||||||
|
header_line = "| " + " | ".join(headers) + " |"
|
||||||
|
sep_line = "|" + "|".join("------" for _ in headers) + "|"
|
||||||
|
data_lines = "\n".join(
|
||||||
|
"| " + " | ".join(str(c) for c in row) + " |" for row in rows
|
||||||
|
)
|
||||||
|
return f"{header_line}\n{sep_line}\n{data_lines}"
|
||||||
|
|
||||||
|
def _format_batch_result(
|
||||||
|
title: str,
|
||||||
|
summary: dict[str, str],
|
||||||
|
headers: list[str],
|
||||||
|
rows: list[list[str]],
|
||||||
|
output_path: str,
|
||||||
|
) -> str:
|
||||||
|
"""Build a standard batch-operation result with summary, table, and save footer.
|
||||||
|
|
||||||
|
Used by batch fix handlers (flash frames, rapid trim, fill gaps) that all
|
||||||
|
share the same markdown structure: ``# Title → ## Summary → ## Details table
|
||||||
|
→ Saved to`` footer.
|
||||||
|
"""
|
||||||
|
summary_lines = "\n".join(f"- **{k}**: {v}" for k, v in summary.items())
|
||||||
|
table = _markdown_table(headers, rows)
|
||||||
|
return (
|
||||||
|
f"# {title}\n\n"
|
||||||
|
f"## Summary\n{summary_lines}\n\n"
|
||||||
|
f"## Details\n{table}\n\n"
|
||||||
|
f"Saved to: `{output_path}`"
|
||||||
|
)
|
||||||
|
|
||||||
|
def _fmt_suggestions(suggestions: list[str]) -> str:
|
||||||
|
"""Format pacing suggestions as markdown list (Python 3.10 compatible)."""
|
||||||
|
if not suggestions:
|
||||||
|
return "- Pacing looks good!"
|
||||||
|
nl = "\n"
|
||||||
|
return nl.join(f"- {s}" for s in suggestions)
|
||||||
|
|
||||||
|
def _voice_analysis_config_text(config: dict) -> str:
|
||||||
|
w = config["emphasis_weights"]
|
||||||
|
text = "# Voice Analysis Settings\n\n"
|
||||||
|
text += _markdown_table(
|
||||||
|
["Setting", "Value"],
|
||||||
|
[
|
||||||
|
["Energy threshold", f"{config['energy_threshold']:.2f}"],
|
||||||
|
["Peak selection", f"top {config['peak_percentile']:.1%} of words"],
|
||||||
|
["Emphasis floor", f"{config['emphasis_floor']:.2f}"],
|
||||||
|
["Emotion detection", "on" if config["emotion_enabled"] else "off"],
|
||||||
|
["Emotion sensitivity", f"{config['emotion_sensitivity']:.2f}"],
|
||||||
|
],
|
||||||
|
) + "\n\n## Emphasis Weights\n"
|
||||||
|
text += _markdown_table(
|
||||||
|
["Factor", "Weight"],
|
||||||
|
[[k.replace("_", " ").title(), f"{v:.2f}"] for k, v in w.items()],
|
||||||
|
)
|
||||||
|
return text
|
||||||
|
|
||||||
|
def _speaker_table(profiles: Sequence[dict]) -> str:
|
||||||
|
"""Who was detected, ordered by how much of the runtime each holds."""
|
||||||
|
return _markdown_table(
|
||||||
|
["ID", "Name", "Share", "Speaking", "Lines", "Avg line"],
|
||||||
|
[
|
||||||
|
[
|
||||||
|
p["id"],
|
||||||
|
p.get("name", ""),
|
||||||
|
f"{p['share']:.0%}",
|
||||||
|
format_duration(p["speaking_seconds"]),
|
||||||
|
str(p["segment_count"]),
|
||||||
|
f"{p['avg_segment']:.1f}s",
|
||||||
|
]
|
||||||
|
for p in profiles
|
||||||
|
],
|
||||||
|
)
|
||||||
|
|
||||||
@@ -0,0 +1,274 @@
|
|||||||
|
"""Mídia e transcrição: cache, corte por trecho falado e ações posicionadas.
|
||||||
|
|
||||||
|
Extraído de _shared.py — ver server_tools/_shared/__init__.py.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from fcpxml.media_intel import media_src_to_path
|
||||||
|
from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config
|
||||||
|
from fcpxml.models import (
|
||||||
|
TimeValue,
|
||||||
|
)
|
||||||
|
from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text
|
||||||
|
from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe
|
||||||
|
|
||||||
|
from .formatting import _markdown_table, format_duration
|
||||||
|
from .paths import _validate_output_path
|
||||||
|
from .project import _text_result
|
||||||
|
|
||||||
|
AUDIO_MEDIA_EXTENSIONS = (
|
||||||
|
'.wav', '.aif', '.aiff', '.mp3', '.m4a', '.aac', '.flac', '.mov', '.mp4',
|
||||||
|
)
|
||||||
|
|
||||||
|
_DIARIZATION_INSTALL_HINT = (
|
||||||
|
"\n\nInstall the optional diarization extra:\n\n"
|
||||||
|
" pip install 'fcp-mcp-server[diarization]'\n\n"
|
||||||
|
"and set a HuggingFace token with access to "
|
||||||
|
"pyannote/speaker-diarization-3.1 (pass hf_token= or persist one via "
|
||||||
|
"save_hf_token)."
|
||||||
|
)
|
||||||
|
|
||||||
|
_FEATURES_INSTALL_HINT = (
|
||||||
|
"\n\nInstall the optional media-intelligence extra:\n\n"
|
||||||
|
" pip install 'fcp-mcp-server[intelligence]'"
|
||||||
|
)
|
||||||
|
|
||||||
|
def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str:
|
||||||
|
"""Apply one non-cut action to the clip that hosts it.
|
||||||
|
|
||||||
|
``clip_start`` is where that clip begins on the timeline; the writer
|
||||||
|
wants times relative to the clip's own head, so the rebase happens here
|
||||||
|
— the single place that knows about the conversion. The clip *element*
|
||||||
|
is passed through rather than its name: after a cut the pieces share a
|
||||||
|
name, and a name lookup would land every edit on the first piece.
|
||||||
|
"""
|
||||||
|
rel_start = action.start - clip_start
|
||||||
|
rel_end = action.end - clip_start
|
||||||
|
|
||||||
|
if action.kind == "zoom":
|
||||||
|
config = load_voice_analysis_config()
|
||||||
|
# Only forward an explicit ease — otherwise add_zoom's own default
|
||||||
|
# (a fast ramp in, instant snap back out) is what should apply.
|
||||||
|
zoom_args = {
|
||||||
|
"ease": float(action.params.get("ease", config["zoom_ease_in"])),
|
||||||
|
"ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])),
|
||||||
|
}
|
||||||
|
mode = str(action.params.get("mode", config["zoom_mode"]))
|
||||||
|
if mode == "in":
|
||||||
|
zoom_args["hold_at_end"] = True
|
||||||
|
zoom_args["start_at_peak"] = False
|
||||||
|
elif mode == "out":
|
||||||
|
zoom_args["hold_at_end"] = False
|
||||||
|
zoom_args["start_at_peak"] = True
|
||||||
|
elif mode == "in_out":
|
||||||
|
zoom_args["hold_at_end"] = False
|
||||||
|
zoom_args["start_at_peak"] = False
|
||||||
|
modifier.add_zoom(
|
||||||
|
clip_id=clip_el,
|
||||||
|
start=rel_start,
|
||||||
|
end=rel_end,
|
||||||
|
scale=float(action.params.get("scale", config["zoom_scale"])),
|
||||||
|
**zoom_args,
|
||||||
|
)
|
||||||
|
return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x"
|
||||||
|
|
||||||
|
if action.kind == "text":
|
||||||
|
# Default to the "Legendas Dinâmicas" emphasis style (the font used
|
||||||
|
# to highlight a word in the captions) rather than a hardcoded
|
||||||
|
# Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of
|
||||||
|
# the video's on-screen text instead of looking like a stray default
|
||||||
|
# title. Any of these the action itself specifies still wins.
|
||||||
|
subtitle_cfg = load_dynamic_subtitle_config()
|
||||||
|
font = action.params.get("font", subtitle_cfg["emphasis_font"])
|
||||||
|
face = action.params.get("face", subtitle_cfg["emphasis_face"])
|
||||||
|
font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0)
|
||||||
|
requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"]))
|
||||||
|
requested_kerning = float(action.params.get("kerning", 0.0) or 0.0)
|
||||||
|
|
||||||
|
# Voice-action callouts are not part of the dynamic subtitle block.
|
||||||
|
# When omitted, put them above the subtitle band and shrink wide
|
||||||
|
# phrases to the title-safe width. The previous default (Position 0 0,
|
||||||
|
# full emphasis size) made long callouts like "PRÓTESES DE SILICONE"
|
||||||
|
# collide with captions and run off both sides of a vertical frame.
|
||||||
|
emitted_size = requested_size * font_scale
|
||||||
|
emitted_kerning = requested_kerning * font_scale
|
||||||
|
safe_width = modifier.frame_width() * 0.90
|
||||||
|
width = measure_text(
|
||||||
|
action.params["content"],
|
||||||
|
emitted_size,
|
||||||
|
bold=bool(action.params.get("bold", False)),
|
||||||
|
kerning=emitted_kerning,
|
||||||
|
font=font,
|
||||||
|
face=face,
|
||||||
|
)
|
||||||
|
font_size = requested_size
|
||||||
|
if width > safe_width and width > 0:
|
||||||
|
font_size = max(32, int(requested_size * safe_width / width))
|
||||||
|
position = action.params.get("position")
|
||||||
|
if not position:
|
||||||
|
position = f"0 {modifier.frame_height() * 0.23:g}"
|
||||||
|
|
||||||
|
modifier.add_text_title(
|
||||||
|
clip_el,
|
||||||
|
action.params["content"],
|
||||||
|
offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
|
||||||
|
duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(),
|
||||||
|
position=position,
|
||||||
|
font=font,
|
||||||
|
font_size=font_size,
|
||||||
|
font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]),
|
||||||
|
face=face,
|
||||||
|
bold=action.params.get("bold", False),
|
||||||
|
)
|
||||||
|
return f"text \"{action.params['content'][:24]}\""
|
||||||
|
|
||||||
|
# marker
|
||||||
|
modifier.add_marker(
|
||||||
|
clip_id=clip_el,
|
||||||
|
timecode=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
|
||||||
|
name=action.params.get("content") or action.reason or "Voice action",
|
||||||
|
note=action.reason or None,
|
||||||
|
)
|
||||||
|
return "marker"
|
||||||
|
|
||||||
|
TRANSCRIBE_MAX_MEDIA = 10
|
||||||
|
|
||||||
|
_TRANSCRIBE_INSTALL_HINT = (
|
||||||
|
"\n\nInstall the optional transcription extra:\n\n"
|
||||||
|
" pip install 'fcp-mcp-server[transcribe]'\n\n"
|
||||||
|
"or run via uvx:\n\n"
|
||||||
|
" uvx --from \"fcp-mcp-server[transcribe]\" fcp-mcp-server"
|
||||||
|
)
|
||||||
|
|
||||||
|
def _transcript_json_path(media_path: str, output_dir: str | None = None) -> Path:
|
||||||
|
"""Where the ``_transcript.json`` for ``media_path`` lives.
|
||||||
|
|
||||||
|
When ``output_dir`` (the user-selected project folder) is set, the
|
||||||
|
transcript is saved/read there instead of next to the source media.
|
||||||
|
"""
|
||||||
|
p = Path(media_path)
|
||||||
|
if output_dir:
|
||||||
|
directory = Path(output_dir).expanduser()
|
||||||
|
directory.mkdir(parents=True, exist_ok=True)
|
||||||
|
return directory / f"{p.stem}_transcript.json"
|
||||||
|
return p.with_name(p.stem + "_transcript.json")
|
||||||
|
|
||||||
|
def _load_or_transcribe(
|
||||||
|
media_path: str, model: str, language: str | None, output_dir: str | None = None
|
||||||
|
) -> tuple[dict | None, str]:
|
||||||
|
"""Load a cached ``_transcript.json`` for a media file, else transcribe and cache it.
|
||||||
|
|
||||||
|
Returns ``(transcript, "")`` or ``(None, reason)``. The cache makes
|
||||||
|
transcription a one-time cost per media file across all transcript tools.
|
||||||
|
"""
|
||||||
|
json_path = _transcript_json_path(media_path, output_dir)
|
||||||
|
if json_path.is_file():
|
||||||
|
try:
|
||||||
|
with open(json_path) as f:
|
||||||
|
data = json.load(f)
|
||||||
|
if isinstance(data, dict) and isinstance(data.get("words"), list):
|
||||||
|
return data, ""
|
||||||
|
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
||||||
|
pass # unreadable cache falls through to re-transcribe
|
||||||
|
result = transcribe(media_path, model_size=model, language=language)
|
||||||
|
if result is None:
|
||||||
|
return None, "untranscribable (faster-whisper not installed or media unreadable)"
|
||||||
|
anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent)
|
||||||
|
out_path = _validate_output_path(str(json_path), anchor_dir=anchor)
|
||||||
|
with open(out_path, "w") as f:
|
||||||
|
json.dump({"source": Path(media_path).name, **result}, f, indent=2)
|
||||||
|
return result, ""
|
||||||
|
|
||||||
|
def _cut_transcript_spans(modifier, clip_filter, model, language, padding, spans_fn, keep_only=False, output_dir=None):
|
||||||
|
"""Shared cut engine for transcript-driven editing.
|
||||||
|
|
||||||
|
``spans_fn(words) -> [(start, end), ...]`` in source seconds. Spans are
|
||||||
|
padded, clamped to each clip's used source window, optionally inverted
|
||||||
|
(keep_only), snapped to the frame grid, and cut with ripple.
|
||||||
|
"""
|
||||||
|
to_frame = modifier.snap_seconds_to_frame
|
||||||
|
|
||||||
|
cache: dict[str, tuple] = {}
|
||||||
|
cuts_made: list[tuple[str, int, float]] = []
|
||||||
|
skipped: list[tuple[str, str]] = []
|
||||||
|
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
||||||
|
for el in spine_clips:
|
||||||
|
name = el.get("name", "")
|
||||||
|
if clip_filter and name != clip_filter:
|
||||||
|
continue
|
||||||
|
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
||||||
|
media_path = media_src_to_path(src)
|
||||||
|
if not media_path or not Path(media_path).is_file():
|
||||||
|
skipped.append((name, "media file missing"))
|
||||||
|
continue
|
||||||
|
if media_path not in cache:
|
||||||
|
if len(cache) >= TRANSCRIBE_MAX_MEDIA:
|
||||||
|
skipped.append((name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)"))
|
||||||
|
continue
|
||||||
|
cache[media_path] = _load_or_transcribe(media_path, model, language, output_dir)
|
||||||
|
data, reason = cache[media_path]
|
||||||
|
if data is None:
|
||||||
|
skipped.append((name, reason))
|
||||||
|
continue
|
||||||
|
|
||||||
|
clip_source_start = modifier.source_file_start(el).to_seconds()
|
||||||
|
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
||||||
|
window_start = clip_source_start
|
||||||
|
window_end = clip_source_start + clip_duration
|
||||||
|
|
||||||
|
spans = spans_fn(data.get("words", []))
|
||||||
|
padded = merge_ranges([(s - padding, e + padding) for s, e in spans])
|
||||||
|
clamped = [
|
||||||
|
(max(s, window_start), min(e, window_end))
|
||||||
|
for s, e in padded
|
||||||
|
if min(e, window_end) > max(s, window_start)
|
||||||
|
]
|
||||||
|
if keep_only:
|
||||||
|
if not clamped:
|
||||||
|
# Never delete a whole clip just because nothing matched in it.
|
||||||
|
skipped.append((name, "no phrase matches — left untouched (keep_only)"))
|
||||||
|
continue
|
||||||
|
cut_source = invert_ranges(clamped, window_start, window_end)
|
||||||
|
else:
|
||||||
|
cut_source = clamped
|
||||||
|
cut_ranges = [
|
||||||
|
(to_frame(s - clip_source_start), to_frame(e - clip_source_start))
|
||||||
|
for s, e in cut_source
|
||||||
|
]
|
||||||
|
cut_ranges = [(a, b) for a, b in cut_ranges if b > a]
|
||||||
|
if not cut_ranges:
|
||||||
|
continue
|
||||||
|
removed = modifier.cut_clip_ranges(el, cut_ranges)
|
||||||
|
if removed > TimeValue.zero():
|
||||||
|
cuts_made.append((name, len(cut_ranges), removed.to_seconds()))
|
||||||
|
return cuts_made, skipped
|
||||||
|
|
||||||
|
def _transcript_cut_report(title, summary_lines, cuts_made, skipped, output_path, footer):
|
||||||
|
if not cuts_made:
|
||||||
|
text = f"# {title}\n\nNo cuts to make — file unchanged (nothing saved)."
|
||||||
|
if skipped:
|
||||||
|
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
||||||
|
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
|
||||||
|
)
|
||||||
|
if any("faster-whisper" in reason for _, reason in skipped):
|
||||||
|
text += _TRANSCRIBE_INSTALL_HINT
|
||||||
|
return _text_result(text)
|
||||||
|
total_removed = sum(seconds for _, _, seconds in cuts_made)
|
||||||
|
result = f"# {title}\n\n## Summary\n"
|
||||||
|
result += "\n".join(summary_lines) + "\n"
|
||||||
|
result += f"- **Clips Cut**: {len(cuts_made)}\n- **Total Removed**: {format_duration(total_removed)}\n"
|
||||||
|
result += "\n## Cuts\n"
|
||||||
|
result += _markdown_table(
|
||||||
|
["Clip", "Ranges Cut", "Removed"],
|
||||||
|
[[name, str(count), f"{seconds:.2f}s"] for name, count, seconds in cuts_made],
|
||||||
|
) + "\n"
|
||||||
|
if skipped:
|
||||||
|
result += "\n## Skipped Clips\n" + _markdown_table(
|
||||||
|
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
|
||||||
|
) + "\n"
|
||||||
|
result += f"\nSaved to: {output_path}\n\n{footer}"
|
||||||
|
return _text_result(result)
|
||||||
@@ -0,0 +1,226 @@
|
|||||||
|
"""Caminhos: validação contra a sandbox, limites de tamanho, saída derivada.
|
||||||
|
|
||||||
|
Extraído de _shared.py — ver server_tools/_shared/__init__.py.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
PROJECTS_DIR = os.environ.get("FCP_PROJECTS_DIR", os.path.expanduser("~/Movies"))
|
||||||
|
|
||||||
|
_SANDBOX_ENABLED = "FCP_PROJECTS_DIR" in os.environ
|
||||||
|
|
||||||
|
MAX_FILE_SIZE = 100 * 1024 * 1024
|
||||||
|
|
||||||
|
MAX_MEDIA_FILE_SIZE = 32 * 1024 * 1024 * 1024
|
||||||
|
|
||||||
|
_MAX_JSON_DEPTH = 50
|
||||||
|
|
||||||
|
def _check_json_depth(obj: object, _depth: int = 0) -> None:
|
||||||
|
"""Reject JSON structures nested beyond _MAX_JSON_DEPTH.
|
||||||
|
|
||||||
|
Prevents denial-of-service via deeply nested objects that exhaust the
|
||||||
|
call stack or memory during downstream processing. Called after
|
||||||
|
json.load() since Python's json module has no built-in depth limit.
|
||||||
|
"""
|
||||||
|
if _depth > _MAX_JSON_DEPTH:
|
||||||
|
raise ValueError(
|
||||||
|
f"JSON nesting depth exceeds {_MAX_JSON_DEPTH} — "
|
||||||
|
"file may be malformed or adversarial"
|
||||||
|
)
|
||||||
|
if isinstance(obj, dict):
|
||||||
|
for v in obj.values():
|
||||||
|
_check_json_depth(v, _depth + 1)
|
||||||
|
elif isinstance(obj, list):
|
||||||
|
for item in obj:
|
||||||
|
_check_json_depth(item, _depth + 1)
|
||||||
|
|
||||||
|
def _validate_filepath(
|
||||||
|
filepath: str,
|
||||||
|
allowed_extensions: tuple[str, ...] | None = None,
|
||||||
|
max_size: int = MAX_FILE_SIZE,
|
||||||
|
) -> str:
|
||||||
|
"""Validate a user-provided file path against traversal and size attacks.
|
||||||
|
|
||||||
|
Resolves symlinks, blocks null bytes, enforces extension whitelist, and
|
||||||
|
checks file size before any parsing takes place.
|
||||||
|
|
||||||
|
``max_size`` defaults to the document limit; callers handling source
|
||||||
|
media pass ``MAX_MEDIA_FILE_SIZE``, since media is streamed rather than
|
||||||
|
parsed into memory (see the constant for why).
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: For invalid paths (null bytes, bad extensions, oversized).
|
||||||
|
FileNotFoundError: When the resolved path does not exist.
|
||||||
|
"""
|
||||||
|
if '\x00' in filepath:
|
||||||
|
raise ValueError("Invalid file path: null byte detected")
|
||||||
|
|
||||||
|
resolved = Path(filepath).resolve()
|
||||||
|
|
||||||
|
if not resolved.exists():
|
||||||
|
raise FileNotFoundError(f"File not found: {filepath}")
|
||||||
|
|
||||||
|
# .fcpxmld bundles are directories (a package wrapping Info.fcpxml plus
|
||||||
|
# sidecar data files for object tracking / Cinematic mode). The size
|
||||||
|
# check applies to the inner Info.fcpxml, which is what gets parsed.
|
||||||
|
if resolved.is_dir():
|
||||||
|
if resolved.suffix.lower() != '.fcpxmld':
|
||||||
|
raise ValueError(f"Not a regular file: {filepath}")
|
||||||
|
inner = resolved / 'Info.fcpxml'
|
||||||
|
if not inner.is_file():
|
||||||
|
raise ValueError(f"Invalid bundle (no Info.fcpxml): {filepath}")
|
||||||
|
size_target = inner
|
||||||
|
elif not resolved.is_file():
|
||||||
|
raise ValueError(f"Not a regular file: {filepath}")
|
||||||
|
else:
|
||||||
|
size_target = resolved
|
||||||
|
|
||||||
|
if allowed_extensions and resolved.suffix.lower() not in allowed_extensions:
|
||||||
|
raise ValueError(
|
||||||
|
f"Invalid file type '{resolved.suffix}'. "
|
||||||
|
f"Allowed: {', '.join(allowed_extensions)}"
|
||||||
|
)
|
||||||
|
|
||||||
|
if size_target.stat().st_size > max_size:
|
||||||
|
size_mb = size_target.stat().st_size / (1024 * 1024)
|
||||||
|
raise ValueError(f"File too large ({size_mb:.1f} MB). Maximum: {max_size // (1024 * 1024)} MB")
|
||||||
|
|
||||||
|
return str(resolved)
|
||||||
|
|
||||||
|
def _validate_output_path(output_path: str, *, anchor_dir: str | None = None) -> str:
|
||||||
|
"""Validate an output path with optional sandbox enforcement.
|
||||||
|
|
||||||
|
Resolves traversal, blocks null bytes, ensures parent exists, and — when
|
||||||
|
*anchor_dir* is provided — verifies the resolved output lives under that
|
||||||
|
directory. This prevents LLM-generated tool calls from writing to
|
||||||
|
arbitrary filesystem locations (e.g. ``/etc/cron.d/backdoor``).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
output_path: The raw output path to validate.
|
||||||
|
anchor_dir: If set, the resolved output must be a child of this
|
||||||
|
directory. Typically the parent directory of the input file so
|
||||||
|
outputs stay co-located with their sources.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: For null bytes, missing parent, or sandbox escape.
|
||||||
|
"""
|
||||||
|
if '\x00' in output_path:
|
||||||
|
raise ValueError("Invalid output path: null byte detected")
|
||||||
|
|
||||||
|
resolved = Path(output_path).resolve()
|
||||||
|
|
||||||
|
if not resolved.parent.exists():
|
||||||
|
raise ValueError(f"Output directory does not exist: {resolved.parent}")
|
||||||
|
|
||||||
|
if anchor_dir is not None:
|
||||||
|
anchor = Path(anchor_dir).resolve()
|
||||||
|
try:
|
||||||
|
resolved.relative_to(anchor)
|
||||||
|
except ValueError:
|
||||||
|
raise ValueError(
|
||||||
|
f"Output path escapes allowed directory: "
|
||||||
|
f"{resolved} is not under {anchor}"
|
||||||
|
)
|
||||||
|
|
||||||
|
return str(resolved)
|
||||||
|
|
||||||
|
def _validate_directory(directory: str, *, allowed_root: str | None = None) -> str:
|
||||||
|
"""Validate a user-provided directory path against traversal and injection.
|
||||||
|
|
||||||
|
Resolves symlinks, blocks null bytes, and verifies the path is a real
|
||||||
|
directory. When *allowed_root* is given, the resolved path must be a
|
||||||
|
descendant of (or equal to) that root — preventing filesystem enumeration
|
||||||
|
beyond the project workspace.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ValueError: For invalid paths (null bytes, not a directory, sandbox escape).
|
||||||
|
"""
|
||||||
|
if '\x00' in directory:
|
||||||
|
raise ValueError("Invalid directory path: null byte detected")
|
||||||
|
|
||||||
|
resolved = Path(directory).resolve()
|
||||||
|
|
||||||
|
if not resolved.is_dir():
|
||||||
|
raise ValueError(f"Not a valid directory: {directory}")
|
||||||
|
|
||||||
|
if allowed_root is not None:
|
||||||
|
root = Path(allowed_root).resolve()
|
||||||
|
try:
|
||||||
|
resolved.relative_to(root)
|
||||||
|
except ValueError:
|
||||||
|
raise ValueError(
|
||||||
|
f"Directory escapes allowed root: "
|
||||||
|
f"{resolved} is not under {root}"
|
||||||
|
)
|
||||||
|
|
||||||
|
return str(resolved)
|
||||||
|
|
||||||
|
def find_fcpxml_files(directory: str) -> list[str]:
|
||||||
|
"""Find all FCPXML files in a directory."""
|
||||||
|
path = Path(directory)
|
||||||
|
files = list(str(f) for f in path.rglob("*.fcpxml"))
|
||||||
|
files.extend(str(f) for f in path.rglob("*.fcpxmld"))
|
||||||
|
return sorted(files)
|
||||||
|
|
||||||
|
def generate_output_path(input_path: str, suffix: str = "_modified") -> str:
|
||||||
|
"""Generate output path from input path.
|
||||||
|
|
||||||
|
The suffix is sanitized to prevent path-component injection — only
|
||||||
|
alphanumeric, hyphen, underscore, and dot characters survive.
|
||||||
|
"""
|
||||||
|
# Strip anything that could inject path separators or traversal sequences
|
||||||
|
clean_suffix = re.sub(r'[^a-zA-Z0-9._-]', '', suffix)
|
||||||
|
if not clean_suffix:
|
||||||
|
clean_suffix = "_modified"
|
||||||
|
p = Path(input_path)
|
||||||
|
return str(p.parent / f"{p.stem}{clean_suffix}{p.suffix}")
|
||||||
|
|
||||||
|
def _resolve_io_paths(
|
||||||
|
arguments: dict,
|
||||||
|
suffix: str = "_modified",
|
||||||
|
) -> tuple[str, str]:
|
||||||
|
"""Validate input filepath and resolve the output path.
|
||||||
|
|
||||||
|
Shared foundation for every handler that reads an FCPXML and writes
|
||||||
|
a derived file. Validates the input, falls back to a suffixed
|
||||||
|
output name when ``output_path`` is not supplied, and sandbox-checks
|
||||||
|
the result.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
arguments: Tool arguments dict (must contain ``filepath``; may
|
||||||
|
contain ``output_path``).
|
||||||
|
suffix: Default output filename suffix when ``output_path`` is
|
||||||
|
not provided (e.g. ``"_modified"``, ``"_beats"``).
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
``(filepath, output_path)`` tuple with both paths validated.
|
||||||
|
"""
|
||||||
|
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
|
||||||
|
# Anchor write operations to the input file's directory so LLM-generated
|
||||||
|
# tool calls cannot write to arbitrary filesystem locations (e.g.
|
||||||
|
# /etc/cron.d/backdoor). When the explicit sandbox is off, the anchor
|
||||||
|
# still prevents writes outside the source directory tree.
|
||||||
|
# `output_dir` is where the caller wants the file written, not merely a
|
||||||
|
# sandbox boundary: the app's "Pasta do projeto" promises that everything
|
||||||
|
# generated lands there. Deriving the name from the input but keeping the
|
||||||
|
# input's directory made every cross-directory call fail its own anchor
|
||||||
|
# check ("output path escapes allowed directory"), so the setting silently
|
||||||
|
# only worked when it pointed at the directory the file was already going
|
||||||
|
# to. An explicit `output_path` still wins, and still has to sit inside
|
||||||
|
# the anchor.
|
||||||
|
output_dir = arguments.get("output_dir")
|
||||||
|
if output_dir:
|
||||||
|
anchor = _validate_directory(str(output_dir))
|
||||||
|
default_output = str(Path(anchor) / Path(generate_output_path(filepath, suffix)).name)
|
||||||
|
else:
|
||||||
|
anchor = str(Path(filepath).resolve().parent)
|
||||||
|
default_output = generate_output_path(filepath, suffix)
|
||||||
|
output_path = _validate_output_path(
|
||||||
|
arguments.get("output_path") or default_output,
|
||||||
|
anchor_dir=anchor,
|
||||||
|
)
|
||||||
|
return filepath, output_path
|
||||||
@@ -0,0 +1,89 @@
|
|||||||
|
"""Abrir um projeto e preparar modifier/generator para editá-lo.
|
||||||
|
|
||||||
|
Extraído de _shared.py — ver server_tools/_shared/__init__.py.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from mcp.types import TextContent
|
||||||
|
|
||||||
|
from fcpxml.parser import FCPXMLParser
|
||||||
|
from fcpxml.rough_cut import RoughCutGenerator
|
||||||
|
from fcpxml.writer import FCPXMLModifier
|
||||||
|
|
||||||
|
from .paths import _resolve_io_paths, _validate_filepath
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_project(filepath: str):
|
||||||
|
"""Parse an FCPXML file and return the project with its primary timeline."""
|
||||||
|
filepath = _validate_filepath(filepath, ('.fcpxml', '.fcpxmld'))
|
||||||
|
project = FCPXMLParser().parse_file(filepath)
|
||||||
|
if not project.timelines:
|
||||||
|
return None, None
|
||||||
|
return project, project.primary_timeline
|
||||||
|
|
||||||
|
def _text_result(text: str) -> list[TextContent]:
|
||||||
|
"""Wrap a string in the MCP TextContent list that every tool handler returns."""
|
||||||
|
return [TextContent(type="text", text=text)]
|
||||||
|
|
||||||
|
def _no_timeline():
|
||||||
|
"""Standard response when no timelines are found."""
|
||||||
|
return _text_result("No timelines found")
|
||||||
|
|
||||||
|
def _require_timeline(filepath: str):
|
||||||
|
"""Parse FCPXML and return (project, timeline), raising if no timeline exists.
|
||||||
|
|
||||||
|
Centralises the repeated _parse_project + _no_timeline guard that
|
||||||
|
appears in every read-only timeline handler. Returns a tuple so
|
||||||
|
callers can destructure directly::
|
||||||
|
|
||||||
|
project, tl = _require_timeline(arguments["filepath"])
|
||||||
|
"""
|
||||||
|
project, tl = _parse_project(filepath)
|
||||||
|
if not tl:
|
||||||
|
raise _NoTimelineError()
|
||||||
|
return project, tl
|
||||||
|
|
||||||
|
class _NoTimelineError(Exception):
|
||||||
|
"""Sentinel raised by _require_timeline when no timelines exist."""
|
||||||
|
|
||||||
|
def _setup_modifier(
|
||||||
|
arguments: dict,
|
||||||
|
suffix: str = "_modified",
|
||||||
|
) -> tuple[str, str, "FCPXMLModifier"]:
|
||||||
|
"""Common setup for write handlers: validate paths and create modifier.
|
||||||
|
|
||||||
|
Consolidates the repeated validate-filepath → resolve-output-path →
|
||||||
|
create-modifier boilerplate shared by 18+ write handlers.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
arguments: Tool arguments dict (must contain ``filepath``; may
|
||||||
|
contain ``output_path``).
|
||||||
|
suffix: Default output filename suffix when ``output_path`` is
|
||||||
|
not provided (e.g. ``"_modified"``, ``"_flash_fixed"``).
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
``(filepath, output_path, modifier)`` tuple ready for the
|
||||||
|
handler's domain-specific operation.
|
||||||
|
"""
|
||||||
|
filepath, output_path = _resolve_io_paths(arguments, suffix)
|
||||||
|
modifier = FCPXMLModifier(filepath)
|
||||||
|
return filepath, output_path, modifier
|
||||||
|
|
||||||
|
def _setup_generator(
|
||||||
|
arguments: dict,
|
||||||
|
suffix: str = "_roughcut",
|
||||||
|
) -> tuple[str, str, "RoughCutGenerator"]:
|
||||||
|
"""Common setup for generation handlers: validate paths and create generator.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
arguments: Tool arguments dict (must contain ``filepath`` and
|
||||||
|
``output_path``).
|
||||||
|
suffix: Default output filename suffix.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
``(filepath, output_path, generator)`` tuple.
|
||||||
|
"""
|
||||||
|
filepath, output_path = _resolve_io_paths(arguments, suffix)
|
||||||
|
generator = RoughCutGenerator(filepath)
|
||||||
|
return filepath, output_path, generator
|
||||||
@@ -46,7 +46,7 @@ class TestDiarizeMediaHandler:
|
|||||||
await handle_diarize_media({"media_path": str(bad)})
|
await handle_diarize_media({"media_path": str(bad)})
|
||||||
|
|
||||||
async def test_writes_diarization_json_and_reports(self, tmp_path, monkeypatch):
|
async def test_writes_diarization_json_and_reports(self, tmp_path, monkeypatch):
|
||||||
import server_tools._shared as _shared_mod
|
import server_tools._shared.media as _shared_mod
|
||||||
import server_tools.voice as server_mod
|
import server_tools.voice as server_mod
|
||||||
from server import handle_diarize_media
|
from server import handle_diarize_media
|
||||||
|
|
||||||
@@ -86,7 +86,7 @@ class TestDiarizeMediaHandler:
|
|||||||
assert data["words"][1]["speaker_id"] == "SPEAKER_01"
|
assert data["words"][1]["speaker_id"] == "SPEAKER_01"
|
||||||
|
|
||||||
async def test_reports_when_diarization_fails(self, tmp_path, monkeypatch):
|
async def test_reports_when_diarization_fails(self, tmp_path, monkeypatch):
|
||||||
import server_tools._shared as _shared_mod
|
import server_tools._shared.media as _shared_mod
|
||||||
import server_tools.voice as server_mod
|
import server_tools.voice as server_mod
|
||||||
from server import handle_diarize_media
|
from server import handle_diarize_media
|
||||||
|
|
||||||
|
|||||||
@@ -32,7 +32,7 @@ def two_candidates(monkeypatch):
|
|||||||
than one candidate to cap.
|
than one candidate to cap.
|
||||||
"""
|
"""
|
||||||
import fcpxml.voice_timeline as vt
|
import fcpxml.voice_timeline as vt
|
||||||
import server_tools._shared as _shared_mod
|
import server_tools._shared.media as _shared_mod
|
||||||
|
|
||||||
transcript = {
|
transcript = {
|
||||||
"language": "pt",
|
"language": "pt",
|
||||||
|
|||||||
@@ -43,7 +43,7 @@ def wav(tmp_path):
|
|||||||
@pytest.fixture
|
@pytest.fixture
|
||||||
def patched_analysis(monkeypatch):
|
def patched_analysis(monkeypatch):
|
||||||
"""Make the tool's transcription + librosa extractors deterministic."""
|
"""Make the tool's transcription + librosa extractors deterministic."""
|
||||||
import server_tools._shared as _shared_mod
|
import server_tools._shared.media as _shared_mod
|
||||||
import server_tools.voice as server_mod
|
import server_tools.voice as server_mod
|
||||||
|
|
||||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
|
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
|
||||||
@@ -111,7 +111,7 @@ class TestAnalyzeVoiceFeaturesHandler:
|
|||||||
assert "emphasis_weights" in data["config"]
|
assert "emphasis_weights" in data["config"]
|
||||||
|
|
||||||
async def test_empty_transcript_reports_instead_of_crashing(self, wav, monkeypatch):
|
async def test_empty_transcript_reports_instead_of_crashing(self, wav, monkeypatch):
|
||||||
import server_tools._shared as _shared_mod
|
import server_tools._shared.media as _shared_mod
|
||||||
import server_tools.voice as server_mod
|
import server_tools.voice as server_mod
|
||||||
from server import handle_analyze_voice_features
|
from server import handle_analyze_voice_features
|
||||||
|
|
||||||
@@ -123,7 +123,7 @@ class TestAnalyzeVoiceFeaturesHandler:
|
|||||||
assert "no words" in result[0].text.lower()
|
assert "no words" in result[0].text.lower()
|
||||||
|
|
||||||
async def test_untranscribable_media_reports_install_hint(self, wav, monkeypatch):
|
async def test_untranscribable_media_reports_install_hint(self, wav, monkeypatch):
|
||||||
import server_tools._shared as _shared_mod
|
import server_tools._shared.media as _shared_mod
|
||||||
import server_tools.voice as server_mod
|
import server_tools.voice as server_mod
|
||||||
from server import handle_analyze_voice_features
|
from server import handle_analyze_voice_features
|
||||||
|
|
||||||
@@ -133,7 +133,7 @@ class TestAnalyzeVoiceFeaturesHandler:
|
|||||||
assert "faster-whisper" in result[0].text
|
assert "faster-whisper" in result[0].text
|
||||||
|
|
||||||
async def test_missing_pitch_track_degrades_without_crashing(self, wav, monkeypatch):
|
async def test_missing_pitch_track_degrades_without_crashing(self, wav, monkeypatch):
|
||||||
import server_tools._shared as _shared_mod
|
import server_tools._shared.media as _shared_mod
|
||||||
import server_tools.voice as server_mod
|
import server_tools.voice as server_mod
|
||||||
from server import handle_analyze_voice_features
|
from server import handle_analyze_voice_features
|
||||||
|
|
||||||
|
|||||||
@@ -34,7 +34,7 @@ def wav(tmp_path):
|
|||||||
def patched(monkeypatch):
|
def patched(monkeypatch):
|
||||||
"""Deterministic transcription + acoustics, no optional extras needed."""
|
"""Deterministic transcription + acoustics, no optional extras needed."""
|
||||||
import fcpxml.voice_timeline as vt
|
import fcpxml.voice_timeline as vt
|
||||||
import server_tools._shared as _shared_mod
|
import server_tools._shared.media as _shared_mod
|
||||||
|
|
||||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _TRANSCRIPT)
|
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _TRANSCRIPT)
|
||||||
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: [(2.4, 260.0), (0.2, 120.0)])
|
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: [(2.4, 260.0), (0.2, 120.0)])
|
||||||
@@ -71,7 +71,7 @@ class TestBuildVoiceTimelineHandler:
|
|||||||
await handle_build_voice_timeline({"media_path": str(bad)})
|
await handle_build_voice_timeline({"media_path": str(bad)})
|
||||||
|
|
||||||
async def test_untranscribable_media_reports_hint(self, wav, monkeypatch):
|
async def test_untranscribable_media_reports_hint(self, wav, monkeypatch):
|
||||||
import server_tools._shared as _shared_mod
|
import server_tools._shared.media as _shared_mod
|
||||||
from server import handle_build_voice_timeline
|
from server import handle_build_voice_timeline
|
||||||
|
|
||||||
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)
|
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)
|
||||||
|
|||||||
Reference in New Issue
Block a user