Eram 882 linhas de seis papéis sem relação, sob um nome que só dizia
"compartilhado" — o depósito onde tudo que servia a mais de um handler
acabava caindo.
media 316 transcrição em cache, corte por fala, relatório
paths 206 sandbox, limites, caminho de saída
project 116 abrir projeto, preparar modifier/generator
captions 112 SRT, VTT, listas com timestamp
detection 99 flash frames, buracos, duplicados
formatting 86 tabelas e relatórios dos handlers
O __init__ reexporta os 46 nomes, então os treze pontos que importam daqui
não mudaram.
_transcript_cut_report saiu de formatting para media: ele precisa do hint de
instalação e do _text_result, ou seja, é relatório de transcrição e não
formatação genérica — mover foi mais honesto que cruzar imports entre os
dois módulos.
Quatro testes patchavam `server_tools._shared.transcribe`; o nome agora é
ligado por _shared/media.py, então o patch passou a apontar para lá — mesmo
padrão da experiência #23.
Lint zerado, 1454 testes passando.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
227 lines
8.6 KiB
Python
227 lines
8.6 KiB
Python
"""Caminhos: validação contra a sandbox, limites de tamanho, saída derivada.
|
|
|
|
Extraído de _shared.py — ver server_tools/_shared/__init__.py.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
import re
|
|
from pathlib import Path
|
|
|
|
PROJECTS_DIR = os.environ.get("FCP_PROJECTS_DIR", os.path.expanduser("~/Movies"))
|
|
|
|
_SANDBOX_ENABLED = "FCP_PROJECTS_DIR" in os.environ
|
|
|
|
MAX_FILE_SIZE = 100 * 1024 * 1024
|
|
|
|
MAX_MEDIA_FILE_SIZE = 32 * 1024 * 1024 * 1024
|
|
|
|
_MAX_JSON_DEPTH = 50
|
|
|
|
def _check_json_depth(obj: object, _depth: int = 0) -> None:
|
|
"""Reject JSON structures nested beyond _MAX_JSON_DEPTH.
|
|
|
|
Prevents denial-of-service via deeply nested objects that exhaust the
|
|
call stack or memory during downstream processing. Called after
|
|
json.load() since Python's json module has no built-in depth limit.
|
|
"""
|
|
if _depth > _MAX_JSON_DEPTH:
|
|
raise ValueError(
|
|
f"JSON nesting depth exceeds {_MAX_JSON_DEPTH} — "
|
|
"file may be malformed or adversarial"
|
|
)
|
|
if isinstance(obj, dict):
|
|
for v in obj.values():
|
|
_check_json_depth(v, _depth + 1)
|
|
elif isinstance(obj, list):
|
|
for item in obj:
|
|
_check_json_depth(item, _depth + 1)
|
|
|
|
def _validate_filepath(
|
|
filepath: str,
|
|
allowed_extensions: tuple[str, ...] | None = None,
|
|
max_size: int = MAX_FILE_SIZE,
|
|
) -> str:
|
|
"""Validate a user-provided file path against traversal and size attacks.
|
|
|
|
Resolves symlinks, blocks null bytes, enforces extension whitelist, and
|
|
checks file size before any parsing takes place.
|
|
|
|
``max_size`` defaults to the document limit; callers handling source
|
|
media pass ``MAX_MEDIA_FILE_SIZE``, since media is streamed rather than
|
|
parsed into memory (see the constant for why).
|
|
|
|
Raises:
|
|
ValueError: For invalid paths (null bytes, bad extensions, oversized).
|
|
FileNotFoundError: When the resolved path does not exist.
|
|
"""
|
|
if '\x00' in filepath:
|
|
raise ValueError("Invalid file path: null byte detected")
|
|
|
|
resolved = Path(filepath).resolve()
|
|
|
|
if not resolved.exists():
|
|
raise FileNotFoundError(f"File not found: {filepath}")
|
|
|
|
# .fcpxmld bundles are directories (a package wrapping Info.fcpxml plus
|
|
# sidecar data files for object tracking / Cinematic mode). The size
|
|
# check applies to the inner Info.fcpxml, which is what gets parsed.
|
|
if resolved.is_dir():
|
|
if resolved.suffix.lower() != '.fcpxmld':
|
|
raise ValueError(f"Not a regular file: {filepath}")
|
|
inner = resolved / 'Info.fcpxml'
|
|
if not inner.is_file():
|
|
raise ValueError(f"Invalid bundle (no Info.fcpxml): {filepath}")
|
|
size_target = inner
|
|
elif not resolved.is_file():
|
|
raise ValueError(f"Not a regular file: {filepath}")
|
|
else:
|
|
size_target = resolved
|
|
|
|
if allowed_extensions and resolved.suffix.lower() not in allowed_extensions:
|
|
raise ValueError(
|
|
f"Invalid file type '{resolved.suffix}'. "
|
|
f"Allowed: {', '.join(allowed_extensions)}"
|
|
)
|
|
|
|
if size_target.stat().st_size > max_size:
|
|
size_mb = size_target.stat().st_size / (1024 * 1024)
|
|
raise ValueError(f"File too large ({size_mb:.1f} MB). Maximum: {max_size // (1024 * 1024)} MB")
|
|
|
|
return str(resolved)
|
|
|
|
def _validate_output_path(output_path: str, *, anchor_dir: str | None = None) -> str:
|
|
"""Validate an output path with optional sandbox enforcement.
|
|
|
|
Resolves traversal, blocks null bytes, ensures parent exists, and — when
|
|
*anchor_dir* is provided — verifies the resolved output lives under that
|
|
directory. This prevents LLM-generated tool calls from writing to
|
|
arbitrary filesystem locations (e.g. ``/etc/cron.d/backdoor``).
|
|
|
|
Args:
|
|
output_path: The raw output path to validate.
|
|
anchor_dir: If set, the resolved output must be a child of this
|
|
directory. Typically the parent directory of the input file so
|
|
outputs stay co-located with their sources.
|
|
|
|
Raises:
|
|
ValueError: For null bytes, missing parent, or sandbox escape.
|
|
"""
|
|
if '\x00' in output_path:
|
|
raise ValueError("Invalid output path: null byte detected")
|
|
|
|
resolved = Path(output_path).resolve()
|
|
|
|
if not resolved.parent.exists():
|
|
raise ValueError(f"Output directory does not exist: {resolved.parent}")
|
|
|
|
if anchor_dir is not None:
|
|
anchor = Path(anchor_dir).resolve()
|
|
try:
|
|
resolved.relative_to(anchor)
|
|
except ValueError:
|
|
raise ValueError(
|
|
f"Output path escapes allowed directory: "
|
|
f"{resolved} is not under {anchor}"
|
|
)
|
|
|
|
return str(resolved)
|
|
|
|
def _validate_directory(directory: str, *, allowed_root: str | None = None) -> str:
|
|
"""Validate a user-provided directory path against traversal and injection.
|
|
|
|
Resolves symlinks, blocks null bytes, and verifies the path is a real
|
|
directory. When *allowed_root* is given, the resolved path must be a
|
|
descendant of (or equal to) that root — preventing filesystem enumeration
|
|
beyond the project workspace.
|
|
|
|
Raises:
|
|
ValueError: For invalid paths (null bytes, not a directory, sandbox escape).
|
|
"""
|
|
if '\x00' in directory:
|
|
raise ValueError("Invalid directory path: null byte detected")
|
|
|
|
resolved = Path(directory).resolve()
|
|
|
|
if not resolved.is_dir():
|
|
raise ValueError(f"Not a valid directory: {directory}")
|
|
|
|
if allowed_root is not None:
|
|
root = Path(allowed_root).resolve()
|
|
try:
|
|
resolved.relative_to(root)
|
|
except ValueError:
|
|
raise ValueError(
|
|
f"Directory escapes allowed root: "
|
|
f"{resolved} is not under {root}"
|
|
)
|
|
|
|
return str(resolved)
|
|
|
|
def find_fcpxml_files(directory: str) -> list[str]:
|
|
"""Find all FCPXML files in a directory."""
|
|
path = Path(directory)
|
|
files = list(str(f) for f in path.rglob("*.fcpxml"))
|
|
files.extend(str(f) for f in path.rglob("*.fcpxmld"))
|
|
return sorted(files)
|
|
|
|
def generate_output_path(input_path: str, suffix: str = "_modified") -> str:
|
|
"""Generate output path from input path.
|
|
|
|
The suffix is sanitized to prevent path-component injection — only
|
|
alphanumeric, hyphen, underscore, and dot characters survive.
|
|
"""
|
|
# Strip anything that could inject path separators or traversal sequences
|
|
clean_suffix = re.sub(r'[^a-zA-Z0-9._-]', '', suffix)
|
|
if not clean_suffix:
|
|
clean_suffix = "_modified"
|
|
p = Path(input_path)
|
|
return str(p.parent / f"{p.stem}{clean_suffix}{p.suffix}")
|
|
|
|
def _resolve_io_paths(
|
|
arguments: dict,
|
|
suffix: str = "_modified",
|
|
) -> tuple[str, str]:
|
|
"""Validate input filepath and resolve the output path.
|
|
|
|
Shared foundation for every handler that reads an FCPXML and writes
|
|
a derived file. Validates the input, falls back to a suffixed
|
|
output name when ``output_path`` is not supplied, and sandbox-checks
|
|
the result.
|
|
|
|
Args:
|
|
arguments: Tool arguments dict (must contain ``filepath``; may
|
|
contain ``output_path``).
|
|
suffix: Default output filename suffix when ``output_path`` is
|
|
not provided (e.g. ``"_modified"``, ``"_beats"``).
|
|
|
|
Returns:
|
|
``(filepath, output_path)`` tuple with both paths validated.
|
|
"""
|
|
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
|
|
# Anchor write operations to the input file's directory so LLM-generated
|
|
# tool calls cannot write to arbitrary filesystem locations (e.g.
|
|
# /etc/cron.d/backdoor). When the explicit sandbox is off, the anchor
|
|
# still prevents writes outside the source directory tree.
|
|
# `output_dir` is where the caller wants the file written, not merely a
|
|
# sandbox boundary: the app's "Pasta do projeto" promises that everything
|
|
# generated lands there. Deriving the name from the input but keeping the
|
|
# input's directory made every cross-directory call fail its own anchor
|
|
# check ("output path escapes allowed directory"), so the setting silently
|
|
# only worked when it pointed at the directory the file was already going
|
|
# to. An explicit `output_path` still wins, and still has to sit inside
|
|
# the anchor.
|
|
output_dir = arguments.get("output_dir")
|
|
if output_dir:
|
|
anchor = _validate_directory(str(output_dir))
|
|
default_output = str(Path(anchor) / Path(generate_output_path(filepath, suffix)).name)
|
|
else:
|
|
anchor = str(Path(filepath).resolve().parent)
|
|
default_output = generate_output_path(filepath, suffix)
|
|
output_path = _validate_output_path(
|
|
arguments.get("output_path") or default_output,
|
|
anchor_dir=anchor,
|
|
)
|
|
return filepath, output_path
|