Files
gart/code/server_tools/_shared/paths.py
João HenriqueandClaude Opus 5 ffaebb3f72 refactor: _shared.py vira subpacote, um módulo por papel
Eram 882 linhas de seis papéis sem relação, sob um nome que só dizia
"compartilhado" — o depósito onde tudo que servia a mais de um handler
acabava caindo.

    media       316   transcrição em cache, corte por fala, relatório
    paths       206   sandbox, limites, caminho de saída
    project     116   abrir projeto, preparar modifier/generator
    captions    112   SRT, VTT, listas com timestamp
    detection    99   flash frames, buracos, duplicados
    formatting   86   tabelas e relatórios dos handlers

O __init__ reexporta os 46 nomes, então os treze pontos que importam daqui
não mudaram.

_transcript_cut_report saiu de formatting para media: ele precisa do hint de
instalação e do _text_result, ou seja, é relatório de transcrição e não
formatação genérica — mover foi mais honesto que cruzar imports entre os
dois módulos.

Quatro testes patchavam `server_tools._shared.transcribe`; o nome agora é
ligado por _shared/media.py, então o patch passou a apontar para lá — mesmo
padrão da experiência #23.

Lint zerado, 1454 testes passando.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-19 22:35:02 -04:00

227 lines
8.6 KiB
Python

"""Caminhos: validação contra a sandbox, limites de tamanho, saída derivada.
Extraído de _shared.py — ver server_tools/_shared/__init__.py.
"""
from __future__ import annotations
import os
import re
from pathlib import Path
PROJECTS_DIR = os.environ.get("FCP_PROJECTS_DIR", os.path.expanduser("~/Movies"))
_SANDBOX_ENABLED = "FCP_PROJECTS_DIR" in os.environ
MAX_FILE_SIZE = 100 * 1024 * 1024
MAX_MEDIA_FILE_SIZE = 32 * 1024 * 1024 * 1024
_MAX_JSON_DEPTH = 50
def _check_json_depth(obj: object, _depth: int = 0) -> None:
"""Reject JSON structures nested beyond _MAX_JSON_DEPTH.
Prevents denial-of-service via deeply nested objects that exhaust the
call stack or memory during downstream processing. Called after
json.load() since Python's json module has no built-in depth limit.
"""
if _depth > _MAX_JSON_DEPTH:
raise ValueError(
f"JSON nesting depth exceeds {_MAX_JSON_DEPTH} — "
"file may be malformed or adversarial"
)
if isinstance(obj, dict):
for v in obj.values():
_check_json_depth(v, _depth + 1)
elif isinstance(obj, list):
for item in obj:
_check_json_depth(item, _depth + 1)
def _validate_filepath(
filepath: str,
allowed_extensions: tuple[str, ...] | None = None,
max_size: int = MAX_FILE_SIZE,
) -> str:
"""Validate a user-provided file path against traversal and size attacks.
Resolves symlinks, blocks null bytes, enforces extension whitelist, and
checks file size before any parsing takes place.
``max_size`` defaults to the document limit; callers handling source
media pass ``MAX_MEDIA_FILE_SIZE``, since media is streamed rather than
parsed into memory (see the constant for why).
Raises:
ValueError: For invalid paths (null bytes, bad extensions, oversized).
FileNotFoundError: When the resolved path does not exist.
"""
if '\x00' in filepath:
raise ValueError("Invalid file path: null byte detected")
resolved = Path(filepath).resolve()
if not resolved.exists():
raise FileNotFoundError(f"File not found: {filepath}")
# .fcpxmld bundles are directories (a package wrapping Info.fcpxml plus
# sidecar data files for object tracking / Cinematic mode). The size
# check applies to the inner Info.fcpxml, which is what gets parsed.
if resolved.is_dir():
if resolved.suffix.lower() != '.fcpxmld':
raise ValueError(f"Not a regular file: {filepath}")
inner = resolved / 'Info.fcpxml'
if not inner.is_file():
raise ValueError(f"Invalid bundle (no Info.fcpxml): {filepath}")
size_target = inner
elif not resolved.is_file():
raise ValueError(f"Not a regular file: {filepath}")
else:
size_target = resolved
if allowed_extensions and resolved.suffix.lower() not in allowed_extensions:
raise ValueError(
f"Invalid file type '{resolved.suffix}'. "
f"Allowed: {', '.join(allowed_extensions)}"
)
if size_target.stat().st_size > max_size:
size_mb = size_target.stat().st_size / (1024 * 1024)
raise ValueError(f"File too large ({size_mb:.1f} MB). Maximum: {max_size // (1024 * 1024)} MB")
return str(resolved)
def _validate_output_path(output_path: str, *, anchor_dir: str | None = None) -> str:
"""Validate an output path with optional sandbox enforcement.
Resolves traversal, blocks null bytes, ensures parent exists, and — when
*anchor_dir* is provided — verifies the resolved output lives under that
directory. This prevents LLM-generated tool calls from writing to
arbitrary filesystem locations (e.g. ``/etc/cron.d/backdoor``).
Args:
output_path: The raw output path to validate.
anchor_dir: If set, the resolved output must be a child of this
directory. Typically the parent directory of the input file so
outputs stay co-located with their sources.
Raises:
ValueError: For null bytes, missing parent, or sandbox escape.
"""
if '\x00' in output_path:
raise ValueError("Invalid output path: null byte detected")
resolved = Path(output_path).resolve()
if not resolved.parent.exists():
raise ValueError(f"Output directory does not exist: {resolved.parent}")
if anchor_dir is not None:
anchor = Path(anchor_dir).resolve()
try:
resolved.relative_to(anchor)
except ValueError:
raise ValueError(
f"Output path escapes allowed directory: "
f"{resolved} is not under {anchor}"
)
return str(resolved)
def _validate_directory(directory: str, *, allowed_root: str | None = None) -> str:
"""Validate a user-provided directory path against traversal and injection.
Resolves symlinks, blocks null bytes, and verifies the path is a real
directory. When *allowed_root* is given, the resolved path must be a
descendant of (or equal to) that root — preventing filesystem enumeration
beyond the project workspace.
Raises:
ValueError: For invalid paths (null bytes, not a directory, sandbox escape).
"""
if '\x00' in directory:
raise ValueError("Invalid directory path: null byte detected")
resolved = Path(directory).resolve()
if not resolved.is_dir():
raise ValueError(f"Not a valid directory: {directory}")
if allowed_root is not None:
root = Path(allowed_root).resolve()
try:
resolved.relative_to(root)
except ValueError:
raise ValueError(
f"Directory escapes allowed root: "
f"{resolved} is not under {root}"
)
return str(resolved)
def find_fcpxml_files(directory: str) -> list[str]:
"""Find all FCPXML files in a directory."""
path = Path(directory)
files = list(str(f) for f in path.rglob("*.fcpxml"))
files.extend(str(f) for f in path.rglob("*.fcpxmld"))
return sorted(files)
def generate_output_path(input_path: str, suffix: str = "_modified") -> str:
"""Generate output path from input path.
The suffix is sanitized to prevent path-component injection — only
alphanumeric, hyphen, underscore, and dot characters survive.
"""
# Strip anything that could inject path separators or traversal sequences
clean_suffix = re.sub(r'[^a-zA-Z0-9._-]', '', suffix)
if not clean_suffix:
clean_suffix = "_modified"
p = Path(input_path)
return str(p.parent / f"{p.stem}{clean_suffix}{p.suffix}")
def _resolve_io_paths(
arguments: dict,
suffix: str = "_modified",
) -> tuple[str, str]:
"""Validate input filepath and resolve the output path.
Shared foundation for every handler that reads an FCPXML and writes
a derived file. Validates the input, falls back to a suffixed
output name when ``output_path`` is not supplied, and sandbox-checks
the result.
Args:
arguments: Tool arguments dict (must contain ``filepath``; may
contain ``output_path``).
suffix: Default output filename suffix when ``output_path`` is
not provided (e.g. ``"_modified"``, ``"_beats"``).
Returns:
``(filepath, output_path)`` tuple with both paths validated.
"""
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
# Anchor write operations to the input file's directory so LLM-generated
# tool calls cannot write to arbitrary filesystem locations (e.g.
# /etc/cron.d/backdoor). When the explicit sandbox is off, the anchor
# still prevents writes outside the source directory tree.
# `output_dir` is where the caller wants the file written, not merely a
# sandbox boundary: the app's "Pasta do projeto" promises that everything
# generated lands there. Deriving the name from the input but keeping the
# input's directory made every cross-directory call fail its own anchor
# check ("output path escapes allowed directory"), so the setting silently
# only worked when it pointed at the directory the file was already going
# to. An explicit `output_path` still wins, and still has to sit inside
# the anchor.
output_dir = arguments.get("output_dir")
if output_dir:
anchor = _validate_directory(str(output_dir))
default_output = str(Path(anchor) / Path(generate_output_path(filepath, suffix)).name)
else:
anchor = str(Path(filepath).resolve().parent)
default_output = generate_output_path(filepath, suffix)
output_path = _validate_output_path(
arguments.get("output_path") or default_output,
anchor_dir=anchor,
)
return filepath, output_path