Ao dividir _shared.py em admin/api/*.py ontem, o cálculo `Path(__file__).resolve().parent.parent / "code"` foi copiado sem ajustar para o nível de diretório novo. No arquivo original (admin/models_api.py, direto em admin/) dois `.parent` chegavam na raiz do repo. Em admin/api/shared.py, um nível mais fundo, dois `.parent` param em admin/ — e admin/code nunca existiu. sys.path nunca recebia code/, então toda ação que passa por `server` (analisar voz, aplicar decisões) crashava o app com ModuleNotFoundError: server_tools. O bug sobreviveu a duas rodadas de validação da sessão anterior — lint zero, 1454 testes verdes, comando testado manualmente pela ponte — porque todos rodam num venv com install editável (__editable__.fcp_mcp_server.pth) que já deixa fcpxml/server_tools importáveis por conta própria, mascarando qualquer erro no cálculo manual de sys.path. Só o app real, no fallback sem uv, expõe o bug. Correção: o cálculo de sys.path sai de cada módulo de comando (estava duplicado em nove arquivos) e passa a existir uma única vez em admin/api/__init__.py, que roda antes de qualquer submódulo — nenhum precisa mais da própria cópia. O teste de regressão precisou de duas tentativas pelo mesmo motivo do bug: a primeira versão também passava com o bug presente, por rodar no mesmo venv "de sorte". Só ficou confiável isolando um subprocess que remove site-packages do sys.path antes de importar — confirmado nos dois sentidos, falha com o bug reintroduzido e passa com a correção (TestCodeDirResolution). Detalhe completo, incluindo por que o comando manual não pegou: Engine/docs/05_EXPERIENCIAS.md #25. Lint zerado, 1457 testes passando (3 novos), app compilado. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
143 lines
5.3 KiB
Python
143 lines
5.3 KiB
Python
"""Base comum dos comandos da ponte: saída JSON, caminhos derivados e cache.
|
|
|
|
A saída passa toda por `emit`. Os módulos de comando chamam `shared.emit(...)`
|
|
pelo módulo, e não pelo nome importado, de propósito: assim trocar `emit` num
|
|
lugar só — como a suíte faz para capturar a saída — continua alcançando todos
|
|
os comandos, o que deixaria de valer se cada um tivesse ligado o nome no seu
|
|
próprio import.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import sys
|
|
import threading
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from fcpxml.diarize import build_speakers
|
|
from fcpxml.media_intel import media_src_to_path
|
|
from fcpxml.parser import parse_fcpxml
|
|
|
|
RECOMMENDED = ("large-v3", "distil-large-v3", "small", "base")
|
|
|
|
def _derived_output(path: str, suffix: str, args: dict) -> str:
|
|
"""Resolve a derived XML path, optionally inside the chosen output folder."""
|
|
output_dir = str(args.get("output_dir", "")).strip()
|
|
if output_dir:
|
|
directory = Path(output_dir).expanduser()
|
|
directory.mkdir(parents=True, exist_ok=True)
|
|
source = Path(path)
|
|
extension = ".fcpxmld" if source.is_dir() else source.suffix
|
|
return str(directory / f"{source.stem}{suffix}{extension}")
|
|
from server import generate_output_path
|
|
return generate_output_path(path, suffix)
|
|
|
|
def _is_no_change_message(message: str) -> bool:
|
|
"""Whether a tool completed cleanly without needing to save a new file."""
|
|
text = message.lower()
|
|
return any(
|
|
token in text
|
|
for token in (
|
|
"no cuts to make",
|
|
"no silence",
|
|
"file unchanged",
|
|
"nothing saved",
|
|
)
|
|
)
|
|
|
|
def _emit_no_change_or_error(path: str, message: str) -> int:
|
|
if _is_no_change_message(message):
|
|
emit({"ok": True, "path": path, "unchanged": True, "message": message})
|
|
return 0
|
|
emit({"ok": False, "error": message})
|
|
return 1
|
|
|
|
|
|
# Serializa a escrita em stdout. A ponte é JSON-lines: dois comandos
|
|
# escrevendo ao mesmo tempo entrelaçariam documentos e o app leria lixo.
|
|
_OUT_LOCK = threading.Lock()
|
|
|
|
def emit(obj: Any) -> None:
|
|
sys.stdout.write(json.dumps(obj, ensure_ascii=False) + "\n")
|
|
sys.stdout.flush()
|
|
|
|
def _transcript_json_path(media_path: str, output_dir: str = "") -> Path:
|
|
"""Where the ``_transcript.json`` for ``media_path`` lives.
|
|
|
|
When ``output_dir`` (the user-selected project folder) is set, the
|
|
transcript is saved/read there — never next to the source media, which
|
|
may sit on a read-only volume or a Final Cut Library the user never
|
|
browses. Falls back to the media's own folder only when no project
|
|
folder has been chosen (legacy/MCP callers).
|
|
"""
|
|
p = Path(media_path)
|
|
if output_dir:
|
|
directory = Path(output_dir).expanduser()
|
|
directory.mkdir(parents=True, exist_ok=True)
|
|
return directory / f"{p.stem}_transcript.json"
|
|
return p.with_name(p.stem + "_transcript.json")
|
|
|
|
def _save_json_atomic(path: Path, data: Any) -> None:
|
|
"""Write ``data`` to ``path`` atomically and validate the result on disk.
|
|
|
|
Mirrors the reference WHISPERX save path: write a ``.tmp``, ``os.replace``
|
|
into place, then confirm the file exists, is non-empty, and parses as JSON.
|
|
"""
|
|
tmp_path = str(path) + ".tmp"
|
|
with open(tmp_path, "w", encoding="utf-8") as fh:
|
|
json.dump(data, fh, ensure_ascii=False, indent=2)
|
|
os.replace(tmp_path, path)
|
|
if not path.exists() or os.path.getsize(path) == 0:
|
|
raise RuntimeError("O arquivo salvo está vazio ou não foi encontrado.")
|
|
with open(path, encoding="utf-8") as fh:
|
|
json.load(fh)
|
|
|
|
def _project_media_paths(path: str) -> list[str]:
|
|
proj = parse_fcpxml(path)
|
|
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
|
|
media_paths: list[str] = []
|
|
if tl is not None:
|
|
for clip in getattr(tl, "clips", []):
|
|
mp = media_src_to_path(clip.media_path or "")
|
|
if mp and Path(mp).is_file() and mp not in media_paths:
|
|
media_paths.append(mp)
|
|
return media_paths
|
|
|
|
def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path:
|
|
p = Path(media_path)
|
|
if output_dir:
|
|
directory = Path(output_dir).expanduser()
|
|
directory.mkdir(parents=True, exist_ok=True)
|
|
return directory / f"{p.stem}_voice_timeline.json"
|
|
return p.with_name(p.stem + "_voice_timeline.json")
|
|
|
|
def _load_cached_voice_timeline(json_path: Path, media_path: str) -> dict | None:
|
|
try:
|
|
with open(json_path, encoding="utf-8") as fh:
|
|
data = json.load(fh)
|
|
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
|
return None
|
|
if not isinstance(data, dict):
|
|
return None
|
|
if data.get("source") != Path(media_path).name:
|
|
return None
|
|
if not isinstance(data.get("segments"), list):
|
|
return None
|
|
return data
|
|
|
|
def _load_cached_transcript(json_path: Path) -> dict | None:
|
|
"""Return a valid cached transcript dict, or ``None`` if absent/unreadable."""
|
|
if not json_path.is_file():
|
|
return None
|
|
try:
|
|
data = json.loads(json_path.read_text(encoding="utf-8"))
|
|
except (OSError, ValueError):
|
|
return None
|
|
if isinstance(data, dict) and isinstance(data.get("words"), list):
|
|
if "speakers" not in data:
|
|
data["speakers"] = build_speakers(data.get("segments", []))
|
|
return data
|
|
return None
|