Files
gart/admin/api/shared.py
T
João HenriqueandClaude Sonnet 5 7b5aed79ee feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-21 18:26:04 -04:00

157 lines
6.0 KiB
Python

"""Base comum dos comandos da ponte: saída JSON, caminhos derivados e cache.
A saída passa toda por `emit`. Os módulos de comando chamam `shared.emit(...)`
pelo módulo, e não pelo nome importado, de propósito: assim trocar `emit` num
lugar só — como a suíte faz para capturar a saída — continua alcançando todos
os comandos, o que deixaria de valer se cada um tivesse ligado o nome no seu
próprio import.
"""
from __future__ import annotations
import json
import os
import sys
import threading
from pathlib import Path
from typing import Any
from fcpxml.diarize import build_speakers
from fcpxml.media_intel import media_src_to_path
from fcpxml.parser import parse_fcpxml
RECOMMENDED = ("large-v3", "distil-large-v3", "small", "base")
def _derived_output(path: str, suffix: str, args: dict) -> str:
"""Resolve a derived XML path, optionally inside the chosen output folder."""
output_dir = str(args.get("output_dir", "")).strip()
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
source = Path(path)
extension = ".fcpxmld" if source.is_dir() else source.suffix
return str(directory / f"{source.stem}{suffix}{extension}")
from server import generate_output_path
return generate_output_path(path, suffix)
def _is_no_change_message(message: str) -> bool:
"""Whether a tool completed cleanly without needing to save a new file."""
text = message.lower()
return any(
token in text
for token in (
"no cuts to make",
"no silence",
"file unchanged",
"nothing saved",
)
)
def _emit_no_change_or_error(path: str, message: str) -> int:
if _is_no_change_message(message):
emit({"ok": True, "path": path, "unchanged": True, "message": message})
return 0
emit({"ok": False, "error": message})
return 1
# Serializa a escrita em stdout. A ponte é JSON-lines: dois comandos
# escrevendo ao mesmo tempo entrelaçariam documentos e o app leria lixo.
_OUT_LOCK = threading.Lock()
def emit(obj: Any) -> None:
sys.stdout.write(json.dumps(obj, ensure_ascii=False) + "\n")
sys.stdout.flush()
def _transcript_json_path(media_path: str, output_dir: str = "") -> Path:
"""Where the ``_transcript.json`` for ``media_path`` lives.
When ``output_dir`` (the user-selected project folder) is set, the
transcript is saved/read there — never next to the source media, which
may sit on a read-only volume or a Final Cut Library the user never
browses. Falls back to the media's own folder only when no project
folder has been chosen (legacy/MCP callers).
"""
p = Path(media_path)
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
return directory / f"{p.stem}_transcript.json"
return p.with_name(p.stem + "_transcript.json")
def _save_json_atomic(path: Path, data: Any) -> None:
"""Write ``data`` to ``path`` atomically and validate the result on disk.
Mirrors the reference WHISPERX save path: write a ``.tmp``, ``os.replace``
into place, then confirm the file exists, is non-empty, and parses as JSON.
"""
tmp_path = str(path) + ".tmp"
with open(tmp_path, "w", encoding="utf-8") as fh:
json.dump(data, fh, ensure_ascii=False, indent=2)
os.replace(tmp_path, path)
if not path.exists() or os.path.getsize(path) == 0:
raise RuntimeError("O arquivo salvo está vazio ou não foi encontrado.")
with open(path, encoding="utf-8") as fh:
json.load(fh)
def _project_media_paths(path: str) -> list[str]:
proj = parse_fcpxml(path)
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
media_paths: list[str] = []
if tl is not None:
for clip in getattr(tl, "clips", []):
mp = media_src_to_path(clip.media_path or "")
if mp and Path(mp).is_file() and mp not in media_paths:
media_paths.append(mp)
return media_paths
def _project_media_rotations(path: str) -> dict[str, float]:
"""Degrees each source media was rotated by via a Transform filter on its
clip in the FCPXML — keyed by the same resolved media path
``_project_media_paths`` returns, so the two can be joined by media_path."""
proj = parse_fcpxml(path)
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
rotations: dict[str, float] = {}
if tl is not None:
for clip in getattr(tl, "clips", []):
mp = media_src_to_path(clip.media_path or "")
if mp and clip.rotation:
rotations[mp] = clip.rotation
return rotations
def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path:
p = Path(media_path)
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
return directory / f"{p.stem}_voice_timeline.json"
return p.with_name(p.stem + "_voice_timeline.json")
def _load_cached_voice_timeline(json_path: Path, media_path: str) -> dict | None:
try:
with open(json_path, encoding="utf-8") as fh:
data = json.load(fh)
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
return None
if not isinstance(data, dict):
return None
if data.get("source") != Path(media_path).name:
return None
if not isinstance(data.get("segments"), list):
return None
return data
def _load_cached_transcript(json_path: Path) -> dict | None:
"""Return a valid cached transcript dict, or ``None`` if absent/unreadable."""
if not json_path.is_file():
return None
try:
data = json.loads(json_path.read_text(encoding="utf-8"))
except (OSError, ValueError):
return None
if isinstance(data, dict) and isinstance(data.get("words"), list):
if "speakers" not in data:
data["speakers"] = build_speakers(data.get("segments", []))
return data
return None