Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
157 lines
6.0 KiB
Python
157 lines
6.0 KiB
Python
"""Base comum dos comandos da ponte: saída JSON, caminhos derivados e cache.
|
|
|
|
A saída passa toda por `emit`. Os módulos de comando chamam `shared.emit(...)`
|
|
pelo módulo, e não pelo nome importado, de propósito: assim trocar `emit` num
|
|
lugar só — como a suíte faz para capturar a saída — continua alcançando todos
|
|
os comandos, o que deixaria de valer se cada um tivesse ligado o nome no seu
|
|
próprio import.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import sys
|
|
import threading
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
from fcpxml.diarize import build_speakers
|
|
from fcpxml.media_intel import media_src_to_path
|
|
from fcpxml.parser import parse_fcpxml
|
|
|
|
RECOMMENDED = ("large-v3", "distil-large-v3", "small", "base")
|
|
|
|
def _derived_output(path: str, suffix: str, args: dict) -> str:
|
|
"""Resolve a derived XML path, optionally inside the chosen output folder."""
|
|
output_dir = str(args.get("output_dir", "")).strip()
|
|
if output_dir:
|
|
directory = Path(output_dir).expanduser()
|
|
directory.mkdir(parents=True, exist_ok=True)
|
|
source = Path(path)
|
|
extension = ".fcpxmld" if source.is_dir() else source.suffix
|
|
return str(directory / f"{source.stem}{suffix}{extension}")
|
|
from server import generate_output_path
|
|
return generate_output_path(path, suffix)
|
|
|
|
def _is_no_change_message(message: str) -> bool:
|
|
"""Whether a tool completed cleanly without needing to save a new file."""
|
|
text = message.lower()
|
|
return any(
|
|
token in text
|
|
for token in (
|
|
"no cuts to make",
|
|
"no silence",
|
|
"file unchanged",
|
|
"nothing saved",
|
|
)
|
|
)
|
|
|
|
def _emit_no_change_or_error(path: str, message: str) -> int:
|
|
if _is_no_change_message(message):
|
|
emit({"ok": True, "path": path, "unchanged": True, "message": message})
|
|
return 0
|
|
emit({"ok": False, "error": message})
|
|
return 1
|
|
|
|
|
|
# Serializa a escrita em stdout. A ponte é JSON-lines: dois comandos
|
|
# escrevendo ao mesmo tempo entrelaçariam documentos e o app leria lixo.
|
|
_OUT_LOCK = threading.Lock()
|
|
|
|
def emit(obj: Any) -> None:
|
|
sys.stdout.write(json.dumps(obj, ensure_ascii=False) + "\n")
|
|
sys.stdout.flush()
|
|
|
|
def _transcript_json_path(media_path: str, output_dir: str = "") -> Path:
|
|
"""Where the ``_transcript.json`` for ``media_path`` lives.
|
|
|
|
When ``output_dir`` (the user-selected project folder) is set, the
|
|
transcript is saved/read there — never next to the source media, which
|
|
may sit on a read-only volume or a Final Cut Library the user never
|
|
browses. Falls back to the media's own folder only when no project
|
|
folder has been chosen (legacy/MCP callers).
|
|
"""
|
|
p = Path(media_path)
|
|
if output_dir:
|
|
directory = Path(output_dir).expanduser()
|
|
directory.mkdir(parents=True, exist_ok=True)
|
|
return directory / f"{p.stem}_transcript.json"
|
|
return p.with_name(p.stem + "_transcript.json")
|
|
|
|
def _save_json_atomic(path: Path, data: Any) -> None:
|
|
"""Write ``data`` to ``path`` atomically and validate the result on disk.
|
|
|
|
Mirrors the reference WHISPERX save path: write a ``.tmp``, ``os.replace``
|
|
into place, then confirm the file exists, is non-empty, and parses as JSON.
|
|
"""
|
|
tmp_path = str(path) + ".tmp"
|
|
with open(tmp_path, "w", encoding="utf-8") as fh:
|
|
json.dump(data, fh, ensure_ascii=False, indent=2)
|
|
os.replace(tmp_path, path)
|
|
if not path.exists() or os.path.getsize(path) == 0:
|
|
raise RuntimeError("O arquivo salvo está vazio ou não foi encontrado.")
|
|
with open(path, encoding="utf-8") as fh:
|
|
json.load(fh)
|
|
|
|
def _project_media_paths(path: str) -> list[str]:
|
|
proj = parse_fcpxml(path)
|
|
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
|
|
media_paths: list[str] = []
|
|
if tl is not None:
|
|
for clip in getattr(tl, "clips", []):
|
|
mp = media_src_to_path(clip.media_path or "")
|
|
if mp and Path(mp).is_file() and mp not in media_paths:
|
|
media_paths.append(mp)
|
|
return media_paths
|
|
|
|
def _project_media_rotations(path: str) -> dict[str, float]:
|
|
"""Degrees each source media was rotated by via a Transform filter on its
|
|
clip in the FCPXML — keyed by the same resolved media path
|
|
``_project_media_paths`` returns, so the two can be joined by media_path."""
|
|
proj = parse_fcpxml(path)
|
|
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
|
|
rotations: dict[str, float] = {}
|
|
if tl is not None:
|
|
for clip in getattr(tl, "clips", []):
|
|
mp = media_src_to_path(clip.media_path or "")
|
|
if mp and clip.rotation:
|
|
rotations[mp] = clip.rotation
|
|
return rotations
|
|
|
|
def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path:
|
|
p = Path(media_path)
|
|
if output_dir:
|
|
directory = Path(output_dir).expanduser()
|
|
directory.mkdir(parents=True, exist_ok=True)
|
|
return directory / f"{p.stem}_voice_timeline.json"
|
|
return p.with_name(p.stem + "_voice_timeline.json")
|
|
|
|
def _load_cached_voice_timeline(json_path: Path, media_path: str) -> dict | None:
|
|
try:
|
|
with open(json_path, encoding="utf-8") as fh:
|
|
data = json.load(fh)
|
|
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
|
return None
|
|
if not isinstance(data, dict):
|
|
return None
|
|
if data.get("source") != Path(media_path).name:
|
|
return None
|
|
if not isinstance(data.get("segments"), list):
|
|
return None
|
|
return data
|
|
|
|
def _load_cached_transcript(json_path: Path) -> dict | None:
|
|
"""Return a valid cached transcript dict, or ``None`` if absent/unreadable."""
|
|
if not json_path.is_file():
|
|
return None
|
|
try:
|
|
data = json.loads(json_path.read_text(encoding="utf-8"))
|
|
except (OSError, ValueError):
|
|
return None
|
|
if isinstance(data, dict) and isinstance(data.get("words"), list):
|
|
if "speakers" not in data:
|
|
data["speakers"] = build_speakers(data.get("segments", []))
|
|
return data
|
|
return None
|