Files
gart/admin/models_api.py

1147 lines
44 KiB
Python

#!/usr/bin/env python3
"""JSON bridge between the SwiftUI app and the fcp-mcp-server Python engine.
The SwiftUI app (MacApp/) launches this script as a subprocess with a command
and optional JSON arguments, then reads a single JSON document (or
newline-delimited JSON for progress) on stdout.
Commands:
catalog
-> {"models": [{display_name, internal_name, size, storage,
accuracy, speed}], "installed": [names],
"selected": name, "models_dir": path, "installed_count": n,
"recommended": [names]}
download {"model": "small"}
-> JSON-lines: {"type":"progress","fraction":0.42}
{"type":"done","installed":true}
{"type":"error","message":"..."}
cancel {"model": "small"}
-> {"ok": true}
select {"model": "small"}
-> {"ok": true, "selected": "small"}
set_language {"language": "pt"} | "auto"
-> {"ok": true, "language": "pt"}
delete {"model": "small"}
-> {"ok": true}
open_finder {"model": "small"}
-> {"ok": true}
set_models_dir {"dir": "/path"}
-> {"ok": true, "models_dir": "/path"}
inspect {"path": "/path/to/project.fcpxml"}
-> {"ok": true, "path": "...", "name": "...", "fcpxml_version": "1.13",
"timelines": [{name, duration_seconds, frame_rate, width, height,
clips, cuts, connected, markers}]}
or {"ok": false, "error": "..."}
analyze_voice {"path": "...", "output_dir": "...", "model": "...",
"language": "pt"|"auto"|null, "hf_token": "..."|null,
"num_speakers": ""|null}
Build the voice timeline (transcript+diarization+acoustics) for
every unique source media — analysis only, writes _voice_timeline.json
next to each media, `path` passes through unchanged. Meant as one
entry in the batch operations list (see processBatchStep), so
`refine_voice_timeline` never has to reopen the audio later.
-> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."}
dynamic_subtitle_config {}
-> {"ok": true, "band_height", "block_center_y", "line_gap", "font",
"font_size", "emphasis_font", "emphasis_face", "emphasis_size",
"active_color", "emphasis_color", "text_scale"}
set_dynamic_subtitle_config {<any of the fields above>}
Persists only the given fields to ~/.fcp-mcp-server/config.json.
generate_dynamic_subtitles reads this as its own fallback default.
-> {"ok": true, <same shape as dynamic_subtitle_config>}
silence_config {}
-> {"ok": true, "noise_db": -30.0, "min_silence": 0.5, "padding": 0.05}
set_silence_config {"noise_db": -30.0, "min_silence": 0.5, "padding": 0.05}
Persists only the given fields. detect_media_silence and
remove_media_silence read this as their own fallback default.
-> {"ok": true, <same shape as silence_config>}
transcribe {"path": "...", "model": "small", "language": "pt"|null,
"hf_token": "..."|null, "num_speakers": ""|null}
-> JSON-lines:
{"type":"progress","fraction":0.5,"stage":"Transcrevendo..."}
{"type":"result","transcripts":[{"media","language","words",
"duration","preview","saved",
"speakers"}]}
{"type":"error","message":"..."}
edit_by_transcript {"path": "...", "phrases": ["frase um", "frase dois"],
"mode": "remove"|"keep_only", "clip_name": "..."|null,
"padding": 0.0, "model": "small", "language": "pt"|null}
-> {"ok": true, "path": "..._transcript_edit.fcpxml", "message": "..."}
or {"ok": false, "error": "..."}
remove_filler_words {"path": "...", "fillers": ["um","uh"]|null,
"clip_name": "..."|null, "padding": 0.02,
"model": "small", "language": "pt"|null}
-> {"ok": true, "path": "..._defillered.fcpxml", "message": "..."}
or {"ok": false, "error": "..."}
transcript_markers {"path": "...", "clip_name": "..."|null,
"marker_type": "chapter", "max_label_length": 50,
"model": "small", "language": "pt"|null}
-> {"ok": true, "path": "..._transcript_markers.fcpxml", "message": "..."}
or {"ok": false, "error": "..."}
add_zoom {"path": "...", "clip_id": "...", "start": 10.0, "end": 16.0,
"scale": 1.3, "ease": 0.3, "position": "0 0"|null}
-> {"ok": true, "path": "..._zoom.fcpxml", "message": "..."}
or {"ok": false, "error": "..."}
generate_dynamic_subtitles {"path": "...", "clip_name": "..."|null,
"band_height": 0.22, "block_center_y": -167,
"font": "Helvetica Neue", "font_size": 128,
"emphasis_font": "Playfair Display",
"emphasis_face": "Medium Italic", "emphasis_size": 265,
"active_color": "1 1 1 1", "emphasis_color": "1 1 1 1",
"model": "small", "language": "pt"|null}
-> {"ok": true, "path": "..._dynamic_subtitles.fcpxml", "message": "..."}
or {"ok": false, "error": "..."}
rename_speakers {"path": "/to/media_transcript.json",
"speakers": {"SPEAKER_01": "Nome"}}
-> {"ok": true, "speakers": [...]}
set_diarization {"token": "hf_...", "num_speakers": ""}
-> {"ok": true, "diarization": bool, "diarization_message": "...",
"num_speakers": "..."}
voice_analysis
-> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85,
"emphasis_weights": {...}, "emotion_enabled": false,
"emotion_sensitivity": 0.5}
set_voice_analysis {"energy_threshold": 0.6, "emphasis_threshold": 0.9,
"emphasis_weights": {"energy": 0.4}|null,
"emotion_enabled": true, "emotion_sensitivity": 0.5}
-> same shape as voice_analysis (only given fields change)
Exit code 0 on success, 1 on error.
"""
from __future__ import annotations
import asyncio
import json
import os
import shutil
import subprocess
import sys
import threading
from pathlib import Path
from typing import Any
# code/ is the package root for fcpxml and server modules.
_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code")
if _CODE_DIR not in sys.path:
sys.path.insert(0, _CODE_DIR)
from fcpxml.diarize import ( # noqa: E402
assign_speakers,
build_speakers,
diarization_capability,
diarize,
)
from fcpxml.media_intel import media_src_to_path # noqa: E402
from fcpxml.model_manager import ( # noqa: E402
download_model,
get_models_dir,
is_model_downloaded,
list_installed_models,
load_catalog,
load_dynamic_subtitle_config,
load_hf_token,
load_num_speakers,
load_project_config,
load_selected_model,
load_silence_config,
load_transcript_language,
load_voice_analysis_config,
model_cache_dir,
save_dynamic_subtitle_config,
save_hf_token,
save_models_dir,
save_num_speakers,
save_project_config,
save_selected_model,
save_silence_config,
save_transcript_language,
save_voice_analysis_config,
)
from fcpxml.parser import parse_fcpxml # noqa: E402
from fcpxml.transcribe import transcribe # noqa: E402
from fcpxml.writer import FCPXMLModifier # noqa: E402
RECOMMENDED = ("large-v3", "distil-large-v3", "small", "base")
def _derived_output(path: str, suffix: str, args: dict) -> str:
"""Resolve a derived XML path, optionally inside the chosen output folder."""
output_dir = str(args.get("output_dir", "")).strip()
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
source = Path(path)
extension = ".fcpxmld" if source.is_dir() else source.suffix
return str(directory / f"{source.stem}{suffix}{extension}")
from server import generate_output_path
return generate_output_path(path, suffix)
# Download cancellation events, keyed by model name.
_CANCEL: dict[str, threading.Event] = {}
_LOCK = threading.Lock()
def _emit(obj: Any) -> None:
sys.stdout.write(json.dumps(obj, ensure_ascii=False) + "\n")
sys.stdout.flush()
def _transcript_json_path(media_path: str, output_dir: str = "") -> Path:
"""Where the ``_transcript.json`` for ``media_path`` lives.
When ``output_dir`` (the user-selected project folder) is set, the
transcript is saved/read there — never next to the source media, which
may sit on a read-only volume or a Final Cut Library the user never
browses. Falls back to the media's own folder only when no project
folder has been chosen (legacy/MCP callers).
"""
p = Path(media_path)
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
return directory / f"{p.stem}_transcript.json"
return p.with_name(p.stem + "_transcript.json")
def _save_json_atomic(path: Path, data: Any) -> None:
"""Write ``data`` to ``path`` atomically and validate the result on disk.
Mirrors the reference WHISPERX save path: write a ``.tmp``, ``os.replace``
into place, then confirm the file exists, is non-empty, and parses as JSON.
"""
tmp_path = str(path) + ".tmp"
with open(tmp_path, "w", encoding="utf-8") as fh:
json.dump(data, fh, ensure_ascii=False, indent=2)
os.replace(tmp_path, path)
if not path.exists() or os.path.getsize(path) == 0:
raise RuntimeError("O arquivo salvo está vazio ou não foi encontrado.")
with open(path, encoding="utf-8") as fh:
json.load(fh)
# ── commands ────────────────────────────────────────────────────────────────
def cmd_catalog() -> None:
catalog = load_catalog()
installed = list_installed_models()
diar_ok, diar_msg = diarization_capability(load_hf_token())
_emit(
{
"models": catalog,
"installed": installed,
"selected": load_selected_model(),
"language": load_transcript_language(),
"models_dir": str(get_models_dir()),
"installed_count": len(installed),
"recommended": list(RECOMMENDED),
"diarization": diar_ok,
"diarization_message": diar_msg,
"hf_token_set": bool(load_hf_token()),
"num_speakers": load_num_speakers(),
}
)
def cmd_download(args: dict) -> int:
model = str(args.get("model", ""))
if model not in _model_names():
_emit({"type": "error", "message": f"Modelo desconhecido: {model}"})
return 1
ev = threading.Event()
with _LOCK:
_CANCEL[model] = ev
try:
download_model(model, progress_cb=lambda f: _emit({"type": "progress", "fraction": f}), cancel_event=ev)
installed = is_model_downloaded(model)
_emit({"type": "done", "installed": installed})
if installed:
save_selected_model(model)
return 0 if installed else 1
except Exception as exc:
_emit({"type": "error", "message": str(exc)})
return 1
finally:
with _LOCK:
_CANCEL.pop(model, None)
def cmd_cancel(args: dict) -> None:
model = str(args.get("model", ""))
ev = _CANCEL.get(model)
if ev is not None:
ev.set()
_emit({"ok": True})
def cmd_select(args: dict) -> None:
model = str(args.get("model", ""))
if not is_model_downloaded(model):
_emit({"ok": False, "error": "Modelo não está instalado."})
return
save_selected_model(model)
_emit({"ok": True, "selected": load_selected_model()})
def cmd_set_language(args: dict) -> int:
"""Persist the transcription language (the default for every transcription)."""
lang = str(args.get("language", "auto"))
try:
saved = save_transcript_language(lang)
except ValueError as exc:
_emit({"ok": False, "error": str(exc)})
return 1
_emit({"ok": True, "language": saved})
return 0
def cmd_delete(args: dict) -> None:
model = str(args.get("model", ""))
try:
shutil.rmtree(model_cache_dir(model), ignore_errors=True)
except Exception:
pass
_emit({"ok": True})
def cmd_open_finder(args: dict) -> None:
target = str(args.get("path") or model_cache_dir(str(args.get("model", ""))))
try:
subprocess.Popen(["open", target])
except OSError:
pass
_emit({"ok": True})
def cmd_remove_silences(args: dict) -> int:
"""Run the canonical server silence remover into a suffixed copy."""
path = str(args.get("path", ""))
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
try:
from server import handle_remove_media_silence
output = _derived_output(path, "_silence_removed", args)
contents = asyncio.run(handle_remove_media_silence({**args, "filepath": path, "output_path": output}))
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists():
_emit({"ok": False, "error": message})
return 1
_emit({"ok": True, "path": output, "message": message})
return 0
except Exception as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_edit_by_transcript(args: dict) -> int:
"""Cut (or keep only) spoken phrases, using each media's cached transcript."""
path = str(args.get("path", ""))
phrases = args.get("phrases") or []
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
if not isinstance(phrases, list) or not [p for p in phrases if str(p).strip()]:
_emit({"ok": False, "error": "Informe ao menos uma frase para cortar."})
return 1
try:
from server import handle_edit_by_transcript
output = _derived_output(path, "_transcript_edit", args)
contents = asyncio.run(handle_edit_by_transcript({**args, "filepath": path, "output_path": output}))
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists():
_emit({"ok": False, "error": message})
return 1
_emit({"ok": True, "path": output, "message": message})
return 0
except Exception as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_remove_filler_words(args: dict) -> int:
"""Cut filler words (um, uh, ...) out, using each media's cached transcript."""
path = str(args.get("path", ""))
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
try:
from server import handle_remove_filler_words
output = _derived_output(path, "_defillered", args)
contents = asyncio.run(handle_remove_filler_words({**args, "filepath": path, "output_path": output}))
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists():
_emit({"ok": False, "error": message})
return 1
_emit({"ok": True, "path": output, "message": message})
return 0
except Exception as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_transcript_markers(args: dict) -> int:
"""Add a marker per transcribed segment, using each media's cached transcript."""
path = str(args.get("path", ""))
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
try:
from server import handle_transcript_markers
output = _derived_output(path, "_transcript_markers", args)
contents = asyncio.run(handle_transcript_markers({**args, "filepath": path, "output_path": output}))
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists():
_emit({"ok": False, "error": message})
return 1
_emit({"ok": True, "path": output, "message": message})
return 0
except Exception as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_generate_dynamic_subtitles(args: dict) -> int:
"""Generate word-by-word ("karaoke") caption compound clips, one per line,
using each media's cached transcript."""
path = str(args.get("path", ""))
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
try:
from server import handle_generate_dynamic_subtitles
output = _derived_output(path, "_dynamic_subtitles", args)
contents = asyncio.run(
handle_generate_dynamic_subtitles({**args, "filepath": path, "output_path": output})
)
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists():
_emit({"ok": False, "error": message})
return 1
_emit({"ok": True, "path": output, "message": message})
return 0
except Exception as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_add_zoom(args: dict) -> int:
"""Add an ease-in/ease-out punch-in zoom to one clip."""
path = str(args.get("path", ""))
clip_id = str(args.get("clip_id", "")).strip()
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
if not clip_id:
_emit({"ok": False, "error": "Informe o nome do clipe."})
return 1
try:
from server import handle_add_zoom
output = _derived_output(path, "_zoom", args)
contents = asyncio.run(handle_add_zoom({**args, "filepath": path, "output_path": output}))
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists():
_emit({"ok": False, "error": message})
return 1
_emit({"ok": True, "path": output, "message": message})
return 0
except Exception as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_zoom_clips(args: dict) -> int:
"""Return timeline clips with enough identity for the zoom picker."""
path = Path(str(args.get("path", "")))
output_dir = str(args.get("output_dir", "")).strip()
if not path.exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
try:
from server import _require_timeline
_, timeline = _require_timeline(str(path))
clips = []
for index, clip in enumerate(timeline.clips):
media = clip.media_path or ""
cached = _load_cached_transcript(_transcript_json_path(media, output_dir)) if media else None
clips.append({
"id": f"{index}:{clip.start.seconds:.6f}",
"index": index,
"name": clip.name,
"start": clip.start.seconds,
"duration": clip.duration_seconds,
"media": Path(media).name if media else "",
"preview": ((cached or {}).get("text", "") or "")[:180],
"has_transcript": cached is not None,
})
_emit({"ok": True, "clips": clips})
return 0
except Exception as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_zoom_segments(args: dict) -> int:
"""Return sentence/word ranges for one timeline clip."""
path = Path(str(args.get("path", "")))
output_dir = str(args.get("output_dir", "")).strip()
try:
from server import _require_timeline
_, timeline = _require_timeline(str(path))
index = int(args.get("index", -1))
if index < 0 or index >= len(timeline.clips):
raise ValueError("Clipe selecionado não existe.")
clip = timeline.clips[index]
if not clip.media_path:
raise ValueError("Este clipe não possui mídia associada.")
data = _load_cached_transcript(_transcript_json_path(clip.media_path, output_dir))
if data is None:
_emit({"ok": True, "segments": [], "message": "Transcreva este clipe primeiro."})
return 0
segments = []
for number, segment in enumerate(data.get("segments", [])):
text = str(segment.get("text", "")).strip()
if text:
segments.append({
"id": number,
"start": float(segment.get("start", 0)),
"end": float(segment.get("end", 0)),
"text": text,
})
_emit({"ok": True, "segments": segments})
return 0
except Exception as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_set_models_dir(args: dict) -> int:
try:
d = save_models_dir(str(args.get("dir", "")))
_emit({"ok": True, "models_dir": d})
return 0
except ValueError as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_inspect(args: dict) -> int:
"""Validate an FCPXML file and return a summary of its projects/timelines."""
path = str(args.get("path", ""))
if not path:
_emit({"ok": False, "error": "Nenhum arquivo informado."})
return 1
if not Path(path).exists():
_emit({"ok": False, "error": "Arquivo não encontrado."})
return 1
try:
proj = parse_fcpxml(path)
except Exception as exc:
_emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
return 1
timelines = []
for tl in proj.timelines:
timelines.append(
{
"name": tl.name,
"duration_seconds": round(tl.duration.seconds, 3),
"frame_rate": round(tl.frame_rate, 3),
"width": tl.width,
"height": tl.height,
"clips": tl.total_clips,
"cuts": tl.total_cuts,
"connected": len(tl.connected_clips),
"markers": len(tl.markers),
}
)
_emit(
{
"ok": True,
"path": path,
"name": proj.name,
"fcpxml_version": proj.fcpxml_version,
"timelines": timelines,
}
)
return 0
def cmd_transcribe(args: dict) -> int:
proj_path = str(args.get("path", ""))
output_dir = str(args.get("output_dir", "")).strip()
# Honra o modelo selecionado no programa quando nenhum é passado.
model = str(args.get("model", "") or load_selected_model() or "")
language = args.get("language")
if language is None:
language = load_transcript_language()
if language == "auto":
language = None
if not proj_path:
_emit({"type": "error", "message": "Nenhum projeto selecionado."})
return 1
if not output_dir:
_emit({"type": "error", "message": "Selecione a pasta do projeto antes de transcrever."})
return 1
if not model or not is_model_downloaded(model):
_emit(
{
"type": "error",
"message": "Nenhum modelo de transcrição instalado. Baixe e selecione um modelo na aba Modelos.",
}
)
return 1
token = str(args.get("hf_token") or load_hf_token() or "")
if args.get("num_speakers") is not None:
num_speakers = str(args.get("num_speakers"))
else:
num_speakers = load_num_speakers()
# Load project.
try:
proj = parse_fcpxml(proj_path)
except Exception as exc:
_emit({"type": "error", "message": f"Erro ao ler o projeto: {exc}"})
return 1
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
media_paths: list[str] = []
if tl is not None:
for clip in getattr(tl, "clips", []):
mp = media_src_to_path(clip.media_path or "")
if mp and Path(mp).is_file() and mp not in media_paths:
media_paths.append(mp)
if not media_paths:
_emit({"type": "error", "message": "Nenhum arquivo de mídia acessível encontrado."})
return 1
total = len(media_paths)
results: list[dict] = []
for i, mp in enumerate(media_paths, 1):
stage = f"Transcrevendo {Path(mp).name} ({i}/{total})…"
_emit({"type": "progress", "fraction": (i - 1) / total, "stage": stage})
json_path = _transcript_json_path(mp, output_dir)
cached = _load_cached_transcript(json_path)
if cached is not None:
_emit({"type": "progress", "fraction": i / total, "stage": stage})
results.append(_result_row(mp, cached))
continue
def _on_progress(file_fraction: float, _i: int = i, _stage: str = stage) -> None:
# Blend this file's own progress into the overall fraction so a
# single-media project doesn't jump straight to 100% before the
# actual (slow) decoding work has even started.
overall = (_i - 1 + file_fraction) / total
_emit({"type": "progress", "fraction": overall, "stage": _stage})
data = transcribe(mp, model_size=model, language=language, progress_cb=_on_progress)
if data is None:
_emit({"type": "error", "message": f"Não foi possível transcrever: {Path(mp).name}"})
return 1
# Diarização opcional (necessita token HF): assina speaker por segmento/palavra.
if token:
tracks = diarize(mp, token, num_speakers)
segments, words = assign_speakers(
data.get("segments", []), data.get("words", []), tracks
)
data = {**data, "segments": segments, "words": words}
data["speakers"] = build_speakers(data.get("segments", []))
payload = {
"schema_version": "1.0",
"source": Path(mp).name,
"model": model,
**data,
}
try:
_save_json_atomic(json_path, payload)
except (OSError, RuntimeError, ValueError) as exc:
_emit({"type": "error", "message": f"Não foi possível salvar o JSON: {exc}"})
return 1
results.append(_result_row(mp, data))
_emit({"type": "result", "transcripts": results})
return 0
def cmd_analyze_voice(args: dict) -> int:
"""Build the voice timeline (transcript+diarization+acoustics -> emphasis)
for every unique source media in the project, so `refine_voice_timeline`
and friends have something to read without ever reopening the audio.
Analysis only — writes _voice_timeline.json next to each media, doesn't
touch the project XML. `path` passes through unchanged so it composes
with the other batch steps (silence removal, captions) regardless of
where in the list it runs.
"""
path = str(args.get("path", ""))
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
model = str(args.get("model", "") or load_selected_model() or "")
language = args.get("language")
if language is None:
language = load_transcript_language()
if language == "auto":
language = None
token = str(args.get("hf_token") or load_hf_token() or "")
num_speakers = str(args.get("num_speakers") or load_num_speakers() or "")
try:
proj = parse_fcpxml(path)
except Exception as exc:
_emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
return 1
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
media_paths: list[str] = []
if tl is not None:
for clip in getattr(tl, "clips", []):
mp = media_src_to_path(clip.media_path or "")
if mp and Path(mp).is_file() and mp not in media_paths:
media_paths.append(mp)
if not media_paths:
_emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."})
return 1
from server import handle_build_voice_timeline
messages: list[str] = []
for mp in media_paths:
try:
contents = asyncio.run(handle_build_voice_timeline({
"media_path": mp, "model": model, "language": language,
"hf_token": token, "num_speakers": num_speakers,
"output_dir": args.get("output_dir"),
}))
except Exception as exc:
_emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"})
return 1
messages.append("\n".join(getattr(c, "text", str(c)) for c in contents))
_emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)})
return 0
def cmd_export_srt(args: dict) -> int:
"""Write a captions .srt synced to the edited timeline.
Each transcribed segment is mapped from its SOURCE-media timestamp to its
real TIMELINE position (``clip_offset + (seg_start - clip_source_start)``),
so captions only cover the frames that remain after cuts/silence removal —
not the whole source file. One .srt is produced per media, in timeline order.
"""
path = str(args.get("path", ""))
output_dir = str(args.get("output_dir", "")).strip()
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
try:
modifier = FCPXMLModifier(path)
except Exception as exc:
_emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
return 1
# Group spine clips by media so each transcript is loaded once.
by_media: dict[str, list] = {}
for _, el in modifier._iter_spine_clips():
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
mp = media_src_to_path(src)
if not mp or not Path(mp).is_file():
continue
by_media.setdefault(mp, []).append(el)
# Never emit a caption past the end of the project — Final Cut rejects an
# SRT whose last cue overruns the timeline ("subtitle extends beyond project
# duration"). Clamp every mapped cue end to this ceiling.
timeline_total = modifier._timeline_duration().to_seconds()
srt_paths: list[str] = []
for mp, clips in by_media.items():
cached = _load_cached_transcript(_transcript_json_path(mp, output_dir))
if cached is None:
continue
segments = cached.get("segments") or []
if not segments:
continue
rows: list[tuple[float, float, str, int]] = []
for el in clips:
clip_source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
clip_offset = modifier._parse_time(el.get("offset", "0s")).to_seconds()
window_end = clip_source_start + clip_duration
for seg_index, seg in enumerate(segments):
seg_start = float(seg.get("start", 0.0))
seg_end = float(seg.get("end", seg_start))
text = seg.get("text", "").strip()
if not text or seg_end <= seg_start:
continue
# Intersect the complete source segment with this kept clip.
# Testing only seg_start loses speech whose first words fall in
# a removed range; interval intersection preserves the part
# that remains and avoids duplicating a segment wholesale.
source_start = max(seg_start, clip_source_start)
source_end = min(seg_end, window_end)
if source_end <= source_start:
continue
tl_start = clip_offset + (source_start - clip_source_start)
tl_end = clip_offset + (source_end - clip_source_start)
tl_start = max(0.0, min(tl_start, timeline_total))
tl_end = max(0.0, min(tl_end, timeline_total))
if tl_end > tl_start:
rows.append((tl_start, tl_end, text, seg_index))
if not rows:
continue
rows.sort(key=lambda r: (r[0], r[1], r[3]))
# Merge only pieces from the same original Whisper segment when their
# mapped intervals touch. Never merge unrelated speech or invent time.
merged: list[tuple[float, float, str, int]] = []
for row in rows:
if merged and row[3] == merged[-1][3] and row[0] <= merged[-1][1] + 0.001:
prev = merged[-1]
merged[-1] = (prev[0], max(prev[1], row[1]), prev[2], prev[3])
else:
merged.append(row)
blocks = []
for index, (s, e, text, _) in enumerate(merged, 1):
start_stamp = srt_stamp(s)
end_stamp = srt_stamp(e)
# Millisecond SRT precision can collapse a sub-millisecond span;
# omit it rather than emit an invalid zero-duration cue.
if start_stamp == end_stamp:
continue
blocks.append(f"{index}\n{start_stamp} --> {end_stamp}\n{text}\n")
if not blocks:
continue
out = (
Path(output_dir).expanduser() / f"{Path(mp).stem}_captions.srt"
if output_dir
else Path(mp).with_name(Path(mp).stem + "_captions.srt")
)
if output_dir:
out.parent.mkdir(parents=True, exist_ok=True)
try:
out.write_text("\n".join(blocks), encoding="utf-8")
except OSError as exc:
_emit({"ok": False, "error": f"Não foi possível salvar a legenda: {exc}"})
return 1
srt_paths.append(str(out))
if not srt_paths:
_emit({"ok": False, "error": "Nenhuma transcrição encontrada. Transcreva o projeto primeiro."})
return 1
_emit({"ok": True, "paths": srt_paths, "message": f"{len(srt_paths)} legenda(s) .srt sincronizada(s) com o corte."})
return 0
def srt_stamp(seconds: float) -> str:
"""Format float seconds as ``HH:MM:SS,mmm`` (SRT uses a comma).
Uses ``floor`` (not ``round``) so a timestamp never rounds up past a frame
boundary — an SRT cue ending on the last frame must not overrun the
project duration, or Final Cut flags it as extending beyond the project.
"""
ms = int((seconds if seconds > 0 else 0.0) * 1000)
h, rem = divmod(ms, 3600000)
m, rem = divmod(rem, 60000)
s, ms = divmod(rem, 1000)
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
def _load_cached_transcript(json_path: Path) -> dict | None:
"""Return a valid cached transcript dict, or ``None`` if absent/unreadable."""
if not json_path.is_file():
return None
try:
data = json.loads(json_path.read_text(encoding="utf-8"))
except (OSError, ValueError):
return None
if isinstance(data, dict) and isinstance(data.get("words"), list):
if "speakers" not in data:
data["speakers"] = build_speakers(data.get("segments", []))
return data
return None
def cmd_rename_speakers(args: dict) -> int:
"""Apply real names to speakers already saved in a transcript JSON."""
json_path = Path(str(args.get("path", "")))
names = args.get("speakers") or {}
if not json_path.is_file():
_emit({"type": "error", "message": "Transcrição não encontrada."})
return 1
try:
data = json.loads(json_path.read_text(encoding="utf-8"))
except (OSError, ValueError) as exc:
_emit({"type": "error", "message": f"Não foi possível ler o JSON: {exc}"})
return 1
mapping = {str(sid): str(name).strip() for sid, name in (names or {}).items()}
for sp in data.get("speakers", []):
sid = str(sp.get("id", ""))
if sid in mapping and mapping[sid]:
sp["name"] = mapping[sid]
try:
_save_json_atomic(json_path, data)
except (OSError, RuntimeError, ValueError) as exc:
_emit({"type": "error", "message": f"Não foi possível salvar: {exc}"})
return 1
_emit({"ok": True, "speakers": data.get("speakers", [])})
return 0
def cmd_set_diarization(args: dict) -> int:
"""Persist the HuggingFace token and expected speaker count for diarization."""
token = args.get("token")
num = args.get("num_speakers")
if token is not None:
save_hf_token(str(token))
if num is not None:
save_num_speakers(str(num))
ok, msg = diarization_capability(load_hf_token())
_emit({"ok": True, "diarization": ok, "diarization_message": msg, "num_speakers": load_num_speakers()})
return 0
def cmd_voice_analysis(args: dict) -> int:
"""Read the persisted voice-analysis settings (energy/emphasis/emotion)."""
_emit({"ok": True, **load_voice_analysis_config()})
return 0
def cmd_set_voice_analysis(args: dict) -> int:
"""Persist voice-analysis settings. Only the given fields change."""
weights = args.get("emphasis_weights")
config = save_voice_analysis_config(
energy_threshold=args.get("energy_threshold"),
emphasis_weights=weights if isinstance(weights, dict) else None,
emphasis_threshold=args.get("emphasis_threshold"),
emotion_enabled=args.get("emotion_enabled"),
emotion_sensitivity=args.get("emotion_sensitivity"),
)
_emit({"ok": True, **config})
return 0
def cmd_dynamic_subtitle_config(args: dict) -> int:
"""Read the persisted dynamic-subtitle style (font, size, color, layout)."""
_emit({"ok": True, **load_dynamic_subtitle_config()})
return 0
def cmd_set_dynamic_subtitle_config(args: dict) -> int:
"""Persist dynamic-subtitle style fields. Only the given fields change."""
config = save_dynamic_subtitle_config(**{
k: args.get(k) for k in (
"band_height", "block_center_y", "line_gap", "font", "font_size",
"emphasis_font", "emphasis_face", "emphasis_size",
"active_color", "emphasis_color", "text_scale",
)
})
_emit({"ok": True, **config})
return 0
def cmd_silence_config(args: dict) -> int:
"""Read the persisted silence thresholds (noise floor, duration, padding)."""
_emit({"ok": True, **load_silence_config()})
return 0
def cmd_set_silence_config(args: dict) -> int:
"""Persist silence thresholds. Only the given fields change."""
config = save_silence_config(
noise_db=args.get("noise_db"),
min_silence=args.get("min_silence"),
padding=args.get("padding"),
)
_emit({"ok": True, **config})
return 0
def cmd_apply_voice_actions(args: dict) -> int:
"""Apply a decision list (cuts/zooms/texts/markers) to the project XML.
The list is produced by a model reading the _voice_timeline.json — this
is the step that turns those decisions into an edit, and the one the
batch chain was missing: without it the app could measure the voice and
caption the result, but never cut by it.
`actions_path` points at the JSON; either a bare list or the
``{"actions": [...]}`` wrapper the skill emits is accepted. Times stay in
ORIGINAL source seconds — the handler resolves cuts first and shifts
everything else itself.
"""
path = str(args.get("path", ""))
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
actions = args.get("actions")
if actions is None:
actions_path = str(args.get("actions_path", ""))
if not actions_path or not Path(actions_path).exists():
_emit({"ok": False, "error": "Arquivo de decisões (JSON) não encontrado."})
return 1
try:
with open(actions_path, encoding="utf-8") as fh:
loaded = json.load(fh)
except (OSError, ValueError) as exc:
_emit({"ok": False, "error": f"Erro ao ler as decisões: {exc}"})
return 1
actions = loaded.get("actions") if isinstance(loaded, dict) else loaded
if not isinstance(actions, list) or not actions:
_emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."})
return 1
from server import handle_apply_voice_actions
try:
contents = asyncio.run(handle_apply_voice_actions({
"filepath": path,
"actions": actions,
"output_dir": args.get("output_dir"),
}))
except Exception as exc:
_emit({"ok": False, "error": f"Falha ao aplicar as decisões: {exc}"})
return 1
message = "\n".join(getattr(c, "text", str(c)) for c in contents)
# The handler reports dropped/rejected actions individually; hand the
# whole report back so the app can surface them instead of only the count.
out_path = path
for line in message.splitlines():
if line.startswith("- **Saved to**:"):
out_path = line.split("`")[1] if "`" in line else path
break
_emit({"ok": True, "path": out_path, "message": message})
return 0
def cmd_project_config(args: dict) -> int:
"""Read the last project folder/file the app was working on."""
_emit({"ok": True, **load_project_config()})
return 0
def cmd_set_project_config(args: dict) -> int:
"""Persist the last project folder/file. Only the given fields change."""
config = save_project_config(folder=args.get("folder"), file=args.get("file"))
_emit({"ok": True, **config})
return 0
def _result_row(mp: str, data: dict) -> dict:
words = data.get("words", [])
preview = (data.get("text", "") or "")[:160]
speakers = data.get("speakers") or []
return {
"media": Path(mp).name,
"language": data.get("language", "?"),
"words": len(words),
"duration": float(data.get("duration", 0.0)),
"preview": preview,
"saved": str(_transcript_json_path(mp)),
"speakers": [s.get("name", s.get("id", "")) for s in speakers],
}
def _model_names() -> list[str]:
return [m["internal_name"] for m in load_catalog()]
def main() -> int:
args = sys.argv[1:]
if not args:
print("usage: models_api.py <command> [json_args]", file=sys.stderr)
return 1
command = args[0]
try:
data: dict = json.loads(args[1]) if len(args) > 1 else {}
except json.JSONDecodeError:
print("invalid JSON args", file=sys.stderr)
return 1
handlers = {
"catalog": cmd_catalog,
"download": cmd_download,
"cancel": cmd_cancel,
"select": cmd_select,
"set_language": cmd_set_language,
"delete": cmd_delete,
"open_finder": cmd_open_finder,
"set_models_dir": cmd_set_models_dir,
"inspect": cmd_inspect,
"transcribe": cmd_transcribe,
"export_srt": cmd_export_srt,
"remove_silences": cmd_remove_silences,
"edit_by_transcript": cmd_edit_by_transcript,
"remove_filler_words": cmd_remove_filler_words,
"transcript_markers": cmd_transcript_markers,
"generate_dynamic_subtitles": cmd_generate_dynamic_subtitles,
"add_zoom": cmd_add_zoom,
"zoom_clips": cmd_zoom_clips,
"zoom_segments": cmd_zoom_segments,
"rename_speakers": cmd_rename_speakers,
"set_diarization": cmd_set_diarization,
"voice_analysis": cmd_voice_analysis,
"set_voice_analysis": cmd_set_voice_analysis,
"analyze_voice": cmd_analyze_voice,
"dynamic_subtitle_config": cmd_dynamic_subtitle_config,
"set_dynamic_subtitle_config": cmd_set_dynamic_subtitle_config,
"apply_voice_actions": cmd_apply_voice_actions,
"project_config": cmd_project_config,
"set_project_config": cmd_set_project_config,
"silence_config": cmd_silence_config,
"set_silence_config": cmd_set_silence_config,
}
handler = handlers.get(command)
if handler is None:
print(f"unknown command: {command}", file=sys.stderr)
return 1
try:
result = handler(data)
except TypeError:
result = handler()
return result or 0
if __name__ == "__main__":
sys.exit(main())