diff --git a/admin/api/__init__.py b/admin/api/__init__.py new file mode 100644 index 0000000..0dbfd0d --- /dev/null +++ b/admin/api/__init__.py @@ -0,0 +1 @@ +"""Comandos da ponte JSON usada pelo app, agrupados por assunto.""" diff --git a/admin/api/editing.py b/admin/api/editing.py new file mode 100644 index 0000000..ece85f2 --- /dev/null +++ b/admin/api/editing.py @@ -0,0 +1,128 @@ +"""Edições no projeto: silêncio, corte por texto, preenchimento, marcadores. + +Extraído de models_api.py — a tabela de comandos segue lá. +""" + +from __future__ import annotations + +import asyncio +import sys +from pathlib import Path + +# code/ is the package root for fcpxml and server modules. +_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") +if _CODE_DIR not in sys.path: + sys.path.insert(0, _CODE_DIR) + +from fcpxml.model_manager import ( + load_silence_config, + save_silence_config, +) + +from . import shared +from .shared import ( + _derived_output, + _emit_no_change_or_error, +) + + +def cmd_remove_silences(args: dict) -> int: + """Run the canonical server silence remover into a suffixed copy.""" + path = str(args.get("path", "")) + if not path or not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + try: + from server import handle_remove_media_silence + + output = _derived_output(path, "_silence_removed", args) + contents = asyncio.run(handle_remove_media_silence({**args, "filepath": path, "output_path": output})) + message = "\n".join(getattr(content, "text", str(content)) for content in contents) + if not Path(output).exists(): + return _emit_no_change_or_error(path, message) + shared.emit({"ok": True, "path": output, "message": message}) + return 0 + except Exception as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 + +def cmd_edit_by_transcript(args: dict) -> int: + """Cut (or keep only) spoken phrases, using each media's cached transcript.""" + path = str(args.get("path", "")) + phrases = args.get("phrases") or [] + if not path or not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + if not isinstance(phrases, list) or not [p for p in phrases if str(p).strip()]: + shared.emit({"ok": False, "error": "Informe ao menos uma frase para cortar."}) + return 1 + try: + from server import handle_edit_by_transcript + + output = _derived_output(path, "_transcript_edit", args) + contents = asyncio.run(handle_edit_by_transcript({**args, "filepath": path, "output_path": output})) + message = "\n".join(getattr(content, "text", str(content)) for content in contents) + if not Path(output).exists(): + shared.emit({"ok": False, "error": message}) + return 1 + shared.emit({"ok": True, "path": output, "message": message}) + return 0 + except Exception as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 + +def cmd_remove_filler_words(args: dict) -> int: + """Cut filler words (um, uh, ...) out, using each media's cached transcript.""" + path = str(args.get("path", "")) + if not path or not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + try: + from server import handle_remove_filler_words + + output = _derived_output(path, "_defillered", args) + contents = asyncio.run(handle_remove_filler_words({**args, "filepath": path, "output_path": output})) + message = "\n".join(getattr(content, "text", str(content)) for content in contents) + if not Path(output).exists(): + return _emit_no_change_or_error(path, message) + shared.emit({"ok": True, "path": output, "message": message}) + return 0 + except Exception as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 + +def cmd_transcript_markers(args: dict) -> int: + """Add a marker per transcribed segment, using each media's cached transcript.""" + path = str(args.get("path", "")) + if not path or not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + try: + from server import handle_transcript_markers + + output = _derived_output(path, "_transcript_markers", args) + contents = asyncio.run(handle_transcript_markers({**args, "filepath": path, "output_path": output})) + message = "\n".join(getattr(content, "text", str(content)) for content in contents) + if not Path(output).exists(): + shared.emit({"ok": False, "error": message}) + return 1 + shared.emit({"ok": True, "path": output, "message": message}) + return 0 + except Exception as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 + +def cmd_silence_config(args: dict) -> int: + """Read the persisted silence thresholds (noise floor, duration, padding).""" + shared.emit({"ok": True, **load_silence_config()}) + return 0 + +def cmd_set_silence_config(args: dict) -> int: + """Persist silence thresholds. Only the given fields change.""" + config = save_silence_config( + noise_db=args.get("noise_db"), + min_silence=args.get("min_silence"), + padding=args.get("padding"), + ) + shared.emit({"ok": True, **config}) + return 0 diff --git a/admin/api/models.py b/admin/api/models.py new file mode 100644 index 0000000..fb75ef6 --- /dev/null +++ b/admin/api/models.py @@ -0,0 +1,145 @@ +"""Catálogo de modelos: listar, baixar, escolher, apagar. + +Extraído de models_api.py — a tabela de comandos segue lá. +""" + +from __future__ import annotations + +import shutil +import subprocess +import sys +import threading +from pathlib import Path + +# code/ is the package root for fcpxml and server modules. +_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") +if _CODE_DIR not in sys.path: + sys.path.insert(0, _CODE_DIR) + +from fcpxml.diarize import ( + diarization_capability, +) +from fcpxml.model_manager import ( + download_model, + get_models_dir, + is_model_downloaded, + list_installed_models, + load_catalog, + load_hf_token, + load_num_speakers, + load_selected_model, + load_transcript_language, + model_cache_dir, + save_models_dir, + save_selected_model, + save_transcript_language, +) + +from . import shared +from .shared import ( + RECOMMENDED, +) + + +# Downloads em andamento, para o comando `cancel` conseguir interrompê-los. +# Mora aqui, e não no shared, porque só `download` e `cancel` o tocam — e o +# lock é próprio: ele protege este dicionário, não a saída em stdout. +_CANCEL: dict[str, threading.Event] = {} +_CANCEL_LOCK = threading.Lock() + + +def cmd_catalog() -> None: + catalog = load_catalog() + installed = list_installed_models() + diar_ok, diar_msg = diarization_capability(load_hf_token()) + shared.emit( + { + "models": catalog, + "installed": installed, + "selected": load_selected_model(), + "language": load_transcript_language(), + "models_dir": str(get_models_dir()), + "installed_count": len(installed), + "recommended": list(RECOMMENDED), + "diarization": diar_ok, + "diarization_message": diar_msg, + "hf_token_set": bool(load_hf_token()), + "num_speakers": load_num_speakers(), + } + ) + +def cmd_download(args: dict) -> int: + model = str(args.get("model", "")) + if model not in _model_names(): + shared.emit({"type": "error", "message": f"Modelo desconhecido: {model}"}) + return 1 + ev = threading.Event() + with _CANCEL_LOCK: + _CANCEL[model] = ev + try: + download_model(model, progress_cb=lambda f: shared.emit({"type": "progress", "fraction": f}), cancel_event=ev) + installed = is_model_downloaded(model) + shared.emit({"type": "done", "installed": installed}) + if installed: + save_selected_model(model) + return 0 if installed else 1 + except Exception as exc: + shared.emit({"type": "error", "message": str(exc)}) + return 1 + finally: + with _CANCEL_LOCK: + _CANCEL.pop(model, None) + +def cmd_cancel(args: dict) -> None: + model = str(args.get("model", "")) + ev = _CANCEL.get(model) + if ev is not None: + ev.set() + shared.emit({"ok": True}) + +def cmd_select(args: dict) -> None: + model = str(args.get("model", "")) + if not is_model_downloaded(model): + shared.emit({"ok": False, "error": "Modelo não está instalado."}) + return + save_selected_model(model) + shared.emit({"ok": True, "selected": load_selected_model()}) + +def cmd_set_language(args: dict) -> int: + """Persist the transcription language (the default for every transcription).""" + lang = str(args.get("language", "auto")) + try: + saved = save_transcript_language(lang) + except ValueError as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 + shared.emit({"ok": True, "language": saved}) + return 0 + +def cmd_delete(args: dict) -> None: + model = str(args.get("model", "")) + try: + shutil.rmtree(model_cache_dir(model), ignore_errors=True) + except Exception: + pass + shared.emit({"ok": True}) + +def cmd_open_finder(args: dict) -> None: + target = str(args.get("path") or model_cache_dir(str(args.get("model", "")))) + try: + subprocess.Popen(["open", target]) + except OSError: + pass + shared.emit({"ok": True}) + +def cmd_set_models_dir(args: dict) -> int: + try: + d = save_models_dir(str(args.get("dir", ""))) + shared.emit({"ok": True, "models_dir": d}) + return 0 + except ValueError as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 + +def _model_names() -> list[str]: + return [m["internal_name"] for m in load_catalog()] diff --git a/admin/api/project.py b/admin/api/project.py new file mode 100644 index 0000000..c5ad377 --- /dev/null +++ b/admin/api/project.py @@ -0,0 +1,75 @@ +"""Projeto: inspecionar o .fcpxml e lembrar a pasta/arquivo em uso. + +Extraído de models_api.py — a tabela de comandos segue lá. +""" + +from __future__ import annotations + +import sys +from pathlib import Path + +# code/ is the package root for fcpxml and server modules. +_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") +if _CODE_DIR not in sys.path: + sys.path.insert(0, _CODE_DIR) + +from fcpxml.model_manager import ( + load_project_config, + save_project_config, +) +from fcpxml.parser import parse_fcpxml + +from . import shared + + +def cmd_inspect(args: dict) -> int: + """Validate an FCPXML file and return a summary of its projects/timelines.""" + path = str(args.get("path", "")) + if not path: + shared.emit({"ok": False, "error": "Nenhum arquivo informado."}) + return 1 + if not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo não encontrado."}) + return 1 + try: + proj = parse_fcpxml(path) + except Exception as exc: + shared.emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"}) + return 1 + + timelines = [] + for tl in proj.timelines: + timelines.append( + { + "name": tl.name, + "duration_seconds": round(tl.duration.seconds, 3), + "frame_rate": round(tl.frame_rate, 3), + "width": tl.width, + "height": tl.height, + "clips": tl.total_clips, + "cuts": tl.total_cuts, + "connected": len(tl.connected_clips), + "markers": len(tl.markers), + } + ) + shared.emit( + { + "ok": True, + "path": path, + "name": proj.name, + "fcpxml_version": proj.fcpxml_version, + "timelines": timelines, + } + ) + return 0 + +def cmd_project_config(args: dict) -> int: + """Read the last project folder/file the app was working on.""" + shared.emit({"ok": True, **load_project_config()}) + return 0 + +def cmd_set_project_config(args: dict) -> int: + """Persist the last project folder/file. Only the given fields change.""" + config = save_project_config(folder=args.get("folder"), file=args.get("file")) + shared.emit({"ok": True, **config}) + return 0 diff --git a/admin/api/review.py b/admin/api/review.py new file mode 100644 index 0000000..6618926 --- /dev/null +++ b/admin/api/review.py @@ -0,0 +1,97 @@ +"""Revisão de frases: montar a tela de ênfases e salvar o que foi decidido. + +Extraído de models_api.py — a tabela de comandos segue lá. +""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +# code/ is the package root for fcpxml and server modules. +_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") +if _CODE_DIR not in sys.path: + sys.path.insert(0, _CODE_DIR) + + +from . import shared + + +def cmd_build_phrase_review(args: dict) -> int: + """Build the reviewable script (phrases + the AI's decisions) for the wizard. + + `voice_timeline` points at the _voice_timeline.json; `actions` carries the + decision list the model returned (inline, in any of the shapes the skill + emits). The review is always rebuilt from the current analysis, then the + decisions saved on a previous visit are laid back over it — reopening the + step must show the edits the user left there without freezing the acoustics + as they were when they left. + """ + from fcpxml.phrase_review import ( + build_phrase_review, + load_phrase_review, + merge_saved_decisions, + ) + + timeline_path = str(args.get("voice_timeline", "")) + if not timeline_path or not Path(timeline_path).exists(): + shared.emit({"ok": False, "error": "Análise de voz (voice_timeline.json) não encontrada."}) + return 1 + + try: + with open(timeline_path, encoding="utf-8") as fh: + timeline = json.load(fh) + except (OSError, ValueError) as exc: + shared.emit({"ok": False, "error": f"Erro ao ler a análise de voz: {exc}"}) + return 1 + + extra = [d for d in (args.get("output_dir"), args.get("media_dir")) if d] + review = build_phrase_review( + timeline, + args.get("actions"), + voice_timeline_path=timeline_path, + extra_dirs=extra, + ) + + saved = None if args.get("fresh") else load_phrase_review(timeline_path) + review = merge_saved_decisions(review, saved) + shared.emit({"ok": True, "reused": saved is not None, **review}) + return 0 + +def cmd_save_phrase_review(args: dict) -> int: + """Persist the edited review and the actions derived from it.""" + from fcpxml.phrase_review import save_phrase_review + + timeline_path = str(args.get("voice_timeline", "")) + if not timeline_path: + shared.emit({"ok": False, "error": "Caminho da análise de voz não informado."}) + return 1 + + phrases = args.get("phrases") + if not isinstance(phrases, list): + shared.emit({"ok": False, "error": "Nenhuma frase para salvar."}) + return 1 + + review = { + "version": args.get("version", "1.0"), + "source": args.get("source", ""), + "duration": args.get("duration", 0.0), + "speakers": args.get("speakers", []), + "phrases": phrases, + "zooms": args.get("zooms", []), + } + try: + review_path, actions_path = save_phrase_review(timeline_path, review) + except OSError as exc: + shared.emit({"ok": False, "error": f"Erro ao salvar a revisão: {exc}"}) + return 1 + + shared.emit({ + "ok": True, + "review_path": str(review_path), + "actions_path": str(actions_path), + "emphasis_count": sum(1 for p in phrases if int(p.get("emphasis", 0) or 0) >= 1), + "removed_count": sum(1 for p in phrases if not p.get("active", True)), + }) + return 0 diff --git a/admin/api/shared.py b/admin/api/shared.py new file mode 100644 index 0000000..cbddcc8 --- /dev/null +++ b/admin/api/shared.py @@ -0,0 +1,306 @@ +"""Base comum dos comandos da ponte: saída JSON, caminhos derivados e cache. + +A saída passa toda por `emit`. Os módulos de comando chamam `shared.emit(...)` +pelo módulo, e não pelo nome importado, de propósito: assim trocar `emit` num +lugar só — como a suíte faz para capturar a saída — continua alcançando todos +os comandos, o que deixaria de valer se cada um tivesse ligado o nome no seu +próprio import. +""" + +from __future__ import annotations + +import json +import os +import sys +import threading +from pathlib import Path +from typing import Any + +# code/ is the package root for fcpxml and server modules. +_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") +if _CODE_DIR not in sys.path: + sys.path.insert(0, _CODE_DIR) + +from fcpxml.media_intel import media_src_to_path +from fcpxml.parser import parse_fcpxml +from fcpxml.diarize import build_speakers # noqa: E402 + +"""JSON bridge between the SwiftUI app and the fcp-mcp-server Python engine. + +The SwiftUI app (MacApp/) launches this script as a subprocess with a command +and optional JSON arguments, then reads a single JSON document (or +newline-delimited JSON for progress) on stdout. + +Commands: + catalog + -> {"models": [{display_name, internal_name, size, storage, + accuracy, speed}], "installed": [names], + "selected": name, "models_dir": path, "installed_count": n, + "recommended": [names]} + + download {"model": "small"} + -> JSON-lines: {"type":"progress","fraction":0.42} + {"type":"done","installed":true} + {"type":"error","message":"..."} + + cancel {"model": "small"} + -> {"ok": true} + + select {"model": "small"} + -> {"ok": true, "selected": "small"} + + set_language {"language": "pt"} | "auto" + -> {"ok": true, "language": "pt"} + + delete {"model": "small"} + -> {"ok": true} + + open_finder {"model": "small"} + -> {"ok": true} + + set_models_dir {"dir": "/path"} + -> {"ok": true, "models_dir": "/path"} + + inspect {"path": "/path/to/project.fcpxml"} + -> {"ok": true, "path": "...", "name": "...", "fcpxml_version": "1.13", + "timelines": [{name, duration_seconds, frame_rate, width, height, + clips, cuts, connected, markers}]} + or {"ok": false, "error": "..."} + + analyze_voice {"path": "...", "output_dir": "...", "model": "...", + "language": "pt"|"auto"|null, "hf_token": "..."|null, + "num_speakers": ""|null} + Build the voice timeline (transcript+diarization+acoustics) for + every unique source media — analysis only, writes _voice_timeline.json + next to each media, `path` passes through unchanged. Meant as one + entry in the batch operations list (see processBatchStep), so + `refine_voice_timeline` never has to reopen the audio later. + -> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."} + + build_phrase_review {"voice_timeline": "..._voice_timeline.json", + "actions": {...}|[...]|null, "fresh": false} + The reviewable script for the wizard's emphasis step: every phrase with + the AI's decision already applied (active/emphasis/trim). A review saved + earlier for the same timeline is returned as-is unless `fresh` is true. + -> {"ok": true, "reused": bool, "source", "duration", "speakers", + "phrases": [{index, start, end, trim_start, trim_end, text, speaker, + active, emphasis (0-3), track, peak_emphasis, + take_boundary, gap_before, reason, words}], + "errors": [...]} + + save_phrase_review {"voice_timeline": "...", "phrases": [...], "source": "...", + "duration": 0.0, "speakers": [...]} + Writes _phrase_review.json plus the _phrase_actions.json derived from it. + -> {"ok": true, "review_path", "actions_path", "emphasis_count", + "removed_count"} + + dynamic_subtitle_config {} + -> {"ok": true, "band_height", "block_center_y", "line_gap", "font", + "font_size", "emphasis_font", "emphasis_face", "emphasis_size", + "active_color", "emphasis_color", "text_scale"} + + set_dynamic_subtitle_config {} + Persists only the given fields to ~/.fcp-mcp-server/config.json. + generate_dynamic_subtitles reads this as its own fallback default. + -> {"ok": true, } + + silence_config {} + -> {"ok": true, "noise_db": -30.0, "min_silence": 0.5, "padding": 0.05} + + set_silence_config {"noise_db": -30.0, "min_silence": 0.5, "padding": 0.05} + Persists only the given fields. detect_media_silence and + remove_media_silence read this as their own fallback default. + -> {"ok": true, } + + transcribe {"path": "...", "model": "small", "language": "pt"|null, + "hf_token": "..."|null, "num_speakers": ""|null} + -> JSON-lines: + {"type":"progress","fraction":0.5,"stage":"Transcrevendo..."} + {"type":"result","transcripts":[{"media","language","words", + "duration","preview","saved", + "speakers"}]} + {"type":"error","message":"..."} + + edit_by_transcript {"path": "...", "phrases": ["frase um", "frase dois"], + "mode": "remove"|"keep_only", "clip_name": "..."|null, + "padding": 0.0, "model": "small", "language": "pt"|null} + -> {"ok": true, "path": "..._transcript_edit.fcpxml", "message": "..."} + or {"ok": false, "error": "..."} + + remove_filler_words {"path": "...", "fillers": ["um","uh"]|null, + "clip_name": "..."|null, "padding": 0.02, + "model": "small", "language": "pt"|null} + -> {"ok": true, "path": "..._defillered.fcpxml", "message": "..."} + or {"ok": false, "error": "..."} + + transcript_markers {"path": "...", "clip_name": "..."|null, + "marker_type": "chapter", "max_label_length": 50, + "model": "small", "language": "pt"|null} + -> {"ok": true, "path": "..._transcript_markers.fcpxml", "message": "..."} + or {"ok": false, "error": "..."} + + add_zoom {"path": "...", "clip_id": "...", "start": 10.0, "end": 16.0, + "scale": 1.3, "ease": 0.3, "position": "0 0"|null} + -> {"ok": true, "path": "..._zoom.fcpxml", "message": "..."} + or {"ok": false, "error": "..."} + + generate_dynamic_subtitles {"path": "...", "clip_name": "..."|null, + "band_height": 0.22, "block_center_y": -167, + "font": "Helvetica Neue", "font_size": 128, + "emphasis_font": "Playfair Display", + "emphasis_face": "Medium Italic", "emphasis_size": 265, + "active_color": "1 1 1 1", "emphasis_color": "1 1 1 1", + "model": "small", "language": "pt"|null} + -> {"ok": true, "path": "..._dynamic_subtitles.fcpxml", "message": "..."} + or {"ok": false, "error": "..."} + + rename_speakers {"path": "/to/media_transcript.json", + "speakers": {"SPEAKER_01": "Nome"}} + -> {"ok": true, "speakers": [...]} + + set_diarization {"token": "hf_...", "num_speakers": ""} + -> {"ok": true, "diarization": bool, "diarization_message": "...", + "num_speakers": "..."} + + acoustics_capability + Whether librosa (pitch/energy for voice analysis) is installed. + -> {"ok": true, "available": bool, "message": "..."} + + voice_analysis + -> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85, + "emphasis_weights": {...}, "emotion_enabled": false, + "emotion_sensitivity": 0.5} + + set_voice_analysis {"energy_threshold": 0.6, "emphasis_threshold": 0.9, + "emphasis_weights": {"energy": 0.4}|null, + "emotion_enabled": true, "emotion_sensitivity": 0.5} + -> same shape as voice_analysis (only given fields change) + +Exit code 0 on success, 1 on error. +""" + +_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") + +if _CODE_DIR not in sys.path: + sys.path.insert(0, _CODE_DIR) + +RECOMMENDED = ("large-v3", "distil-large-v3", "small", "base") + +def _derived_output(path: str, suffix: str, args: dict) -> str: + """Resolve a derived XML path, optionally inside the chosen output folder.""" + output_dir = str(args.get("output_dir", "")).strip() + if output_dir: + directory = Path(output_dir).expanduser() + directory.mkdir(parents=True, exist_ok=True) + source = Path(path) + extension = ".fcpxmld" if source.is_dir() else source.suffix + return str(directory / f"{source.stem}{suffix}{extension}") + from server import generate_output_path + return generate_output_path(path, suffix) + +def _is_no_change_message(message: str) -> bool: + """Whether a tool completed cleanly without needing to save a new file.""" + text = message.lower() + return any( + token in text + for token in ( + "no cuts to make", + "no silence", + "file unchanged", + "nothing saved", + ) + ) + +def _emit_no_change_or_error(path: str, message: str) -> int: + if _is_no_change_message(message): + emit({"ok": True, "path": path, "unchanged": True, "message": message}) + return 0 + emit({"ok": False, "error": message}) + return 1 + + +# Serializa a escrita em stdout. A ponte é JSON-lines: dois comandos +# escrevendo ao mesmo tempo entrelaçariam documentos e o app leria lixo. +_OUT_LOCK = threading.Lock() + +def emit(obj: Any) -> None: + sys.stdout.write(json.dumps(obj, ensure_ascii=False) + "\n") + sys.stdout.flush() + +def _transcript_json_path(media_path: str, output_dir: str = "") -> Path: + """Where the ``_transcript.json`` for ``media_path`` lives. + + When ``output_dir`` (the user-selected project folder) is set, the + transcript is saved/read there — never next to the source media, which + may sit on a read-only volume or a Final Cut Library the user never + browses. Falls back to the media's own folder only when no project + folder has been chosen (legacy/MCP callers). + """ + p = Path(media_path) + if output_dir: + directory = Path(output_dir).expanduser() + directory.mkdir(parents=True, exist_ok=True) + return directory / f"{p.stem}_transcript.json" + return p.with_name(p.stem + "_transcript.json") + +def _save_json_atomic(path: Path, data: Any) -> None: + """Write ``data`` to ``path`` atomically and validate the result on disk. + + Mirrors the reference WHISPERX save path: write a ``.tmp``, ``os.replace`` + into place, then confirm the file exists, is non-empty, and parses as JSON. + """ + tmp_path = str(path) + ".tmp" + with open(tmp_path, "w", encoding="utf-8") as fh: + json.dump(data, fh, ensure_ascii=False, indent=2) + os.replace(tmp_path, path) + if not path.exists() or os.path.getsize(path) == 0: + raise RuntimeError("O arquivo salvo está vazio ou não foi encontrado.") + with open(path, encoding="utf-8") as fh: + json.load(fh) + +def _project_media_paths(path: str) -> list[str]: + proj = parse_fcpxml(path) + tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None) + media_paths: list[str] = [] + if tl is not None: + for clip in getattr(tl, "clips", []): + mp = media_src_to_path(clip.media_path or "") + if mp and Path(mp).is_file() and mp not in media_paths: + media_paths.append(mp) + return media_paths + +def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path: + p = Path(media_path) + if output_dir: + directory = Path(output_dir).expanduser() + directory.mkdir(parents=True, exist_ok=True) + return directory / f"{p.stem}_voice_timeline.json" + return p.with_name(p.stem + "_voice_timeline.json") + +def _load_cached_voice_timeline(json_path: Path, media_path: str) -> dict | None: + try: + with open(json_path, encoding="utf-8") as fh: + data = json.load(fh) + except (OSError, json.JSONDecodeError, UnicodeDecodeError): + return None + if not isinstance(data, dict): + return None + if data.get("source") != Path(media_path).name: + return None + if not isinstance(data.get("segments"), list): + return None + return data + +def _load_cached_transcript(json_path: Path) -> dict | None: + """Return a valid cached transcript dict, or ``None`` if absent/unreadable.""" + if not json_path.is_file(): + return None + try: + data = json.loads(json_path.read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + if isinstance(data, dict) and isinstance(data.get("words"), list): + if "speakers" not in data: + data["speakers"] = build_speakers(data.get("segments", [])) + return data + return None diff --git a/admin/api/subtitles.py b/admin/api/subtitles.py new file mode 100644 index 0000000..9a49730 --- /dev/null +++ b/admin/api/subtitles.py @@ -0,0 +1,240 @@ +"""Legendas: dinâmicas, comuns, SRT e as configurações de estilo. + +Extraído de models_api.py — a tabela de comandos segue lá. +""" + +from __future__ import annotations + +import asyncio +import sys +from pathlib import Path + +# code/ is the package root for fcpxml and server modules. +_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") +if _CODE_DIR not in sys.path: + sys.path.insert(0, _CODE_DIR) + +from fcpxml.media_intel import media_src_to_path +from fcpxml.model_manager import ( + load_dynamic_subtitle_config, + load_plain_subtitle_config, + save_dynamic_subtitle_config, + save_plain_subtitle_config, +) +from fcpxml.writer import FCPXMLModifier + +from . import shared +from .shared import ( + _derived_output, + _emit_no_change_or_error, + _transcript_json_path, +) +from .shared import _load_cached_transcript # noqa: E402 + + +def cmd_generate_dynamic_subtitles(args: dict) -> int: + """Generate word-by-word ("karaoke") caption compound clips, one per line, + using each media's cached transcript.""" + path = str(args.get("path", "")) + if not path or not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + try: + from server import handle_generate_dynamic_subtitles + + output = _derived_output(path, "_dynamic_subtitles", args) + contents = asyncio.run( + handle_generate_dynamic_subtitles({**args, "filepath": path, "output_path": output}) + ) + message = "\n".join(getattr(content, "text", str(content)) for content in contents) + if not Path(output).exists(): + shared.emit({"ok": False, "error": message}) + return 1 + shared.emit({"ok": True, "path": output, "message": message}) + return 0 + except Exception as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 + +def cmd_generate_plain_subtitles(args: dict) -> int: + """Generate simple static editable subtitle title clips.""" + path = str(args.get("path", "")) + if not path or not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + try: + from server import handle_generate_plain_subtitles + + output = _derived_output(path, "_plain_subtitles", args) + contents = asyncio.run( + handle_generate_plain_subtitles({**args, "filepath": path, "output_path": output}) + ) + message = "\n".join(getattr(content, "text", str(content)) for content in contents) + if not Path(output).exists(): + return _emit_no_change_or_error(path, message) + shared.emit({"ok": True, "path": output, "message": message}) + return 0 + except Exception as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 + +def cmd_export_srt(args: dict) -> int: + """Write a captions .srt synced to the edited timeline. + + Each transcribed segment is mapped from its SOURCE-media timestamp to its + real TIMELINE position (``clip_offset + (seg_start - clip_source_start)``), + so captions only cover the frames that remain after cuts/silence removal — + not the whole source file. One .srt is produced per media, in timeline order. + """ + path = str(args.get("path", "")) + output_dir = str(args.get("output_dir", "")).strip() + if not path or not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + try: + modifier = FCPXMLModifier(path) + except Exception as exc: + shared.emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"}) + return 1 + + # Group spine clips by media so each transcript is loaded once. + by_media: dict[str, list] = {} + for _, el in modifier._iter_spine_clips(): + src = modifier.resources.get(el.get("ref", ""), {}).get("src", "") + mp = media_src_to_path(src) + if not mp or not Path(mp).is_file(): + continue + by_media.setdefault(mp, []).append(el) + + # Never emit a caption past the end of the project — Final Cut rejects an + # SRT whose last cue overruns the timeline ("subtitle extends beyond project + # duration"). Clamp every mapped cue end to this ceiling. + timeline_total = modifier._timeline_duration().to_seconds() + + srt_paths: list[str] = [] + for mp, clips in by_media.items(): + cached = _load_cached_transcript(_transcript_json_path(mp, output_dir)) + if cached is None: + continue + segments = cached.get("segments") or [] + if not segments: + continue + + rows: list[tuple[float, float, str, int]] = [] + for el in clips: + clip_source_start = modifier.source_file_start(el).to_seconds() + clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() + clip_offset = modifier._parse_time(el.get("offset", "0s")).to_seconds() + window_end = clip_source_start + clip_duration + for seg_index, seg in enumerate(segments): + seg_start = float(seg.get("start", 0.0)) + seg_end = float(seg.get("end", seg_start)) + text = seg.get("text", "").strip() + if not text or seg_end <= seg_start: + continue + # Intersect the complete source segment with this kept clip. + # Testing only seg_start loses speech whose first words fall in + # a removed range; interval intersection preserves the part + # that remains and avoids duplicating a segment wholesale. + source_start = max(seg_start, clip_source_start) + source_end = min(seg_end, window_end) + if source_end <= source_start: + continue + tl_start = clip_offset + (source_start - clip_source_start) + tl_end = clip_offset + (source_end - clip_source_start) + tl_start = max(0.0, min(tl_start, timeline_total)) + tl_end = max(0.0, min(tl_end, timeline_total)) + if tl_end > tl_start: + rows.append((tl_start, tl_end, text, seg_index)) + + if not rows: + continue + rows.sort(key=lambda r: (r[0], r[1], r[3])) + # Merge only pieces from the same original Whisper segment when their + # mapped intervals touch. Never merge unrelated speech or invent time. + merged: list[tuple[float, float, str, int]] = [] + for row in rows: + if merged and row[3] == merged[-1][3] and row[0] <= merged[-1][1] + 0.001: + prev = merged[-1] + merged[-1] = (prev[0], max(prev[1], row[1]), prev[2], prev[3]) + else: + merged.append(row) + + blocks = [] + for index, (s, e, text, _) in enumerate(merged, 1): + start_stamp = srt_stamp(s) + end_stamp = srt_stamp(e) + # Millisecond SRT precision can collapse a sub-millisecond span; + # omit it rather than emit an invalid zero-duration cue. + if start_stamp == end_stamp: + continue + blocks.append(f"{index}\n{start_stamp} --> {end_stamp}\n{text}\n") + if not blocks: + continue + + out = ( + Path(output_dir).expanduser() / f"{Path(mp).stem}_captions.srt" + if output_dir + else Path(mp).with_name(Path(mp).stem + "_captions.srt") + ) + if output_dir: + out.parent.mkdir(parents=True, exist_ok=True) + try: + out.write_text("\n".join(blocks), encoding="utf-8") + except OSError as exc: + shared.emit({"ok": False, "error": f"Não foi possível salvar a legenda: {exc}"}) + return 1 + srt_paths.append(str(out)) + + if not srt_paths: + shared.emit({"ok": False, "error": "Nenhuma transcrição encontrada. Transcreva o projeto primeiro."}) + return 1 + + shared.emit({"ok": True, "paths": srt_paths, "message": f"{len(srt_paths)} legenda(s) .srt sincronizada(s) com o corte."}) + return 0 + +def srt_stamp(seconds: float) -> str: + """Format float seconds as ``HH:MM:SS,mmm`` (SRT uses a comma). + + Uses ``floor`` (not ``round``) so a timestamp never rounds up past a frame + boundary — an SRT cue ending on the last frame must not overrun the + project duration, or Final Cut flags it as extending beyond the project. + """ + ms = int((seconds if seconds > 0 else 0.0) * 1000) + h, rem = divmod(ms, 3600000) + m, rem = divmod(rem, 60000) + s, ms = divmod(rem, 1000) + return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}" + +def cmd_dynamic_subtitle_config(args: dict) -> int: + """Read the persisted dynamic-subtitle style (font, size, color, layout).""" + shared.emit({"ok": True, **load_dynamic_subtitle_config()}) + return 0 + +def cmd_set_dynamic_subtitle_config(args: dict) -> int: + """Persist dynamic-subtitle style fields. Only the given fields change.""" + config = save_dynamic_subtitle_config(**{ + k: args.get(k) for k in ( + "band_height", "block_center_y", "line_gap", "font", "font_size", + "emphasis_font", "emphasis_face", "emphasis_size", + "active_color", "emphasis_color", "text_scale", + ) + }) + shared.emit({"ok": True, **config}) + return 0 + +def cmd_plain_subtitle_config(args: dict) -> int: + """Read the persisted simple subtitle style.""" + shared.emit({"ok": True, **load_plain_subtitle_config()}) + return 0 + +def cmd_set_plain_subtitle_config(args: dict) -> int: + """Persist simple subtitle style fields. Only the given fields change.""" + config = save_plain_subtitle_config(**{ + k: args.get(k) for k in ( + "font", "font_size", "font_color", "max_words", + "position_y", "uppercase", "keep_punctuation", "text_scale", + ) + }) + shared.emit({"ok": True, **config}) + return 0 diff --git a/admin/api/transcription.py b/admin/api/transcription.py new file mode 100644 index 0000000..27a04c6 --- /dev/null +++ b/admin/api/transcription.py @@ -0,0 +1,191 @@ +"""Transcrição e locutores. + +Extraído de models_api.py — a tabela de comandos segue lá. +""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +# code/ is the package root for fcpxml and server modules. +_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") +if _CODE_DIR not in sys.path: + sys.path.insert(0, _CODE_DIR) + +from fcpxml.diarize import ( + assign_speakers, + build_speakers, + diarization_capability, + diarize, +) +from fcpxml.media_intel import media_src_to_path +from fcpxml.model_manager import ( + is_model_downloaded, + load_hf_token, + load_num_speakers, + load_selected_model, + load_transcript_language, + save_hf_token, + save_num_speakers, +) +from fcpxml.parser import parse_fcpxml +from fcpxml.transcribe import transcribe + +from . import shared +from .shared import ( + _save_json_atomic, + _transcript_json_path, +) +from .shared import _load_cached_transcript # noqa: E402 + + +def cmd_transcribe(args: dict) -> int: + proj_path = str(args.get("path", "")) + output_dir = str(args.get("output_dir", "")).strip() + # Honra o modelo selecionado no programa quando nenhum é passado. + model = str(args.get("model", "") or load_selected_model() or "") + language = args.get("language") + if language is None: + language = load_transcript_language() + if language == "auto": + language = None + if not proj_path: + shared.emit({"type": "error", "message": "Nenhum projeto selecionado."}) + return 1 + if not output_dir: + shared.emit({"type": "error", "message": "Selecione a pasta do projeto antes de transcrever."}) + return 1 + if not model or not is_model_downloaded(model): + shared.emit( + { + "type": "error", + "message": "Nenhum modelo de transcrição instalado. Baixe e selecione um modelo na aba Modelos.", + } + ) + return 1 + + token = str(args.get("hf_token") or load_hf_token() or "") + if args.get("num_speakers") is not None: + num_speakers = str(args.get("num_speakers")) + else: + num_speakers = load_num_speakers() + + # Load project. + try: + proj = parse_fcpxml(proj_path) + except Exception as exc: + shared.emit({"type": "error", "message": f"Erro ao ler o projeto: {exc}"}) + return 1 + tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None) + media_paths: list[str] = [] + if tl is not None: + for clip in getattr(tl, "clips", []): + mp = media_src_to_path(clip.media_path or "") + if mp and Path(mp).is_file() and mp not in media_paths: + media_paths.append(mp) + if not media_paths: + shared.emit({"type": "error", "message": "Nenhum arquivo de mídia acessível encontrado."}) + return 1 + + total = len(media_paths) + results: list[dict] = [] + for i, mp in enumerate(media_paths, 1): + stage = f"Transcrevendo {Path(mp).name} ({i}/{total})…" + shared.emit({"type": "progress", "fraction": (i - 1) / total, "stage": stage}) + json_path = _transcript_json_path(mp, output_dir) + cached = _load_cached_transcript(json_path) + if cached is not None: + shared.emit({"type": "progress", "fraction": i / total, "stage": stage}) + results.append(_result_row(mp, cached)) + continue + + def _on_progress(file_fraction: float, _i: int = i, _stage: str = stage) -> None: + # Blend this file's own progress into the overall fraction so a + # single-media project doesn't jump straight to 100% before the + # actual (slow) decoding work has even started. + overall = (_i - 1 + file_fraction) / total + shared.emit({"type": "progress", "fraction": overall, "stage": _stage}) + + data = transcribe(mp, model_size=model, language=language, progress_cb=_on_progress) + if data is None: + shared.emit({"type": "error", "message": f"Não foi possível transcrever: {Path(mp).name}"}) + return 1 + + # Diarização opcional (necessita token HF): assina speaker por segmento/palavra. + if token: + tracks = diarize(mp, token, num_speakers) + segments, words = assign_speakers( + data.get("segments", []), data.get("words", []), tracks + ) + data = {**data, "segments": segments, "words": words} + data["speakers"] = build_speakers(data.get("segments", [])) + + payload = { + "schema_version": "1.0", + "source": Path(mp).name, + "model": model, + **data, + } + try: + _save_json_atomic(json_path, payload) + except (OSError, RuntimeError, ValueError) as exc: + shared.emit({"type": "error", "message": f"Não foi possível salvar o JSON: {exc}"}) + return 1 + results.append(_result_row(mp, data)) + + shared.emit({"type": "result", "transcripts": results}) + return 0 + + +def cmd_rename_speakers(args: dict) -> int: + """Apply real names to speakers already saved in a transcript JSON.""" + json_path = Path(str(args.get("path", ""))) + names = args.get("speakers") or {} + if not json_path.is_file(): + shared.emit({"type": "error", "message": "Transcrição não encontrada."}) + return 1 + try: + data = json.loads(json_path.read_text(encoding="utf-8")) + except (OSError, ValueError) as exc: + shared.emit({"type": "error", "message": f"Não foi possível ler o JSON: {exc}"}) + return 1 + mapping = {str(sid): str(name).strip() for sid, name in (names or {}).items()} + for sp in data.get("speakers", []): + sid = str(sp.get("id", "")) + if mapping.get(sid): + sp["name"] = mapping[sid] + try: + _save_json_atomic(json_path, data) + except (OSError, RuntimeError, ValueError) as exc: + shared.emit({"type": "error", "message": f"Não foi possível salvar: {exc}"}) + return 1 + shared.emit({"ok": True, "speakers": data.get("speakers", [])}) + return 0 + +def cmd_set_diarization(args: dict) -> int: + """Persist the HuggingFace token and expected speaker count for diarization.""" + token = args.get("token") + num = args.get("num_speakers") + if token is not None: + save_hf_token(str(token)) + if num is not None: + save_num_speakers(str(num)) + ok, msg = diarization_capability(load_hf_token()) + shared.emit({"ok": True, "diarization": ok, "diarization_message": msg, "num_speakers": load_num_speakers()}) + return 0 + +def _result_row(mp: str, data: dict) -> dict: + words = data.get("words", []) + preview = (data.get("text", "") or "")[:160] + speakers = data.get("speakers") or [] + return { + "media": Path(mp).name, + "language": data.get("language", "?"), + "words": len(words), + "duration": float(data.get("duration", 0.0)), + "preview": preview, + "saved": str(_transcript_json_path(mp)), + "speakers": [s.get("name", s.get("id", "")) for s in speakers], + } diff --git a/admin/api/voice.py b/admin/api/voice.py new file mode 100644 index 0000000..7b28e9c --- /dev/null +++ b/admin/api/voice.py @@ -0,0 +1,212 @@ +"""Análise de voz e aplicação das decisões de edição. + +Extraído de models_api.py — a tabela de comandos segue lá. +""" + +from __future__ import annotations + +import asyncio +import json +import sys +from pathlib import Path + +# code/ is the package root for fcpxml and server modules. +_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") +if _CODE_DIR not in sys.path: + sys.path.insert(0, _CODE_DIR) + +from fcpxml.model_manager import ( + load_hf_token, + load_num_speakers, + load_selected_model, + load_transcript_language, + load_voice_analysis_config, + save_voice_analysis_config, +) + +from . import shared +from .shared import ( + _load_cached_voice_timeline, + _project_media_paths, + _transcript_json_path, + _voice_timeline_json_path, +) +from .shared import _load_cached_transcript # noqa: E402 + + +def cmd_analyze_voice(args: dict) -> int: + """Build the voice timeline (transcript+diarization+acoustics -> emphasis) + for every unique source media in the project, so `refine_voice_timeline` + and friends have something to read without ever reopening the audio. + + Analysis only — writes _voice_timeline.json next to each media, doesn't + touch the project XML. `path` passes through unchanged so it composes + with the other batch steps (silence removal, captions) regardless of + where in the list it runs. + """ + path = str(args.get("path", "")) + if not path or not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + + model = str(args.get("model", "") or load_selected_model() or "") + language = args.get("language") + if language is None: + language = load_transcript_language() + if language == "auto": + language = None + token = str(args.get("hf_token") or load_hf_token() or "") + num_speakers = str(args.get("num_speakers") or load_num_speakers() or "") + + try: + media_paths = _project_media_paths(path) + except Exception as exc: + shared.emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"}) + return 1 + if not media_paths: + shared.emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."}) + return 1 + + from server import handle_build_voice_timeline + + messages: list[str] = [] + output_dir = str(args.get("output_dir") or "").strip() + existing: list[Path] = [] + for mp in media_paths: + timeline_path = _voice_timeline_json_path(mp, output_dir) + if _load_cached_voice_timeline(timeline_path, mp) is not None: + existing.append(timeline_path) + if existing and len(existing) == len(media_paths) and not bool(args.get("force_reprocess", False)): + message = "# Voice Timeline Cache\n\n" + message += "Reaproveitando análise de voz existente. Nada foi reprocessado.\n\n" + for timeline_path in existing: + message += f"- **Timeline JSON**: {timeline_path}\n" + shared.emit({ + "ok": True, + "path": path, + "reused": True, + "timelines": [str(p) for p in existing], + "message": message, + }) + return 0 + + for mp in media_paths: + transcript_path = _transcript_json_path(mp, output_dir) + reused_prefix = "" + if _load_cached_transcript(transcript_path) is not None: + reused_prefix = f"# Cache\n\nReaproveitando transcrição existente: `{transcript_path}`\n\n" + try: + contents = asyncio.run(handle_build_voice_timeline({ + "media_path": mp, "model": model, "language": language, + "hf_token": token, "num_speakers": num_speakers, + "output_dir": output_dir, + })) + except Exception as exc: + shared.emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"}) + return 1 + messages.append(reused_prefix + "\n".join(getattr(c, "text", str(c)) for c in contents)) + + shared.emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)}) + return 0 + +def cmd_acoustics_capability(args: dict) -> int: + """Whether librosa (pitch/energy extraction) is installed in this venv. + + Surfaces `features_capability()` — previously computed but never + exposed to the app, so `layers.acoustics: false` in a voice timeline + had no explanation the user could act on. + """ + from fcpxml.voice_features import features_capability + ok, msg = features_capability() + shared.emit({"ok": True, "available": ok, "message": msg}) + return 0 + +def cmd_voice_analysis(args: dict) -> int: + """Read the persisted voice-analysis settings (energy/emphasis/emotion).""" + config = load_voice_analysis_config() + shared.emit({"ok": True, **config, "emphasis_threshold": config["emphasis_floor"]}) + return 0 + +def cmd_set_voice_analysis(args: dict) -> int: + """Persist voice-analysis settings. Only the given fields change.""" + weights = args.get("emphasis_weights") + config = save_voice_analysis_config( + energy_threshold=args.get("energy_threshold"), + emphasis_weights=weights if isinstance(weights, dict) else None, + emphasis_floor=args.get("emphasis_threshold"), + emotion_enabled=args.get("emotion_enabled"), + emotion_sensitivity=args.get("emotion_sensitivity"), + zoom_scale=args.get("zoom_scale"), + zoom_mode=args.get("zoom_mode"), + zoom_ease_in=args.get("zoom_ease_in"), + zoom_ease_out=args.get("zoom_ease_out"), + ) + shared.emit({"ok": True, **config}) + return 0 + +def cmd_apply_voice_actions(args: dict) -> int: + """Apply a decision list (cuts/zooms/texts/markers) to the project XML. + + The list is produced by a model reading the _voice_timeline.json — this + is the step that turns those decisions into an edit, and the one the + batch chain was missing: without it the app could measure the voice and + caption the result, but never cut by it. + + `actions_path` points at the JSON; either a bare list or the + ``{"actions": [...]}`` wrapper the skill emits is accepted. Times stay in + ORIGINAL source seconds — the handler resolves cuts first and shifts + everything else itself. + """ + path = str(args.get("path", "")) + if not path or not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + + actions = args.get("actions") + if actions is None: + actions_path = str(args.get("actions_path", "")) + if not actions_path or not Path(actions_path).exists(): + shared.emit({"ok": False, "error": "Arquivo de decisões (JSON) não encontrado."}) + return 1 + try: + with open(actions_path, encoding="utf-8") as fh: + loaded = json.load(fh) + except (OSError, ValueError) as exc: + shared.emit({"ok": False, "error": f"Erro ao ler as decisões: {exc}"}) + return 1 + actions = loaded.get("actions") if isinstance(loaded, dict) else loaded + + # The documented output format is {"source": ..., "actions": [...]} — + # callers passing that whole object inline (e.g. the wizard pasting the + # skill's JSON verbatim) need the same unwrap the actions_path branch + # above already does, or a well-formed payload gets rejected as + # "malformed" for having one extra layer of nesting. + if isinstance(actions, dict): + actions = actions.get("actions") + + if not isinstance(actions, list) or not actions: + shared.emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."}) + return 1 + + from server import handle_apply_voice_actions + + try: + contents = asyncio.run(handle_apply_voice_actions({ + "filepath": path, + "actions": actions, + "output_dir": args.get("output_dir"), + })) + except Exception as exc: + shared.emit({"ok": False, "error": f"Falha ao aplicar as decisões: {exc}"}) + return 1 + + message = "\n".join(getattr(c, "text", str(c)) for c in contents) + # The handler reports dropped/rejected actions individually; hand the + # whole report back so the app can surface them instead of only the count. + out_path = path + for line in message.splitlines(): + if line.startswith("- **Saved to**:"): + out_path = line.split("`")[1] if "`" in line else path + break + shared.emit({"ok": True, "path": out_path, "message": message}) + return 0 diff --git a/admin/api/zoom.py b/admin/api/zoom.py new file mode 100644 index 0000000..6abe4d2 --- /dev/null +++ b/admin/api/zoom.py @@ -0,0 +1,114 @@ +"""Zoom (punch-in): por janela, por clipe e por trecho da transcrição. + +Extraído de models_api.py — a tabela de comandos segue lá. +""" + +from __future__ import annotations + +import asyncio +import sys +from pathlib import Path + +# code/ is the package root for fcpxml and server modules. +_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") +if _CODE_DIR not in sys.path: + sys.path.insert(0, _CODE_DIR) + + +from . import shared +from .shared import ( + _derived_output, + _transcript_json_path, +) +from .shared import _load_cached_transcript # noqa: E402 + + +def cmd_add_zoom(args: dict) -> int: + """Add an ease-in/ease-out punch-in zoom to one clip.""" + path = str(args.get("path", "")) + clip_id = str(args.get("clip_id", "")).strip() + if not path or not Path(path).exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + if not clip_id: + shared.emit({"ok": False, "error": "Informe o nome do clipe."}) + return 1 + try: + from server import handle_add_zoom + + output = _derived_output(path, "_zoom", args) + contents = asyncio.run(handle_add_zoom({**args, "filepath": path, "output_path": output})) + message = "\n".join(getattr(content, "text", str(content)) for content in contents) + if not Path(output).exists(): + shared.emit({"ok": False, "error": message}) + return 1 + shared.emit({"ok": True, "path": output, "message": message}) + return 0 + except Exception as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 + +def cmd_zoom_clips(args: dict) -> int: + """Return timeline clips with enough identity for the zoom picker.""" + path = Path(str(args.get("path", ""))) + output_dir = str(args.get("output_dir", "")).strip() + if not path.exists(): + shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + try: + from server import _require_timeline + + _, timeline = _require_timeline(str(path)) + clips = [] + for index, clip in enumerate(timeline.clips): + media = clip.media_path or "" + cached = _load_cached_transcript(_transcript_json_path(media, output_dir)) if media else None + clips.append({ + "id": f"{index}:{clip.start.seconds:.6f}", + "index": index, + "name": clip.name, + "start": clip.start.seconds, + "duration": clip.duration_seconds, + "media": Path(media).name if media else "", + "preview": ((cached or {}).get("text", "") or "")[:180], + "has_transcript": cached is not None, + }) + shared.emit({"ok": True, "clips": clips}) + return 0 + except Exception as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 + +def cmd_zoom_segments(args: dict) -> int: + """Return sentence/word ranges for one timeline clip.""" + path = Path(str(args.get("path", ""))) + output_dir = str(args.get("output_dir", "")).strip() + try: + from server import _require_timeline + + _, timeline = _require_timeline(str(path)) + index = int(args.get("index", -1)) + if index < 0 or index >= len(timeline.clips): + raise ValueError("Clipe selecionado não existe.") + clip = timeline.clips[index] + if not clip.media_path: + raise ValueError("Este clipe não possui mídia associada.") + data = _load_cached_transcript(_transcript_json_path(clip.media_path, output_dir)) + if data is None: + shared.emit({"ok": True, "segments": [], "message": "Transcreva este clipe primeiro."}) + return 0 + segments = [] + for number, segment in enumerate(data.get("segments", [])): + text = str(segment.get("text", "")).strip() + if text: + segments.append({ + "id": number, + "start": float(segment.get("start", 0)), + "end": float(segment.get("end", 0)), + "text": text, + }) + shared.emit({"ok": True, "segments": segments}) + return 0 + except Exception as exc: + shared.emit({"ok": False, "error": str(exc)}) + return 1 diff --git a/admin/models_api.py b/admin/models_api.py index 5601a8f..adac3a5 100644 --- a/admin/models_api.py +++ b/admin/models_api.py @@ -153,1180 +153,27 @@ Commands: Exit code 0 on success, 1 on error. """ -from __future__ import annotations - -import asyncio import json -import os -import shutil -import subprocess import sys -import threading from pathlib import Path -from typing import Any -# code/ is the package root for fcpxml and server modules. -_CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") -if _CODE_DIR not in sys.path: - sys.path.insert(0, _CODE_DIR) +# `admin/` não é um pacote instalado — a raiz do repositório precisa estar no +# caminho para `admin.api` resolver quando o app roda este arquivo por path. +_REPO_ROOT = str(Path(__file__).resolve().parent.parent) +if _REPO_ROOT not in sys.path: + sys.path.insert(0, _REPO_ROOT) -from fcpxml.diarize import ( # noqa: E402 - assign_speakers, - build_speakers, - diarization_capability, - diarize, +from admin.api import ( + editing, + models, + project, + review, + subtitles, + transcription, + voice, + zoom, ) -from fcpxml.media_intel import media_src_to_path # noqa: E402 -from fcpxml.model_manager import ( # noqa: E402 - download_model, - get_models_dir, - is_model_downloaded, - list_installed_models, - load_catalog, - load_dynamic_subtitle_config, - load_hf_token, - load_num_speakers, - load_plain_subtitle_config, - load_project_config, - load_selected_model, - load_silence_config, - load_transcript_language, - load_voice_analysis_config, - model_cache_dir, - save_dynamic_subtitle_config, - save_hf_token, - save_models_dir, - save_num_speakers, - save_plain_subtitle_config, - save_project_config, - save_selected_model, - save_silence_config, - save_transcript_language, - save_voice_analysis_config, -) -from fcpxml.parser import parse_fcpxml # noqa: E402 -from fcpxml.transcribe import transcribe # noqa: E402 -from fcpxml.writer import FCPXMLModifier # noqa: E402 - -RECOMMENDED = ("large-v3", "distil-large-v3", "small", "base") - - -def _derived_output(path: str, suffix: str, args: dict) -> str: - """Resolve a derived XML path, optionally inside the chosen output folder.""" - output_dir = str(args.get("output_dir", "")).strip() - if output_dir: - directory = Path(output_dir).expanduser() - directory.mkdir(parents=True, exist_ok=True) - source = Path(path) - extension = ".fcpxmld" if source.is_dir() else source.suffix - return str(directory / f"{source.stem}{suffix}{extension}") - from server import generate_output_path - return generate_output_path(path, suffix) - - -def _is_no_change_message(message: str) -> bool: - """Whether a tool completed cleanly without needing to save a new file.""" - text = message.lower() - return any( - token in text - for token in ( - "no cuts to make", - "no silence", - "file unchanged", - "nothing saved", - ) - ) - - -def _emit_no_change_or_error(path: str, message: str) -> int: - if _is_no_change_message(message): - _emit({"ok": True, "path": path, "unchanged": True, "message": message}) - return 0 - _emit({"ok": False, "error": message}) - return 1 - -# Download cancellation events, keyed by model name. -_CANCEL: dict[str, threading.Event] = {} -_LOCK = threading.Lock() - - -def _emit(obj: Any) -> None: - sys.stdout.write(json.dumps(obj, ensure_ascii=False) + "\n") - sys.stdout.flush() - - -def _transcript_json_path(media_path: str, output_dir: str = "") -> Path: - """Where the ``_transcript.json`` for ``media_path`` lives. - - When ``output_dir`` (the user-selected project folder) is set, the - transcript is saved/read there — never next to the source media, which - may sit on a read-only volume or a Final Cut Library the user never - browses. Falls back to the media's own folder only when no project - folder has been chosen (legacy/MCP callers). - """ - p = Path(media_path) - if output_dir: - directory = Path(output_dir).expanduser() - directory.mkdir(parents=True, exist_ok=True) - return directory / f"{p.stem}_transcript.json" - return p.with_name(p.stem + "_transcript.json") - - -def _save_json_atomic(path: Path, data: Any) -> None: - """Write ``data`` to ``path`` atomically and validate the result on disk. - - Mirrors the reference WHISPERX save path: write a ``.tmp``, ``os.replace`` - into place, then confirm the file exists, is non-empty, and parses as JSON. - """ - tmp_path = str(path) + ".tmp" - with open(tmp_path, "w", encoding="utf-8") as fh: - json.dump(data, fh, ensure_ascii=False, indent=2) - os.replace(tmp_path, path) - if not path.exists() or os.path.getsize(path) == 0: - raise RuntimeError("O arquivo salvo está vazio ou não foi encontrado.") - with open(path, encoding="utf-8") as fh: - json.load(fh) - - -def _project_media_paths(path: str) -> list[str]: - proj = parse_fcpxml(path) - tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None) - media_paths: list[str] = [] - if tl is not None: - for clip in getattr(tl, "clips", []): - mp = media_src_to_path(clip.media_path or "") - if mp and Path(mp).is_file() and mp not in media_paths: - media_paths.append(mp) - return media_paths - - -def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path: - p = Path(media_path) - if output_dir: - directory = Path(output_dir).expanduser() - directory.mkdir(parents=True, exist_ok=True) - return directory / f"{p.stem}_voice_timeline.json" - return p.with_name(p.stem + "_voice_timeline.json") - - -def _load_cached_voice_timeline(json_path: Path, media_path: str) -> dict | None: - try: - with open(json_path, encoding="utf-8") as fh: - data = json.load(fh) - except (OSError, json.JSONDecodeError, UnicodeDecodeError): - return None - if not isinstance(data, dict): - return None - if data.get("source") != Path(media_path).name: - return None - if not isinstance(data.get("segments"), list): - return None - return data - - -# ── commands ──────────────────────────────────────────────────────────────── - - -def cmd_catalog() -> None: - catalog = load_catalog() - installed = list_installed_models() - diar_ok, diar_msg = diarization_capability(load_hf_token()) - _emit( - { - "models": catalog, - "installed": installed, - "selected": load_selected_model(), - "language": load_transcript_language(), - "models_dir": str(get_models_dir()), - "installed_count": len(installed), - "recommended": list(RECOMMENDED), - "diarization": diar_ok, - "diarization_message": diar_msg, - "hf_token_set": bool(load_hf_token()), - "num_speakers": load_num_speakers(), - } - ) - - -def cmd_download(args: dict) -> int: - model = str(args.get("model", "")) - if model not in _model_names(): - _emit({"type": "error", "message": f"Modelo desconhecido: {model}"}) - return 1 - ev = threading.Event() - with _LOCK: - _CANCEL[model] = ev - try: - download_model(model, progress_cb=lambda f: _emit({"type": "progress", "fraction": f}), cancel_event=ev) - installed = is_model_downloaded(model) - _emit({"type": "done", "installed": installed}) - if installed: - save_selected_model(model) - return 0 if installed else 1 - except Exception as exc: - _emit({"type": "error", "message": str(exc)}) - return 1 - finally: - with _LOCK: - _CANCEL.pop(model, None) - - -def cmd_cancel(args: dict) -> None: - model = str(args.get("model", "")) - ev = _CANCEL.get(model) - if ev is not None: - ev.set() - _emit({"ok": True}) - - -def cmd_select(args: dict) -> None: - model = str(args.get("model", "")) - if not is_model_downloaded(model): - _emit({"ok": False, "error": "Modelo não está instalado."}) - return - save_selected_model(model) - _emit({"ok": True, "selected": load_selected_model()}) - - -def cmd_set_language(args: dict) -> int: - """Persist the transcription language (the default for every transcription).""" - lang = str(args.get("language", "auto")) - try: - saved = save_transcript_language(lang) - except ValueError as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 - _emit({"ok": True, "language": saved}) - return 0 - - -def cmd_delete(args: dict) -> None: - model = str(args.get("model", "")) - try: - shutil.rmtree(model_cache_dir(model), ignore_errors=True) - except Exception: - pass - _emit({"ok": True}) - - -def cmd_open_finder(args: dict) -> None: - target = str(args.get("path") or model_cache_dir(str(args.get("model", "")))) - try: - subprocess.Popen(["open", target]) - except OSError: - pass - _emit({"ok": True}) - - -def cmd_remove_silences(args: dict) -> int: - """Run the canonical server silence remover into a suffixed copy.""" - path = str(args.get("path", "")) - if not path or not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - try: - from server import handle_remove_media_silence - - output = _derived_output(path, "_silence_removed", args) - contents = asyncio.run(handle_remove_media_silence({**args, "filepath": path, "output_path": output})) - message = "\n".join(getattr(content, "text", str(content)) for content in contents) - if not Path(output).exists(): - return _emit_no_change_or_error(path, message) - _emit({"ok": True, "path": output, "message": message}) - return 0 - except Exception as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 - - -def cmd_edit_by_transcript(args: dict) -> int: - """Cut (or keep only) spoken phrases, using each media's cached transcript.""" - path = str(args.get("path", "")) - phrases = args.get("phrases") or [] - if not path or not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - if not isinstance(phrases, list) or not [p for p in phrases if str(p).strip()]: - _emit({"ok": False, "error": "Informe ao menos uma frase para cortar."}) - return 1 - try: - from server import handle_edit_by_transcript - - output = _derived_output(path, "_transcript_edit", args) - contents = asyncio.run(handle_edit_by_transcript({**args, "filepath": path, "output_path": output})) - message = "\n".join(getattr(content, "text", str(content)) for content in contents) - if not Path(output).exists(): - _emit({"ok": False, "error": message}) - return 1 - _emit({"ok": True, "path": output, "message": message}) - return 0 - except Exception as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 - - -def cmd_remove_filler_words(args: dict) -> int: - """Cut filler words (um, uh, ...) out, using each media's cached transcript.""" - path = str(args.get("path", "")) - if not path or not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - try: - from server import handle_remove_filler_words - - output = _derived_output(path, "_defillered", args) - contents = asyncio.run(handle_remove_filler_words({**args, "filepath": path, "output_path": output})) - message = "\n".join(getattr(content, "text", str(content)) for content in contents) - if not Path(output).exists(): - return _emit_no_change_or_error(path, message) - _emit({"ok": True, "path": output, "message": message}) - return 0 - except Exception as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 - - -def cmd_transcript_markers(args: dict) -> int: - """Add a marker per transcribed segment, using each media's cached transcript.""" - path = str(args.get("path", "")) - if not path or not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - try: - from server import handle_transcript_markers - - output = _derived_output(path, "_transcript_markers", args) - contents = asyncio.run(handle_transcript_markers({**args, "filepath": path, "output_path": output})) - message = "\n".join(getattr(content, "text", str(content)) for content in contents) - if not Path(output).exists(): - _emit({"ok": False, "error": message}) - return 1 - _emit({"ok": True, "path": output, "message": message}) - return 0 - except Exception as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 - - -def cmd_generate_dynamic_subtitles(args: dict) -> int: - """Generate word-by-word ("karaoke") caption compound clips, one per line, - using each media's cached transcript.""" - path = str(args.get("path", "")) - if not path or not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - try: - from server import handle_generate_dynamic_subtitles - - output = _derived_output(path, "_dynamic_subtitles", args) - contents = asyncio.run( - handle_generate_dynamic_subtitles({**args, "filepath": path, "output_path": output}) - ) - message = "\n".join(getattr(content, "text", str(content)) for content in contents) - if not Path(output).exists(): - _emit({"ok": False, "error": message}) - return 1 - _emit({"ok": True, "path": output, "message": message}) - return 0 - except Exception as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 - - -def cmd_generate_plain_subtitles(args: dict) -> int: - """Generate simple static editable subtitle title clips.""" - path = str(args.get("path", "")) - if not path or not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - try: - from server import handle_generate_plain_subtitles - - output = _derived_output(path, "_plain_subtitles", args) - contents = asyncio.run( - handle_generate_plain_subtitles({**args, "filepath": path, "output_path": output}) - ) - message = "\n".join(getattr(content, "text", str(content)) for content in contents) - if not Path(output).exists(): - return _emit_no_change_or_error(path, message) - _emit({"ok": True, "path": output, "message": message}) - return 0 - except Exception as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 - - -def cmd_add_zoom(args: dict) -> int: - """Add an ease-in/ease-out punch-in zoom to one clip.""" - path = str(args.get("path", "")) - clip_id = str(args.get("clip_id", "")).strip() - if not path or not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - if not clip_id: - _emit({"ok": False, "error": "Informe o nome do clipe."}) - return 1 - try: - from server import handle_add_zoom - - output = _derived_output(path, "_zoom", args) - contents = asyncio.run(handle_add_zoom({**args, "filepath": path, "output_path": output})) - message = "\n".join(getattr(content, "text", str(content)) for content in contents) - if not Path(output).exists(): - _emit({"ok": False, "error": message}) - return 1 - _emit({"ok": True, "path": output, "message": message}) - return 0 - except Exception as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 - - -def cmd_zoom_clips(args: dict) -> int: - """Return timeline clips with enough identity for the zoom picker.""" - path = Path(str(args.get("path", ""))) - output_dir = str(args.get("output_dir", "")).strip() - if not path.exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - try: - from server import _require_timeline - - _, timeline = _require_timeline(str(path)) - clips = [] - for index, clip in enumerate(timeline.clips): - media = clip.media_path or "" - cached = _load_cached_transcript(_transcript_json_path(media, output_dir)) if media else None - clips.append({ - "id": f"{index}:{clip.start.seconds:.6f}", - "index": index, - "name": clip.name, - "start": clip.start.seconds, - "duration": clip.duration_seconds, - "media": Path(media).name if media else "", - "preview": ((cached or {}).get("text", "") or "")[:180], - "has_transcript": cached is not None, - }) - _emit({"ok": True, "clips": clips}) - return 0 - except Exception as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 - - -def cmd_zoom_segments(args: dict) -> int: - """Return sentence/word ranges for one timeline clip.""" - path = Path(str(args.get("path", ""))) - output_dir = str(args.get("output_dir", "")).strip() - try: - from server import _require_timeline - - _, timeline = _require_timeline(str(path)) - index = int(args.get("index", -1)) - if index < 0 or index >= len(timeline.clips): - raise ValueError("Clipe selecionado não existe.") - clip = timeline.clips[index] - if not clip.media_path: - raise ValueError("Este clipe não possui mídia associada.") - data = _load_cached_transcript(_transcript_json_path(clip.media_path, output_dir)) - if data is None: - _emit({"ok": True, "segments": [], "message": "Transcreva este clipe primeiro."}) - return 0 - segments = [] - for number, segment in enumerate(data.get("segments", [])): - text = str(segment.get("text", "")).strip() - if text: - segments.append({ - "id": number, - "start": float(segment.get("start", 0)), - "end": float(segment.get("end", 0)), - "text": text, - }) - _emit({"ok": True, "segments": segments}) - return 0 - except Exception as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 -def cmd_set_models_dir(args: dict) -> int: - try: - d = save_models_dir(str(args.get("dir", ""))) - _emit({"ok": True, "models_dir": d}) - return 0 - except ValueError as exc: - _emit({"ok": False, "error": str(exc)}) - return 1 - - -def cmd_inspect(args: dict) -> int: - """Validate an FCPXML file and return a summary of its projects/timelines.""" - path = str(args.get("path", "")) - if not path: - _emit({"ok": False, "error": "Nenhum arquivo informado."}) - return 1 - if not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo não encontrado."}) - return 1 - try: - proj = parse_fcpxml(path) - except Exception as exc: - _emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"}) - return 1 - - timelines = [] - for tl in proj.timelines: - timelines.append( - { - "name": tl.name, - "duration_seconds": round(tl.duration.seconds, 3), - "frame_rate": round(tl.frame_rate, 3), - "width": tl.width, - "height": tl.height, - "clips": tl.total_clips, - "cuts": tl.total_cuts, - "connected": len(tl.connected_clips), - "markers": len(tl.markers), - } - ) - _emit( - { - "ok": True, - "path": path, - "name": proj.name, - "fcpxml_version": proj.fcpxml_version, - "timelines": timelines, - } - ) - return 0 - - -def cmd_transcribe(args: dict) -> int: - proj_path = str(args.get("path", "")) - output_dir = str(args.get("output_dir", "")).strip() - # Honra o modelo selecionado no programa quando nenhum é passado. - model = str(args.get("model", "") or load_selected_model() or "") - language = args.get("language") - if language is None: - language = load_transcript_language() - if language == "auto": - language = None - if not proj_path: - _emit({"type": "error", "message": "Nenhum projeto selecionado."}) - return 1 - if not output_dir: - _emit({"type": "error", "message": "Selecione a pasta do projeto antes de transcrever."}) - return 1 - if not model or not is_model_downloaded(model): - _emit( - { - "type": "error", - "message": "Nenhum modelo de transcrição instalado. Baixe e selecione um modelo na aba Modelos.", - } - ) - return 1 - - token = str(args.get("hf_token") or load_hf_token() or "") - if args.get("num_speakers") is not None: - num_speakers = str(args.get("num_speakers")) - else: - num_speakers = load_num_speakers() - - # Load project. - try: - proj = parse_fcpxml(proj_path) - except Exception as exc: - _emit({"type": "error", "message": f"Erro ao ler o projeto: {exc}"}) - return 1 - tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None) - media_paths: list[str] = [] - if tl is not None: - for clip in getattr(tl, "clips", []): - mp = media_src_to_path(clip.media_path or "") - if mp and Path(mp).is_file() and mp not in media_paths: - media_paths.append(mp) - if not media_paths: - _emit({"type": "error", "message": "Nenhum arquivo de mídia acessível encontrado."}) - return 1 - - total = len(media_paths) - results: list[dict] = [] - for i, mp in enumerate(media_paths, 1): - stage = f"Transcrevendo {Path(mp).name} ({i}/{total})…" - _emit({"type": "progress", "fraction": (i - 1) / total, "stage": stage}) - json_path = _transcript_json_path(mp, output_dir) - cached = _load_cached_transcript(json_path) - if cached is not None: - _emit({"type": "progress", "fraction": i / total, "stage": stage}) - results.append(_result_row(mp, cached)) - continue - - def _on_progress(file_fraction: float, _i: int = i, _stage: str = stage) -> None: - # Blend this file's own progress into the overall fraction so a - # single-media project doesn't jump straight to 100% before the - # actual (slow) decoding work has even started. - overall = (_i - 1 + file_fraction) / total - _emit({"type": "progress", "fraction": overall, "stage": _stage}) - - data = transcribe(mp, model_size=model, language=language, progress_cb=_on_progress) - if data is None: - _emit({"type": "error", "message": f"Não foi possível transcrever: {Path(mp).name}"}) - return 1 - - # Diarização opcional (necessita token HF): assina speaker por segmento/palavra. - if token: - tracks = diarize(mp, token, num_speakers) - segments, words = assign_speakers( - data.get("segments", []), data.get("words", []), tracks - ) - data = {**data, "segments": segments, "words": words} - data["speakers"] = build_speakers(data.get("segments", [])) - - payload = { - "schema_version": "1.0", - "source": Path(mp).name, - "model": model, - **data, - } - try: - _save_json_atomic(json_path, payload) - except (OSError, RuntimeError, ValueError) as exc: - _emit({"type": "error", "message": f"Não foi possível salvar o JSON: {exc}"}) - return 1 - results.append(_result_row(mp, data)) - - _emit({"type": "result", "transcripts": results}) - return 0 - - -def cmd_analyze_voice(args: dict) -> int: - """Build the voice timeline (transcript+diarization+acoustics -> emphasis) - for every unique source media in the project, so `refine_voice_timeline` - and friends have something to read without ever reopening the audio. - - Analysis only — writes _voice_timeline.json next to each media, doesn't - touch the project XML. `path` passes through unchanged so it composes - with the other batch steps (silence removal, captions) regardless of - where in the list it runs. - """ - path = str(args.get("path", "")) - if not path or not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - - model = str(args.get("model", "") or load_selected_model() or "") - language = args.get("language") - if language is None: - language = load_transcript_language() - if language == "auto": - language = None - token = str(args.get("hf_token") or load_hf_token() or "") - num_speakers = str(args.get("num_speakers") or load_num_speakers() or "") - - try: - media_paths = _project_media_paths(path) - except Exception as exc: - _emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"}) - return 1 - if not media_paths: - _emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."}) - return 1 - - from server import handle_build_voice_timeline - - messages: list[str] = [] - output_dir = str(args.get("output_dir") or "").strip() - existing: list[Path] = [] - for mp in media_paths: - timeline_path = _voice_timeline_json_path(mp, output_dir) - if _load_cached_voice_timeline(timeline_path, mp) is not None: - existing.append(timeline_path) - if existing and len(existing) == len(media_paths) and not bool(args.get("force_reprocess", False)): - message = "# Voice Timeline Cache\n\n" - message += "Reaproveitando análise de voz existente. Nada foi reprocessado.\n\n" - for timeline_path in existing: - message += f"- **Timeline JSON**: {timeline_path}\n" - _emit({ - "ok": True, - "path": path, - "reused": True, - "timelines": [str(p) for p in existing], - "message": message, - }) - return 0 - - for mp in media_paths: - transcript_path = _transcript_json_path(mp, output_dir) - reused_prefix = "" - if _load_cached_transcript(transcript_path) is not None: - reused_prefix = f"# Cache\n\nReaproveitando transcrição existente: `{transcript_path}`\n\n" - try: - contents = asyncio.run(handle_build_voice_timeline({ - "media_path": mp, "model": model, "language": language, - "hf_token": token, "num_speakers": num_speakers, - "output_dir": output_dir, - })) - except Exception as exc: - _emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"}) - return 1 - messages.append(reused_prefix + "\n".join(getattr(c, "text", str(c)) for c in contents)) - - _emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)}) - return 0 - - -def cmd_export_srt(args: dict) -> int: - """Write a captions .srt synced to the edited timeline. - - Each transcribed segment is mapped from its SOURCE-media timestamp to its - real TIMELINE position (``clip_offset + (seg_start - clip_source_start)``), - so captions only cover the frames that remain after cuts/silence removal — - not the whole source file. One .srt is produced per media, in timeline order. - """ - path = str(args.get("path", "")) - output_dir = str(args.get("output_dir", "")).strip() - if not path or not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - try: - modifier = FCPXMLModifier(path) - except Exception as exc: - _emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"}) - return 1 - - # Group spine clips by media so each transcript is loaded once. - by_media: dict[str, list] = {} - for _, el in modifier._iter_spine_clips(): - src = modifier.resources.get(el.get("ref", ""), {}).get("src", "") - mp = media_src_to_path(src) - if not mp or not Path(mp).is_file(): - continue - by_media.setdefault(mp, []).append(el) - - # Never emit a caption past the end of the project — Final Cut rejects an - # SRT whose last cue overruns the timeline ("subtitle extends beyond project - # duration"). Clamp every mapped cue end to this ceiling. - timeline_total = modifier._timeline_duration().to_seconds() - - srt_paths: list[str] = [] - for mp, clips in by_media.items(): - cached = _load_cached_transcript(_transcript_json_path(mp, output_dir)) - if cached is None: - continue - segments = cached.get("segments") or [] - if not segments: - continue - - rows: list[tuple[float, float, str, int]] = [] - for el in clips: - clip_source_start = modifier.source_file_start(el).to_seconds() - clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() - clip_offset = modifier._parse_time(el.get("offset", "0s")).to_seconds() - window_end = clip_source_start + clip_duration - for seg_index, seg in enumerate(segments): - seg_start = float(seg.get("start", 0.0)) - seg_end = float(seg.get("end", seg_start)) - text = seg.get("text", "").strip() - if not text or seg_end <= seg_start: - continue - # Intersect the complete source segment with this kept clip. - # Testing only seg_start loses speech whose first words fall in - # a removed range; interval intersection preserves the part - # that remains and avoids duplicating a segment wholesale. - source_start = max(seg_start, clip_source_start) - source_end = min(seg_end, window_end) - if source_end <= source_start: - continue - tl_start = clip_offset + (source_start - clip_source_start) - tl_end = clip_offset + (source_end - clip_source_start) - tl_start = max(0.0, min(tl_start, timeline_total)) - tl_end = max(0.0, min(tl_end, timeline_total)) - if tl_end > tl_start: - rows.append((tl_start, tl_end, text, seg_index)) - - if not rows: - continue - rows.sort(key=lambda r: (r[0], r[1], r[3])) - # Merge only pieces from the same original Whisper segment when their - # mapped intervals touch. Never merge unrelated speech or invent time. - merged: list[tuple[float, float, str, int]] = [] - for row in rows: - if merged and row[3] == merged[-1][3] and row[0] <= merged[-1][1] + 0.001: - prev = merged[-1] - merged[-1] = (prev[0], max(prev[1], row[1]), prev[2], prev[3]) - else: - merged.append(row) - - blocks = [] - for index, (s, e, text, _) in enumerate(merged, 1): - start_stamp = srt_stamp(s) - end_stamp = srt_stamp(e) - # Millisecond SRT precision can collapse a sub-millisecond span; - # omit it rather than emit an invalid zero-duration cue. - if start_stamp == end_stamp: - continue - blocks.append(f"{index}\n{start_stamp} --> {end_stamp}\n{text}\n") - if not blocks: - continue - - out = ( - Path(output_dir).expanduser() / f"{Path(mp).stem}_captions.srt" - if output_dir - else Path(mp).with_name(Path(mp).stem + "_captions.srt") - ) - if output_dir: - out.parent.mkdir(parents=True, exist_ok=True) - try: - out.write_text("\n".join(blocks), encoding="utf-8") - except OSError as exc: - _emit({"ok": False, "error": f"Não foi possível salvar a legenda: {exc}"}) - return 1 - srt_paths.append(str(out)) - - if not srt_paths: - _emit({"ok": False, "error": "Nenhuma transcrição encontrada. Transcreva o projeto primeiro."}) - return 1 - - _emit({"ok": True, "paths": srt_paths, "message": f"{len(srt_paths)} legenda(s) .srt sincronizada(s) com o corte."}) - return 0 - - -def srt_stamp(seconds: float) -> str: - """Format float seconds as ``HH:MM:SS,mmm`` (SRT uses a comma). - - Uses ``floor`` (not ``round``) so a timestamp never rounds up past a frame - boundary — an SRT cue ending on the last frame must not overrun the - project duration, or Final Cut flags it as extending beyond the project. - """ - ms = int((seconds if seconds > 0 else 0.0) * 1000) - h, rem = divmod(ms, 3600000) - m, rem = divmod(rem, 60000) - s, ms = divmod(rem, 1000) - return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}" - - -def _load_cached_transcript(json_path: Path) -> dict | None: - """Return a valid cached transcript dict, or ``None`` if absent/unreadable.""" - if not json_path.is_file(): - return None - try: - data = json.loads(json_path.read_text(encoding="utf-8")) - except (OSError, ValueError): - return None - if isinstance(data, dict) and isinstance(data.get("words"), list): - if "speakers" not in data: - data["speakers"] = build_speakers(data.get("segments", [])) - return data - return None - - -def cmd_rename_speakers(args: dict) -> int: - """Apply real names to speakers already saved in a transcript JSON.""" - json_path = Path(str(args.get("path", ""))) - names = args.get("speakers") or {} - if not json_path.is_file(): - _emit({"type": "error", "message": "Transcrição não encontrada."}) - return 1 - try: - data = json.loads(json_path.read_text(encoding="utf-8")) - except (OSError, ValueError) as exc: - _emit({"type": "error", "message": f"Não foi possível ler o JSON: {exc}"}) - return 1 - mapping = {str(sid): str(name).strip() for sid, name in (names or {}).items()} - for sp in data.get("speakers", []): - sid = str(sp.get("id", "")) - if sid in mapping and mapping[sid]: - sp["name"] = mapping[sid] - try: - _save_json_atomic(json_path, data) - except (OSError, RuntimeError, ValueError) as exc: - _emit({"type": "error", "message": f"Não foi possível salvar: {exc}"}) - return 1 - _emit({"ok": True, "speakers": data.get("speakers", [])}) - return 0 - - -def cmd_set_diarization(args: dict) -> int: - """Persist the HuggingFace token and expected speaker count for diarization.""" - token = args.get("token") - num = args.get("num_speakers") - if token is not None: - save_hf_token(str(token)) - if num is not None: - save_num_speakers(str(num)) - ok, msg = diarization_capability(load_hf_token()) - _emit({"ok": True, "diarization": ok, "diarization_message": msg, "num_speakers": load_num_speakers()}) - return 0 - - -def cmd_acoustics_capability(args: dict) -> int: - """Whether librosa (pitch/energy extraction) is installed in this venv. - - Surfaces `features_capability()` — previously computed but never - exposed to the app, so `layers.acoustics: false` in a voice timeline - had no explanation the user could act on. - """ - from fcpxml.voice_features import features_capability - ok, msg = features_capability() - _emit({"ok": True, "available": ok, "message": msg}) - return 0 - - -def cmd_voice_analysis(args: dict) -> int: - """Read the persisted voice-analysis settings (energy/emphasis/emotion).""" - config = load_voice_analysis_config() - _emit({"ok": True, **config, "emphasis_threshold": config["emphasis_floor"]}) - return 0 - - -def cmd_set_voice_analysis(args: dict) -> int: - """Persist voice-analysis settings. Only the given fields change.""" - weights = args.get("emphasis_weights") - config = save_voice_analysis_config( - energy_threshold=args.get("energy_threshold"), - emphasis_weights=weights if isinstance(weights, dict) else None, - emphasis_floor=args.get("emphasis_threshold"), - emotion_enabled=args.get("emotion_enabled"), - emotion_sensitivity=args.get("emotion_sensitivity"), - zoom_scale=args.get("zoom_scale"), - zoom_mode=args.get("zoom_mode"), - zoom_ease_in=args.get("zoom_ease_in"), - zoom_ease_out=args.get("zoom_ease_out"), - ) - _emit({"ok": True, **config}) - return 0 - - -def cmd_dynamic_subtitle_config(args: dict) -> int: - """Read the persisted dynamic-subtitle style (font, size, color, layout).""" - _emit({"ok": True, **load_dynamic_subtitle_config()}) - return 0 - - -def cmd_set_dynamic_subtitle_config(args: dict) -> int: - """Persist dynamic-subtitle style fields. Only the given fields change.""" - config = save_dynamic_subtitle_config(**{ - k: args.get(k) for k in ( - "band_height", "block_center_y", "line_gap", "font", "font_size", - "emphasis_font", "emphasis_face", "emphasis_size", - "active_color", "emphasis_color", "text_scale", - ) - }) - _emit({"ok": True, **config}) - return 0 - - -def cmd_plain_subtitle_config(args: dict) -> int: - """Read the persisted simple subtitle style.""" - _emit({"ok": True, **load_plain_subtitle_config()}) - return 0 - - -def cmd_set_plain_subtitle_config(args: dict) -> int: - """Persist simple subtitle style fields. Only the given fields change.""" - config = save_plain_subtitle_config(**{ - k: args.get(k) for k in ( - "font", "font_size", "font_color", "max_words", - "position_y", "uppercase", "keep_punctuation", "text_scale", - ) - }) - _emit({"ok": True, **config}) - return 0 - - -def cmd_silence_config(args: dict) -> int: - """Read the persisted silence thresholds (noise floor, duration, padding).""" - _emit({"ok": True, **load_silence_config()}) - return 0 - - -def cmd_set_silence_config(args: dict) -> int: - """Persist silence thresholds. Only the given fields change.""" - config = save_silence_config( - noise_db=args.get("noise_db"), - min_silence=args.get("min_silence"), - padding=args.get("padding"), - ) - _emit({"ok": True, **config}) - return 0 - - -def cmd_apply_voice_actions(args: dict) -> int: - """Apply a decision list (cuts/zooms/texts/markers) to the project XML. - - The list is produced by a model reading the _voice_timeline.json — this - is the step that turns those decisions into an edit, and the one the - batch chain was missing: without it the app could measure the voice and - caption the result, but never cut by it. - - `actions_path` points at the JSON; either a bare list or the - ``{"actions": [...]}`` wrapper the skill emits is accepted. Times stay in - ORIGINAL source seconds — the handler resolves cuts first and shifts - everything else itself. - """ - path = str(args.get("path", "")) - if not path or not Path(path).exists(): - _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) - return 1 - - actions = args.get("actions") - if actions is None: - actions_path = str(args.get("actions_path", "")) - if not actions_path or not Path(actions_path).exists(): - _emit({"ok": False, "error": "Arquivo de decisões (JSON) não encontrado."}) - return 1 - try: - with open(actions_path, encoding="utf-8") as fh: - loaded = json.load(fh) - except (OSError, ValueError) as exc: - _emit({"ok": False, "error": f"Erro ao ler as decisões: {exc}"}) - return 1 - actions = loaded.get("actions") if isinstance(loaded, dict) else loaded - - # The documented output format is {"source": ..., "actions": [...]} — - # callers passing that whole object inline (e.g. the wizard pasting the - # skill's JSON verbatim) need the same unwrap the actions_path branch - # above already does, or a well-formed payload gets rejected as - # "malformed" for having one extra layer of nesting. - if isinstance(actions, dict): - actions = actions.get("actions") - - if not isinstance(actions, list) or not actions: - _emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."}) - return 1 - - from server import handle_apply_voice_actions - - try: - contents = asyncio.run(handle_apply_voice_actions({ - "filepath": path, - "actions": actions, - "output_dir": args.get("output_dir"), - })) - except Exception as exc: - _emit({"ok": False, "error": f"Falha ao aplicar as decisões: {exc}"}) - return 1 - - message = "\n".join(getattr(c, "text", str(c)) for c in contents) - # The handler reports dropped/rejected actions individually; hand the - # whole report back so the app can surface them instead of only the count. - out_path = path - for line in message.splitlines(): - if line.startswith("- **Saved to**:"): - out_path = line.split("`")[1] if "`" in line else path - break - _emit({"ok": True, "path": out_path, "message": message}) - return 0 - - -def cmd_build_phrase_review(args: dict) -> int: - """Build the reviewable script (phrases + the AI's decisions) for the wizard. - - `voice_timeline` points at the _voice_timeline.json; `actions` carries the - decision list the model returned (inline, in any of the shapes the skill - emits). The review is always rebuilt from the current analysis, then the - decisions saved on a previous visit are laid back over it — reopening the - step must show the edits the user left there without freezing the acoustics - as they were when they left. - """ - from fcpxml.phrase_review import ( - build_phrase_review, - load_phrase_review, - merge_saved_decisions, - ) - - timeline_path = str(args.get("voice_timeline", "")) - if not timeline_path or not Path(timeline_path).exists(): - _emit({"ok": False, "error": "Análise de voz (voice_timeline.json) não encontrada."}) - return 1 - - try: - with open(timeline_path, encoding="utf-8") as fh: - timeline = json.load(fh) - except (OSError, ValueError) as exc: - _emit({"ok": False, "error": f"Erro ao ler a análise de voz: {exc}"}) - return 1 - - extra = [d for d in (args.get("output_dir"), args.get("media_dir")) if d] - review = build_phrase_review( - timeline, - args.get("actions"), - voice_timeline_path=timeline_path, - extra_dirs=extra, - ) - - saved = None if args.get("fresh") else load_phrase_review(timeline_path) - review = merge_saved_decisions(review, saved) - _emit({"ok": True, "reused": saved is not None, **review}) - return 0 - - -def cmd_save_phrase_review(args: dict) -> int: - """Persist the edited review and the actions derived from it.""" - from fcpxml.phrase_review import save_phrase_review - - timeline_path = str(args.get("voice_timeline", "")) - if not timeline_path: - _emit({"ok": False, "error": "Caminho da análise de voz não informado."}) - return 1 - - phrases = args.get("phrases") - if not isinstance(phrases, list): - _emit({"ok": False, "error": "Nenhuma frase para salvar."}) - return 1 - - review = { - "version": args.get("version", "1.0"), - "source": args.get("source", ""), - "duration": args.get("duration", 0.0), - "speakers": args.get("speakers", []), - "phrases": phrases, - "zooms": args.get("zooms", []), - } - try: - review_path, actions_path = save_phrase_review(timeline_path, review) - except OSError as exc: - _emit({"ok": False, "error": f"Erro ao salvar a revisão: {exc}"}) - return 1 - - _emit({ - "ok": True, - "review_path": str(review_path), - "actions_path": str(actions_path), - "emphasis_count": sum(1 for p in phrases if int(p.get("emphasis", 0) or 0) >= 1), - "removed_count": sum(1 for p in phrases if not p.get("active", True)), - }) - return 0 - - -def cmd_project_config(args: dict) -> int: - """Read the last project folder/file the app was working on.""" - _emit({"ok": True, **load_project_config()}) - return 0 - - -def cmd_set_project_config(args: dict) -> int: - """Persist the last project folder/file. Only the given fields change.""" - config = save_project_config(folder=args.get("folder"), file=args.get("file")) - _emit({"ok": True, **config}) - return 0 - - -def _result_row(mp: str, data: dict) -> dict: - words = data.get("words", []) - preview = (data.get("text", "") or "")[:160] - speakers = data.get("speakers") or [] - return { - "media": Path(mp).name, - "language": data.get("language", "?"), - "words": len(words), - "duration": float(data.get("duration", 0.0)), - "preview": preview, - "saved": str(_transcript_json_path(mp)), - "speakers": [s.get("name", s.get("id", "")) for s in speakers], - } - - -def _model_names() -> list[str]: - return [m["internal_name"] for m in load_catalog()] +from admin.api.shared import emit # noqa: F401 def main() -> int: @@ -1342,43 +189,43 @@ def main() -> int: return 1 handlers = { - "catalog": cmd_catalog, - "download": cmd_download, - "cancel": cmd_cancel, - "select": cmd_select, - "set_language": cmd_set_language, - "delete": cmd_delete, - "open_finder": cmd_open_finder, - "set_models_dir": cmd_set_models_dir, - "inspect": cmd_inspect, - "transcribe": cmd_transcribe, - "export_srt": cmd_export_srt, - "remove_silences": cmd_remove_silences, - "edit_by_transcript": cmd_edit_by_transcript, - "remove_filler_words": cmd_remove_filler_words, - "transcript_markers": cmd_transcript_markers, - "generate_dynamic_subtitles": cmd_generate_dynamic_subtitles, - "generate_plain_subtitles": cmd_generate_plain_subtitles, - "add_zoom": cmd_add_zoom, - "zoom_clips": cmd_zoom_clips, - "zoom_segments": cmd_zoom_segments, - "rename_speakers": cmd_rename_speakers, - "set_diarization": cmd_set_diarization, - "acoustics_capability": cmd_acoustics_capability, - "voice_analysis": cmd_voice_analysis, - "set_voice_analysis": cmd_set_voice_analysis, - "analyze_voice": cmd_analyze_voice, - "dynamic_subtitle_config": cmd_dynamic_subtitle_config, - "set_dynamic_subtitle_config": cmd_set_dynamic_subtitle_config, - "plain_subtitle_config": cmd_plain_subtitle_config, - "set_plain_subtitle_config": cmd_set_plain_subtitle_config, - "apply_voice_actions": cmd_apply_voice_actions, - "build_phrase_review": cmd_build_phrase_review, - "save_phrase_review": cmd_save_phrase_review, - "project_config": cmd_project_config, - "set_project_config": cmd_set_project_config, - "silence_config": cmd_silence_config, - "set_silence_config": cmd_set_silence_config, + "catalog": models.cmd_catalog, + "download": models.cmd_download, + "cancel": models.cmd_cancel, + "select": models.cmd_select, + "set_language": models.cmd_set_language, + "delete": models.cmd_delete, + "open_finder": models.cmd_open_finder, + "set_models_dir": models.cmd_set_models_dir, + "inspect": project.cmd_inspect, + "transcribe": transcription.cmd_transcribe, + "export_srt": subtitles.cmd_export_srt, + "remove_silences": editing.cmd_remove_silences, + "edit_by_transcript": editing.cmd_edit_by_transcript, + "remove_filler_words": editing.cmd_remove_filler_words, + "transcript_markers": editing.cmd_transcript_markers, + "generate_dynamic_subtitles": subtitles.cmd_generate_dynamic_subtitles, + "generate_plain_subtitles": subtitles.cmd_generate_plain_subtitles, + "add_zoom": zoom.cmd_add_zoom, + "zoom_clips": zoom.cmd_zoom_clips, + "zoom_segments": zoom.cmd_zoom_segments, + "rename_speakers": transcription.cmd_rename_speakers, + "set_diarization": transcription.cmd_set_diarization, + "acoustics_capability": voice.cmd_acoustics_capability, + "voice_analysis": voice.cmd_voice_analysis, + "set_voice_analysis": voice.cmd_set_voice_analysis, + "analyze_voice": voice.cmd_analyze_voice, + "dynamic_subtitle_config": subtitles.cmd_dynamic_subtitle_config, + "set_dynamic_subtitle_config": subtitles.cmd_set_dynamic_subtitle_config, + "plain_subtitle_config": subtitles.cmd_plain_subtitle_config, + "set_plain_subtitle_config": subtitles.cmd_set_plain_subtitle_config, + "apply_voice_actions": voice.cmd_apply_voice_actions, + "build_phrase_review": review.cmd_build_phrase_review, + "save_phrase_review": review.cmd_save_phrase_review, + "project_config": project.cmd_project_config, + "set_project_config": project.cmd_set_project_config, + "silence_config": editing.cmd_silence_config, + "set_silence_config": editing.cmd_set_silence_config, } handler = handlers.get(command) if handler is None: diff --git a/admin/models_gui.py b/admin/models_gui.py index 1b87b11..3d17b8f 100644 --- a/admin/models_gui.py +++ b/admin/models_gui.py @@ -20,7 +20,6 @@ import subprocess import sys import threading from pathlib import Path -from typing import Optional import flet as ft @@ -29,8 +28,8 @@ _CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") if _CODE_DIR not in sys.path: sys.path.insert(0, _CODE_DIR) -from fcpxml.media_intel import media_src_to_path # noqa: E402 -from fcpxml.model_manager import ( # noqa: E402 +from fcpxml.media_intel import media_src_to_path +from fcpxml.model_manager import ( download_model, get_models_dir, is_model_downloaded, @@ -41,8 +40,8 @@ from fcpxml.model_manager import ( # noqa: E402 save_models_dir, save_selected_model, ) -from fcpxml.parser import parse_fcpxml # noqa: E402 -from fcpxml.transcribe import transcribe # noqa: E402 +from fcpxml.parser import parse_fcpxml +from fcpxml.transcribe import transcribe logger = logging.getLogger(__name__) @@ -101,9 +100,9 @@ class ModelManagerApp: def __init__(self, page: ft.Page) -> None: self.page = page self.selected = load_selected_model() - self.downloading: Optional[str] = None + self.downloading: str | None = None self._cancel_events: dict[str, threading.Event] = {} - self._picker: Optional[ft.FilePicker] = None + self._picker: ft.FilePicker | None = None # ── helpers ──────────────────────────────────────────────────────────── @@ -121,7 +120,7 @@ class ModelManagerApp: self._picker = ft.FilePicker() self._picker.on_result = self._on_file_picked self.page.overlay.append(self._picker) - self._pending_target: Optional[dict] = None + self._pending_target: dict | None = None def _on_file_picked(self, e) -> None: if self._pending_target == "project": diff --git a/code/Engine/docs/05_EXPERIENCIAS.md b/code/Engine/docs/05_EXPERIENCIAS.md index 04dd93f..ef988a7 100644 --- a/code/Engine/docs/05_EXPERIENCIAS.md +++ b/code/Engine/docs/05_EXPERIENCIAS.md @@ -1255,6 +1255,29 @@ o outro; percentil entrega um punhado útil nos dois casos. --- +## 24 — 2026-08-19 — Teste existia, mas estava fora da suíte + +- **Sintoma:** `admin/test_models_api.py` (13 testes) nunca rodava. Não + falhava — simplesmente não era coletado, então `models_api.py` figurava + como "coberto" sem que uma única asserção fosse executada em nenhum + commit. +- **Causa raiz:** `testpaths = ["tests"]` no `pyproject.toml`, com o pytest + rodando de `code/`. O arquivo morava em `admin/`, fora do alcance. Rodá-lo + à mão também falhava (`ModuleNotFoundError: admin`), porque a raiz do + repositório não entra no `sys.path` — ou seja, o único jeito de executá-lo + exigia saber de antemão que ele existia e como. +- **Solução adotada:** movido para `code/tests/test_models_api.py`, com o + insert da raiz do repositório no `sys.path` ao lado do import que precisa + dele. Passou a rodar no gate: 1441 → 1454 testes. +- **Aprendizado:** um teste fora de `testpaths` é pior que teste nenhum — ele + dá a sensação de rede sem ser rede. Ao mover ou criar teste fora da pasta + padrão, confirme que a contagem total subiu; se não subiu, ele não está + rodando. Vale também para o lint: `admin/` ainda não é coberto pelo + `run_after_fix.sh`, que roda só dentro de `code/`. +- **Estado:** `resolvido` + +--- + ## Resumo rápido (índice) | # | Data | Problema | Estado | @@ -1280,5 +1303,6 @@ o outro; percentil entrega um punhado útil nos dois casos. | 21 | 2026-08-19 | Teste ainda afirmava o default `zoom scale=1.3` removido do parser (agora vem do `zoom_scale` do usuário) | `resolvido` | | 22 | 2026-08-19 | `VideoPlayer` (AVKit) aborta em runtime no app compilado por `swiftc` — etapa 5 fechava o app; trocado por `AVPlayerLayer` | `resolvido` | | 23 | 2026-08-19 | Dividir `writer.py` em pacote quebrou `@patch('fcpxml.writer.subprocess')` — a suíte protege comportamento, não localização | `resolvido` | +| 24 | 2026-08-19 | `admin/test_models_api.py` existia mas estava fora de `testpaths` — 13 testes que nunca rodaram | `resolvido` | > Mantenha o índice acima sempre sincronizado com as entradas mais recentes. diff --git a/admin/test_models_api.py b/code/tests/test_models_api.py similarity index 61% rename from admin/test_models_api.py rename to code/tests/test_models_api.py index 2e696db..095ccac 100644 --- a/admin/test_models_api.py +++ b/code/tests/test_models_api.py @@ -1,12 +1,27 @@ -"""Tests for admin/models_api.py — the SwiftUI JSON bridge commands. +"""Tests for the SwiftUI JSON bridge commands (admin/api/). -Focused on the transcription-flow changes: atomic save, speaker renaming, and -the "use the selected model" default plus model-availability guard. +Focused on the transcription flow: atomic save, speaker renaming, and the +"use the selected model" default plus the model-availability guard. + +Patch targets follow one rule: replace a name **in the module that uses it**. +`shared.emit` is the exception that proves it — the command modules call it as +`shared.emit(...)` rather than binding the name locally, precisely so that one +patch keeps capturing the output of all of them. """ import json +import sys +from pathlib import Path -import admin.models_api as api +# `admin/` lives outside `code/`, which is pytest's rootdir — without the repo +# root on the path this module is invisible and the whole file silently stops +# being collected. It spent its life outside `testpaths` for exactly that +# reason, so keep the insert next to the import that needs it. +_REPO_ROOT = Path(__file__).resolve().parent.parent.parent +if str(_REPO_ROOT) not in sys.path: + sys.path.insert(0, str(_REPO_ROOT)) + +from admin.api import models, shared, subtitles, transcription # noqa: E402 def _capture(monkeypatch): @@ -15,13 +30,13 @@ def _capture(monkeypatch): def _emit(obj): captured.append(obj) - monkeypatch.setattr(api, "_emit", _emit) + monkeypatch.setattr(shared, "emit", _emit) return captured def test_save_json_atomic(tmp_path): p = tmp_path / "t.json" - api._save_json_atomic(p, {"a": [1, 2], "text": "olá"}) + shared._save_json_atomic(p, {"a": [1, 2], "text": "olá"}) assert p.exists() assert not (tmp_path / "t.json.tmp").exists() assert json.loads(p.read_text(encoding="utf-8"))["text"] == "olá" @@ -41,7 +56,7 @@ def test_rename_speakers(tmp_path, monkeypatch): ), encoding="utf-8", ) - assert api.cmd_rename_speakers({"path": str(p), "speakers": {"SPEAKER_01": "Erika"}}) == 0 + assert transcription.cmd_rename_speakers({"path": str(p), "speakers": {"SPEAKER_01": "Erika"}}) == 0 assert captured[0]["ok"] is True saved = json.loads(p.read_text(encoding="utf-8")) assert saved["speakers"][0]["name"] == "Speaker 1" @@ -50,32 +65,32 @@ def test_rename_speakers(tmp_path, monkeypatch): def test_rename_speakers_missing_file(monkeypatch): captured = _capture(monkeypatch) - assert api.cmd_rename_speakers({"path": "/nonexistent/x.json"}) == 1 + assert transcription.cmd_rename_speakers({"path": "/nonexistent/x.json"}) == 1 assert captured[0]["type"] == "error" def test_transcribe_requires_output_dir(monkeypatch): captured = _capture(monkeypatch) - monkeypatch.setattr(api, "load_selected_model", lambda: "small") - monkeypatch.setattr(api, "is_model_downloaded", lambda m: True) - assert api.cmd_transcribe({"path": "/some/project.fcpxml"}) == 1 + monkeypatch.setattr(transcription, "load_selected_model", lambda: "small") + monkeypatch.setattr(transcription, "is_model_downloaded", lambda m: True) + assert transcription.cmd_transcribe({"path": "/some/project.fcpxml"}) == 1 assert captured[0]["type"] == "error" assert "pasta do projeto" in captured[0]["message"] def test_transcribe_requires_installed_model(monkeypatch, tmp_path): captured = _capture(monkeypatch) - monkeypatch.setattr(api, "load_selected_model", lambda: "") - monkeypatch.setattr(api, "is_model_downloaded", lambda m: False) - assert api.cmd_transcribe({"path": "/some/project.fcpxml", "output_dir": str(tmp_path)}) == 1 + monkeypatch.setattr(transcription, "load_selected_model", lambda: "") + monkeypatch.setattr(transcription, "is_model_downloaded", lambda m: False) + assert transcription.cmd_transcribe({"path": "/some/project.fcpxml", "output_dir": str(tmp_path)}) == 1 assert captured[0]["type"] == "error" assert "instalado" in captured[0]["message"] def test_transcribe_defaults_to_selected_model(monkeypatch, tmp_path): captured = _capture(monkeypatch) - monkeypatch.setattr(api, "load_selected_model", lambda: "small") - monkeypatch.setattr(api, "is_model_downloaded", lambda m: m == "small") + monkeypatch.setattr(transcription, "load_selected_model", lambda: "small") + monkeypatch.setattr(transcription, "is_model_downloaded", lambda m: m == "small") class FakeTL: clips = [] @@ -84,32 +99,32 @@ def test_transcribe_defaults_to_selected_model(monkeypatch, tmp_path): primary_timeline = None timelines = [FakeTL()] - monkeypatch.setattr(api, "parse_fcpxml", lambda p: FakeProject()) + monkeypatch.setattr(transcription, "parse_fcpxml", lambda p: FakeProject()) # No media accessible -> reaches the media-path check (past model validation). - assert api.cmd_transcribe({"path": "/some/project.fcpxml", "output_dir": str(tmp_path)}) == 1 + assert transcription.cmd_transcribe({"path": "/some/project.fcpxml", "output_dir": str(tmp_path)}) == 1 assert captured[0]["type"] == "error" assert "mídia" in captured[0]["message"] def test_set_language_persists(monkeypatch): captured = _capture(monkeypatch) - assert api.cmd_set_language({"language": "pt"}) == 0 + assert models.cmd_set_language({"language": "pt"}) == 0 assert captured[0]["ok"] is True assert captured[0]["language"] == "pt" - assert api.load_transcript_language() == "pt" + assert models.load_transcript_language() == "pt" def test_set_language_rejects_unknown(monkeypatch): captured = _capture(monkeypatch) - assert api.cmd_set_language({"language": "xx"}) == 1 + assert models.cmd_set_language({"language": "xx"}) == 1 assert captured[0]["ok"] is False assert "language" in captured[0]["error"] def test_transcribe_defaults_language_to_persisted(monkeypatch, tmp_path): - monkeypatch.setattr(api, "load_selected_model", lambda: "small") - monkeypatch.setattr(api, "is_model_downloaded", lambda m: m == "small") - monkeypatch.setattr(api, "load_transcript_language", lambda: "pt") + monkeypatch.setattr(transcription, "load_selected_model", lambda: "small") + monkeypatch.setattr(transcription, "is_model_downloaded", lambda m: m == "small") + monkeypatch.setattr(models, "load_transcript_language", lambda: "pt") media = tmp_path / "clip.mov" media.write_bytes(b"fake") @@ -124,20 +139,21 @@ def test_transcribe_defaults_language_to_persisted(monkeypatch, tmp_path): primary_timeline = None timelines = [FakeTL()] - monkeypatch.setattr(api, "parse_fcpxml", lambda p: FakeProject()) - monkeypatch.setattr(api, "media_src_to_path", lambda mp: str(media)) + monkeypatch.setattr(transcription, "parse_fcpxml", lambda p: FakeProject()) + monkeypatch.setattr(transcription, "media_src_to_path", lambda mp: str(media)) called = {} monkeypatch.setattr( - api, "transcribe", lambda mp, model_size, language, **kw: called.update(lang=language) + transcription, "transcribe", + lambda mp, model_size, language, **kw: called.update(lang=language), ) - assert api.cmd_transcribe({"path": "/some/project.fcpxml", "output_dir": str(tmp_path / "out")}) == 1 + assert transcription.cmd_transcribe({"path": "/some/project.fcpxml", "output_dir": str(tmp_path / "out")}) == 1 assert called["lang"] == "pt" def test_srt_stamp_format(): - assert api.srt_stamp(0.0) == "00:00:00,000" - assert api.srt_stamp(1.5) == "00:00:01,500" - assert api.srt_stamp(3661.234) == "01:01:01,234" + assert subtitles.srt_stamp(0.0) == "00:00:00,000" + assert subtitles.srt_stamp(1.5) == "00:00:01,500" + assert subtitles.srt_stamp(3661.234) == "01:01:01,234" _FCPXML_SAMPLE = """ @@ -181,12 +197,12 @@ def test_cmd_export_srt_maps_to_edited_timeline(tmp_path, monkeypatch): {"start": 50.0, "end": 51.0, "text": "depois do corte"}, ] } - tj = api._transcript_json_path(media) + tj = shared._transcript_json_path(media) tj.parent.mkdir(parents=True, exist_ok=True) - api._save_json_atomic(tj, transcript) - monkeypatch.setattr(api, "media_src_to_path", lambda src: str(media)) + shared._save_json_atomic(tj, transcript) + monkeypatch.setattr(subtitles, "media_src_to_path", lambda src: str(media)) - assert api.cmd_export_srt({"path": str(project)}) == 0 + assert subtitles.cmd_export_srt({"path": str(project)}) == 0 assert captured[0]["ok"] is True srt = tmp_path / "clip_captions.srt" assert srt.exists() @@ -205,8 +221,8 @@ def test_cmd_export_srt_no_transcript(tmp_path, monkeypatch): project.write_text(_FCPXML_SAMPLE, encoding="utf-8") media = tmp_path / "clip.mp4" media.write_bytes(b"fake") - monkeypatch.setattr(api, "media_src_to_path", lambda src: str(media)) - assert api.cmd_export_srt({"path": str(project)}) == 1 + monkeypatch.setattr(subtitles, "media_src_to_path", lambda src: str(media)) + assert subtitles.cmd_export_srt({"path": str(project)}) == 1 assert captured[0]["ok"] is False @@ -229,12 +245,12 @@ def test_cmd_export_srt_clamps_past_project_duration(tmp_path, monkeypatch): {"start": 12.0, "end": 30.0, "text": "longa fala"}, ] } - tj = api._transcript_json_path(media) + tj = shared._transcript_json_path(media) tj.parent.mkdir(parents=True, exist_ok=True) - api._save_json_atomic(tj, transcript) - monkeypatch.setattr(api, "media_src_to_path", lambda src: str(media)) + shared._save_json_atomic(tj, transcript) + monkeypatch.setattr(subtitles, "media_src_to_path", lambda src: str(media)) - assert api.cmd_export_srt({"path": str(project)}) == 0 + assert subtitles.cmd_export_srt({"path": str(project)}) == 0 assert captured[0]["ok"] is True srt = tmp_path / "clip_captions.srt" text = srt.read_text(encoding="utf-8")