"""Análise de voz e aplicação das decisões de edição. Extraído de models_api.py — a tabela de comandos segue lá. """ from __future__ import annotations import asyncio import json import re from pathlib import Path from fcpxml.model_manager import ( load_hf_token, load_num_speakers, load_selected_model, load_transcript_language, load_voice_analysis_config, save_voice_analysis_config, ) from . import shared from .shared import ( _load_cached_transcript, _load_cached_voice_timeline, _project_media_paths, _project_media_rotations, _transcript_json_path, _voice_timeline_json_path, ) def cmd_analyze_voice(args: dict) -> int: """Build the voice timeline (transcript+diarization+acoustics -> emphasis) for every unique source media in the project, so `refine_voice_timeline` and friends have something to read without ever reopening the audio. Analysis only — writes _voice_timeline.json next to each media, doesn't touch the project XML. `path` passes through unchanged so it composes with the other batch steps (silence removal, captions) regardless of where in the list it runs. """ path = str(args.get("path", "")) if not path or not Path(path).exists(): shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) return 1 model = str(args.get("model", "") or load_selected_model() or "") language = args.get("language") if language is None: language = load_transcript_language() if language == "auto": language = None token = str(args.get("hf_token") or load_hf_token() or "") num_speakers = str(args.get("num_speakers") or load_num_speakers() or "") try: media_paths = _project_media_paths(path) rotations = _project_media_rotations(path) except Exception as exc: shared.emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"}) return 1 if not media_paths: shared.emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."}) return 1 from server import handle_build_voice_timeline messages: list[str] = [] output_dir = str(args.get("output_dir") or "").strip() existing: list[Path] = [] for mp in media_paths: timeline_path = _voice_timeline_json_path(mp, output_dir) if _load_cached_voice_timeline(timeline_path, mp) is not None: existing.append(timeline_path) if existing and len(existing) == len(media_paths) and not bool(args.get("force_reprocess", False)): message = "# Voice Timeline Cache\n\n" message += "Reaproveitando análise de voz existente. Nada foi reprocessado.\n\n" for timeline_path in existing: message += f"- **Timeline JSON**: {timeline_path}\n" shared.emit({ "ok": True, "path": path, "reused": True, "timelines": [str(p) for p in existing], "message": message, }) return 0 for mp in media_paths: transcript_path = _transcript_json_path(mp, output_dir) reused_prefix = "" if _load_cached_transcript(transcript_path) is not None: reused_prefix = f"# Cache\n\nReaproveitando transcrição existente: `{transcript_path}`\n\n" try: contents = asyncio.run(handle_build_voice_timeline({ "media_path": mp, "model": model, "language": language, "hf_token": token, "num_speakers": num_speakers, "output_dir": output_dir, "rotation": rotations.get(mp, 0.0), })) except Exception as exc: shared.emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"}) return 1 messages.append(reused_prefix + "\n".join(getattr(c, "text", str(c)) for c in contents)) shared.emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)}) return 0 def cmd_acoustics_capability(args: dict) -> int: """Whether librosa (pitch/energy extraction) is installed in this venv. Surfaces `features_capability()` — previously computed but never exposed to the app, so `layers.acoustics: false` in a voice timeline had no explanation the user could act on. """ from fcpxml.voice_features import features_capability ok, msg = features_capability() shared.emit({"ok": True, "available": ok, "message": msg}) return 0 def cmd_voice_analysis(args: dict) -> int: """Read the persisted voice-analysis settings (energy/emphasis/emotion).""" config = load_voice_analysis_config() shared.emit({"ok": True, **config, "emphasis_threshold": config["emphasis_floor"]}) return 0 def cmd_set_voice_analysis(args: dict) -> int: """Persist voice-analysis settings. Only the given fields change.""" weights = args.get("emphasis_weights") config = save_voice_analysis_config( energy_threshold=args.get("energy_threshold"), emphasis_weights=weights if isinstance(weights, dict) else None, emphasis_floor=args.get("emphasis_threshold"), emotion_enabled=args.get("emotion_enabled"), emotion_sensitivity=args.get("emotion_sensitivity"), zoom_scale=args.get("zoom_scale"), zoom_mode=args.get("zoom_mode"), zoom_ease_in=args.get("zoom_ease_in"), zoom_ease_out=args.get("zoom_ease_out"), ) shared.emit({"ok": True, **config}) return 0 def cmd_apply_voice_actions(args: dict) -> int: """Apply a decision list (cuts/zooms/texts/markers) to the project XML. The list is produced by a model reading the _voice_timeline.json — this is the step that turns those decisions into an edit, and the one the batch chain was missing: without it the app could measure the voice and caption the result, but never cut by it. `actions_path` points at the JSON; either a bare list or the ``{"actions": [...]}`` wrapper the skill emits is accepted. Times stay in ORIGINAL source seconds — the handler resolves cuts first and shifts everything else itself. """ path = str(args.get("path", "")) if not path or not Path(path).exists(): shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) return 1 actions = args.get("actions") if actions is None: actions_path = str(args.get("actions_path", "")) if not actions_path or not Path(actions_path).exists(): shared.emit({"ok": False, "error": "Arquivo de decisões (JSON) não encontrado."}) return 1 try: with open(actions_path, encoding="utf-8") as fh: loaded = json.load(fh) except (OSError, ValueError) as exc: shared.emit({"ok": False, "error": f"Erro ao ler as decisões: {exc}"}) return 1 actions = loaded.get("actions") if isinstance(loaded, dict) else loaded # The documented output format is {"source": ..., "actions": [...]} — # callers passing that whole object inline (e.g. the wizard pasting the # skill's JSON verbatim) need the same unwrap the actions_path branch # above already does, or a well-formed payload gets rejected as # "malformed" for having one extra layer of nesting. if isinstance(actions, dict): actions = actions.get("actions") if not isinstance(actions, list) or not actions: shared.emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."}) return 1 from server import handle_apply_voice_actions try: contents = asyncio.run(handle_apply_voice_actions({ "filepath": path, "actions": actions, "output_dir": args.get("output_dir"), })) except Exception as exc: shared.emit({"ok": False, "error": f"Falha ao aplicar as decisões: {exc}"}) return 1 message = "\n".join(getattr(c, "text", str(c)) for c in contents) # The handler reports dropped/rejected actions individually; hand the # whole report back so the app can surface them instead of only the count. out_path = path for line in message.splitlines(): if line.startswith("- **Saved to**:"): out_path = line.split("`")[1] if "`" in line else path break shared.emit({"ok": True, "path": out_path, "message": message}) return 0 def cmd_generate_voice_script(args: dict) -> int: """Run the ENTIRE voice-edit pass against a LOCAL model, inside the engine. Transcribe (cached) -> build the voice timeline -> hand it to a local Ollama model (Gemma 3 / Llama) that directs the edit -> return the readable script (roteiro) and the action JSON, and optionally apply to a FCPXML. No wizard, no copy-paste: the model's decisions are validated and applied by the same pipeline the rules engine uses. Args (all optional except one of ``media_path`` / ``voice_timeline``): media_path audio/video to analyze and direct (required when there is no voice_timeline yet) voice_timeline path to an existing _voice_timeline.json; when given the analysis is reused and media_path is not required filepath optional FCPXML to apply the decisions to (non-destructive) model local model Ollama serves (default gemma3:12b) base_url Ollama base URL (default http://localhost:11434) model_size whisper size if transcription is needed language ISO language hint for transcription hf_token HuggingFace token for diarization num_speakers known speaker count, if any output_dir folder for the timeline/review/actions JSON apply_to_fcpxml apply to filepath when given (default true) -> {"ok": true, "message": "...", "roteiro_path", "actions_path", "applied_path"} or {"ok": false, "error": "..."} """ media_path = str(args.get("media_path", "")) voice_timeline = str(args.get("voice_timeline", "")) if not voice_timeline and (not media_path or not Path(media_path).exists()): shared.emit({"ok": False, "error": "Arquivo de mídia não encontrado (informe media_path ou voice_timeline)."}) return 1 from server import handle_generate_voice_script try: contents = asyncio.run(handle_generate_voice_script({ "media_path": media_path, "voice_timeline": args.get("voice_timeline"), "filepath": args.get("filepath"), "model": args.get("model"), "base_url": args.get("base_url"), "model_size": args.get("model_size"), "language": args.get("language"), "hf_token": args.get("hf_token"), "num_speakers": args.get("num_speakers"), "output_dir": args.get("output_dir"), "apply_to_fcpxml": args.get("apply_to_fcpxml", True), })) except Exception as exc: shared.emit({"ok": False, "error": f"Falha ao gerar roteiro por IA local: {exc}"}) return 1 message = "\n".join(getattr(c, "text", str(c)) for c in contents) def _path_after(label: str) -> str: m = re.search(rf"\*\*{label}\*\*: (.+)", message) return m.group(1).strip() if m else "" roteiro_path = _path_after(r"Roteiro \(legível\)") actions_path = _path_after("Ações JSON") applied_path = "" for line in message.splitlines(): if line.startswith("- **Saved to**:"): applied_path = line.split("`")[1] if "`" in line else "" break shared.emit({ "ok": True, "message": message, "roteiro_path": roteiro_path, "actions_path": actions_path, "applied_path": applied_path, }) return 0 def cmd_list_ollama_models(args: dict) -> int: """List the models Ollama currently serves, for the app's model picker. Args: base_url Ollama base URL (default http://localhost:11434) -> {"ok": true, "models": ["gemma3:12b", ...]} (empty list if Ollama is unreachable, so the UI can fall back to a text field) """ from fcpxml.llm_local import list_ollama_models base_url = str(args.get("base_url") or "http://localhost:11434") models = list_ollama_models(base_url=base_url) shared.emit({"ok": True, "models": models}) return 0