#!/usr/bin/env python3 """JSON bridge between the SwiftUI app and the fcp-mcp-server Python engine. The SwiftUI app (MacApp/) launches this script as a subprocess with a command and optional JSON arguments, then reads a single JSON document (or newline-delimited JSON for progress) on stdout. Commands: catalog -> {"models": [{display_name, internal_name, size, storage, accuracy, speed}], "installed": [names], "selected": name, "models_dir": path, "installed_count": n, "recommended": [names]} download {"model": "small"} -> JSON-lines: {"type":"progress","fraction":0.42} {"type":"done","installed":true} {"type":"error","message":"..."} cancel {"model": "small"} -> {"ok": true} select {"model": "small"} -> {"ok": true, "selected": "small"} set_language {"language": "pt"} | "auto" -> {"ok": true, "language": "pt"} delete {"model": "small"} -> {"ok": true} open_finder {"model": "small"} -> {"ok": true} set_models_dir {"dir": "/path"} -> {"ok": true, "models_dir": "/path"} inspect {"path": "/path/to/project.fcpxml"} -> {"ok": true, "path": "...", "name": "...", "fcpxml_version": "1.13", "timelines": [{name, duration_seconds, frame_rate, width, height, clips, cuts, connected, markers}]} or {"ok": false, "error": "..."} analyze_voice {"path": "...", "output_dir": "...", "model": "...", "language": "pt"|"auto"|null, "hf_token": "..."|null, "num_speakers": ""|null} Build the voice timeline (transcript+diarization+acoustics) for every unique source media — analysis only, writes _voice_timeline.json next to each media, `path` passes through unchanged. Meant as one entry in the batch operations list (see processBatchStep), so `refine_voice_timeline` never has to reopen the audio later. -> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."} build_phrase_review {"voice_timeline": "..._voice_timeline.json", "actions": {...}|[...]|null, "fresh": false} The reviewable script for the wizard's emphasis step: every phrase with the AI's decision already applied (active/emphasis/trim). A review saved earlier for the same timeline is returned as-is unless `fresh` is true. -> {"ok": true, "reused": bool, "source", "duration", "speakers", "phrases": [{index, start, end, trim_start, trim_end, text, speaker, active, emphasis (0-3), track, peak_emphasis, take_boundary, gap_before, reason, words}], "errors": [...]} save_phrase_review {"voice_timeline": "...", "phrases": [...], "source": "...", "duration": 0.0, "speakers": [...]} Writes _phrase_review.json plus the _phrase_actions.json derived from it. -> {"ok": true, "review_path", "actions_path", "emphasis_count", "removed_count"} generate_voice_script {"media_path": "...", "voice_timeline": "...", "filepath": "...", "model": "gemma3:12b", "base_url": "http://localhost:11434", "model_size": "base", "language": "pt"|"auto"|null, "hf_token": "..."|null, "num_speakers": ""|null, "output_dir": "...", "apply_to_fcpxml": true} The WHOLE voice-edit pass run INSIDE the engine against a LOCAL model (Ollama running Gemma 3 / Llama) — no wizard, no copy-paste. If `voice_timeline` is given (the analysis from an earlier step), it is reused and the transcription/acoustics are skipped; otherwise `media_path` is transcribed and analyzed. Then the local model directs the edit -> returns the readable script (roteiro) + the action JSON, and optionally applies to `filepath` (FCPXML, non-destructive). -> {"ok": true, "message", "roteiro_path", "actions_path", "applied_path"} or {"ok": false, "error": "..."} list_ollama_models {"base_url": "http://localhost:11434"} Lists the models Ollama currently serves, for the app's model picker in step 4 (generate script by local AI). Empty list if Ollama is unreachable, so the UI falls back to a free-text field. -> {"ok": true, "models": ["gemma3:12b", ...]} dynamic_subtitle_config {} -> {"ok": true, "band_height", "block_center_y", "line_gap", "font", "font_size", "emphasis_font", "emphasis_face", "emphasis_size", "active_color", "emphasis_color", "text_scale"} set_dynamic_subtitle_config {} Persists only the given fields to ~/.fcp-mcp-server/config.json. generate_dynamic_subtitles reads this as its own fallback default. -> {"ok": true, } silence_config {} -> {"ok": true, "noise_db": -30.0, "min_silence": 0.5, "padding": 0.05} set_silence_config {"noise_db": -30.0, "min_silence": 0.5, "padding": 0.05} Persists only the given fields. detect_media_silence and remove_media_silence read this as their own fallback default. -> {"ok": true, } transcribe {"path": "...", "model": "small", "language": "pt"|null, "hf_token": "..."|null, "num_speakers": ""|null} -> JSON-lines: {"type":"progress","fraction":0.5,"stage":"Transcrevendo..."} {"type":"result","transcripts":[{"media","language","words", "duration","preview","saved", "speakers"}]} {"type":"error","message":"..."} edit_by_transcript {"path": "...", "phrases": ["frase um", "frase dois"], "mode": "remove"|"keep_only", "clip_name": "..."|null, "padding": 0.0, "model": "small", "language": "pt"|null} -> {"ok": true, "path": "..._transcript_edit.fcpxml", "message": "..."} or {"ok": false, "error": "..."} remove_filler_words {"path": "...", "fillers": ["um","uh"]|null, "clip_name": "..."|null, "padding": 0.02, "model": "small", "language": "pt"|null} -> {"ok": true, "path": "..._defillered.fcpxml", "message": "..."} or {"ok": false, "error": "..."} transcript_markers {"path": "...", "clip_name": "..."|null, "marker_type": "chapter", "max_label_length": 50, "model": "small", "language": "pt"|null} -> {"ok": true, "path": "..._transcript_markers.fcpxml", "message": "..."} or {"ok": false, "error": "..."} add_zoom {"path": "...", "clip_id": "...", "start": 10.0, "end": 16.0, "scale": 1.3, "ease": 0.3, "position": "0 0"|null} -> {"ok": true, "path": "..._zoom.fcpxml", "message": "..."} or {"ok": false, "error": "..."} generate_dynamic_subtitles {"path": "...", "clip_name": "..."|null, "band_height": 0.22, "block_center_y": -167, "font": "Helvetica Neue", "font_size": 128, "emphasis_font": "Playfair Display", "emphasis_face": "Medium Italic", "emphasis_size": 265, "active_color": "1 1 1 1", "emphasis_color": "1 1 1 1", "model": "small", "language": "pt"|null} -> {"ok": true, "path": "..._dynamic_subtitles.fcpxml", "message": "..."} or {"ok": false, "error": "..."} rename_speakers {"path": "/to/media_transcript.json", "speakers": {"SPEAKER_01": "Nome"}} -> {"ok": true, "speakers": [...]} set_diarization {"token": "hf_...", "num_speakers": ""} -> {"ok": true, "diarization": bool, "diarization_message": "...", "num_speakers": "..."} acoustics_capability Whether librosa (pitch/energy for voice analysis) is installed. -> {"ok": true, "available": bool, "message": "..."} voice_analysis -> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85, "emphasis_weights": {...}, "emotion_enabled": false, "emotion_sensitivity": 0.5} set_voice_analysis {"energy_threshold": 0.6, "emphasis_threshold": 0.9, "emphasis_weights": {"energy": 0.4}|null, "emotion_enabled": true, "emotion_sensitivity": 0.5} -> same shape as voice_analysis (only given fields change) Exit code 0 on success, 1 on error. """ import json import sys from pathlib import Path # `admin/` não é um pacote instalado — a raiz do repositório precisa estar no # caminho para `admin.api` resolver quando o app roda este arquivo por path. _REPO_ROOT = str(Path(__file__).resolve().parent.parent) if _REPO_ROOT not in sys.path: sys.path.insert(0, _REPO_ROOT) from admin.api import ( editing, models, project, review, subtitles, transcription, voice, zoom, ) from admin.api.shared import emit # noqa: F401 def main() -> int: args = sys.argv[1:] if not args: print("usage: models_api.py [json_args]", file=sys.stderr) return 1 command = args[0] try: data: dict = json.loads(args[1]) if len(args) > 1 else {} except json.JSONDecodeError: print("invalid JSON args", file=sys.stderr) return 1 handlers = { "catalog": models.cmd_catalog, "download": models.cmd_download, "cancel": models.cmd_cancel, "select": models.cmd_select, "set_language": models.cmd_set_language, "delete": models.cmd_delete, "open_finder": models.cmd_open_finder, "set_models_dir": models.cmd_set_models_dir, "inspect": project.cmd_inspect, "transcribe": transcription.cmd_transcribe, "export_srt": subtitles.cmd_export_srt, "remove_silences": editing.cmd_remove_silences, "edit_by_transcript": editing.cmd_edit_by_transcript, "remove_filler_words": editing.cmd_remove_filler_words, "transcript_markers": editing.cmd_transcript_markers, "generate_dynamic_subtitles": subtitles.cmd_generate_dynamic_subtitles, "generate_plain_subtitles": subtitles.cmd_generate_plain_subtitles, "add_zoom": zoom.cmd_add_zoom, "zoom_clips": zoom.cmd_zoom_clips, "zoom_segments": zoom.cmd_zoom_segments, "rename_speakers": transcription.cmd_rename_speakers, "set_diarization": transcription.cmd_set_diarization, "acoustics_capability": voice.cmd_acoustics_capability, "voice_analysis": voice.cmd_voice_analysis, "set_voice_analysis": voice.cmd_set_voice_analysis, "analyze_voice": voice.cmd_analyze_voice, "dynamic_subtitle_config": subtitles.cmd_dynamic_subtitle_config, "set_dynamic_subtitle_config": subtitles.cmd_set_dynamic_subtitle_config, "plain_subtitle_config": subtitles.cmd_plain_subtitle_config, "set_plain_subtitle_config": subtitles.cmd_set_plain_subtitle_config, "apply_voice_actions": voice.cmd_apply_voice_actions, "generate_voice_script": voice.cmd_generate_voice_script, "list_ollama_models": voice.cmd_list_ollama_models, "build_phrase_review": review.cmd_build_phrase_review, "save_phrase_review": review.cmd_save_phrase_review, "project_config": project.cmd_project_config, "set_project_config": project.cmd_set_project_config, "silence_config": editing.cmd_silence_config, "set_silence_config": editing.cmd_set_silence_config, } handler = handlers.get(command) if handler is None: print(f"unknown command: {command}", file=sys.stderr) return 1 try: result = handler(data) except TypeError: result = handler() return result or 0 if __name__ == "__main__": sys.exit(main())