Fase 0 do roteiro de reestruturação (Engine/docs/10_MAPA_REESTRUTURACAO.md): move code/WHISPERX (2,6 GB de backups órfãos, sem uso ativo, sem .gitmodules) para ~/Archives/G-ART-WHISPERX-backup fora do workspace git; traz admin/ para o gate de lint de run_after_fix.sh; corrige fcpxml/writer/adjustment.py, que gerava um wrapper <adjustment> inexistente no DTD 1.13 (filtros agora vão direto no <clip>, na ordem exigida), com teste de regressão novo. Achado à parte: .gitignore tinha uma regra solta "models/" (pensada só para o cache do Whisper em code/models/) que também escondia do git todo o pacote fcpxml/models/ — nunca commitado, sem proteção nenhuma. Corrigida para /code/models/, ancorada na raiz. Docs atualizados no mesmo commit (02_MODULES, 09_MANUTENCAO, 10_MAPA_REESTRUTURACAO, 05_EXPERIENCIAS #34 e #36), conforme a regra do CLAUDE.md. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
341 lines
16 KiB
Python
341 lines
16 KiB
Python
#!/usr/bin/env python3
|
|
"""JSON bridge between the SwiftUI app and the fcp-mcp-server Python engine.
|
|
|
|
The SwiftUI app (MacApp/) launches this script as a subprocess with a command
|
|
and optional JSON arguments, then reads a single JSON document (or
|
|
newline-delimited JSON for progress) on stdout.
|
|
|
|
Commands:
|
|
catalog
|
|
-> {"models": [{display_name, internal_name, size, storage,
|
|
accuracy, speed}], "installed": [names],
|
|
"selected": name, "models_dir": path, "installed_count": n,
|
|
"recommended": [names]}
|
|
|
|
download {"model": "small"}
|
|
-> JSON-lines: {"type":"progress","fraction":0.42}
|
|
{"type":"done","installed":true}
|
|
{"type":"error","message":"..."}
|
|
|
|
cancel {"model": "small"}
|
|
-> {"ok": true}
|
|
|
|
select {"model": "small"}
|
|
-> {"ok": true, "selected": "small"}
|
|
|
|
set_language {"language": "pt"} | "auto"
|
|
-> {"ok": true, "language": "pt"}
|
|
|
|
delete {"model": "small"}
|
|
-> {"ok": true}
|
|
|
|
open_finder {"model": "small"}
|
|
-> {"ok": true}
|
|
|
|
set_models_dir {"dir": "/path"}
|
|
-> {"ok": true, "models_dir": "/path"}
|
|
|
|
inspect {"path": "/path/to/project.fcpxml"}
|
|
-> {"ok": true, "path": "...", "name": "...", "fcpxml_version": "1.13",
|
|
"timelines": [{name, duration_seconds, frame_rate, width, height,
|
|
clips, cuts, connected, markers}]}
|
|
or {"ok": false, "error": "..."}
|
|
|
|
analyze_voice {"path": "...", "output_dir": "...", "model": "...",
|
|
"language": "pt"|"auto"|null, "hf_token": "..."|null,
|
|
"num_speakers": ""|null}
|
|
Build the voice timeline (transcript+diarization+acoustics) for
|
|
every unique source media — analysis only, writes _voice_timeline.json
|
|
next to each media, `path` passes through unchanged. Meant as one
|
|
entry in the batch operations list (see processBatchStep), so
|
|
`refine_voice_timeline` never has to reopen the audio later.
|
|
-> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."}
|
|
|
|
build_phrase_review {"voice_timeline": "..._voice_timeline.json",
|
|
"actions": {...}|[...]|null, "fresh": false}
|
|
The reviewable script for the wizard's emphasis step: every phrase with
|
|
the AI's decision already applied (active/emphasis/trim). A review saved
|
|
earlier for the same timeline is returned as-is unless `fresh` is true.
|
|
-> {"ok": true, "reused": bool, "source", "duration", "speakers",
|
|
"phrases": [{index, start, end, trim_start, trim_end, text, speaker,
|
|
active, emphasis (0-3), track, peak_emphasis,
|
|
take_boundary, gap_before, reason, words}],
|
|
"errors": [...]}
|
|
|
|
save_phrase_review {"voice_timeline": "...", "phrases": [...], "source": "...",
|
|
"duration": 0.0, "speakers": [...]}
|
|
Writes _phrase_review.json plus the _phrase_actions.json derived from it.
|
|
-> {"ok": true, "review_path", "actions_path", "emphasis_count",
|
|
"removed_count"}
|
|
|
|
build_speaker_review {"voice_timeline": "..._voice_timeline.json", "fresh": false}
|
|
Runs right after `analyze_voice` (wizard step 3): who was detected
|
|
(with speaking share and sample lines, default active/kept) plus one
|
|
row per transcript segment, for a naming + mute + strike-line screen
|
|
before anything reaches the AI. A review saved earlier is merged
|
|
back on top unless `fresh` is true.
|
|
-> {"ok": true, "reused": bool, "source", "duration", "video_type",
|
|
"speakers": [{id, name, display_name, speaking_seconds, share,
|
|
segment_count, avg_segment, word_count, samples,
|
|
active}],
|
|
"segments": [{id, start, end, speaker, text, excluded, words}]}
|
|
|
|
recalc_speaker_review {"voice_timeline": "...", "speakers": [...],
|
|
"segments": [...], "video_type": ""}
|
|
Reruns the same filter+recompute `save_speaker_review` persists to
|
|
_voice_timeline_clean.json, but writes nothing — a live preview for
|
|
the "Recalcular" button so muting a speaker or striking a line
|
|
updates the emphasis/peak numbers shown (pure math over the words
|
|
that survived; no new audio pass).
|
|
-> {"ok": true, "duration", "peak_count",
|
|
"segments": [{start, end, speaker, text, words, ...}]}
|
|
|
|
save_speaker_review {"voice_timeline": "...", "speakers": [...],
|
|
"segments": [...], "video_type": "", "source": "...",
|
|
"duration": 0.0}
|
|
Writes _speaker_review.json (the decisions) and
|
|
_voice_timeline_clean.json (inactive speakers + struck lines
|
|
removed) — the raw _voice_timeline.json is never touched. From here
|
|
on, `copyForChat` and `generate_voice_script` should prefer the
|
|
_clean file when it exists.
|
|
-> {"ok": true, "review_path", "clean_path", "active_speakers",
|
|
"muted_speakers", "excluded_segments"}
|
|
|
|
generate_voice_script {"media_path": "...", "voice_timeline": "...", "filepath": "...",
|
|
"model": "gemma3:12b", "base_url": "http://localhost:11434",
|
|
"model_size": "base", "language": "pt"|"auto"|null,
|
|
"hf_token": "..."|null, "num_speakers": ""|null,
|
|
"output_dir": "...", "apply_to_fcpxml": true}
|
|
The WHOLE voice-edit pass run INSIDE the engine against a LOCAL model
|
|
(Ollama running Gemma 3 / Llama) — no wizard, no copy-paste. If
|
|
`voice_timeline` is given (the analysis from an earlier step), it is
|
|
reused and the transcription/acoustics are skipped; otherwise
|
|
`media_path` is transcribed and analyzed. Then the local model directs
|
|
the edit -> returns the readable script (roteiro) + the action JSON,
|
|
and optionally applies to `filepath` (FCPXML, non-destructive).
|
|
-> {"ok": true, "message", "roteiro_path", "actions_path", "applied_path"}
|
|
or {"ok": false, "error": "..."}
|
|
|
|
list_ollama_models {"base_url": "http://localhost:11434"}
|
|
Lists the models Ollama currently serves, for the app's model picker
|
|
in step 4 (generate script by local AI). Empty list if Ollama is
|
|
unreachable, so the UI falls back to a free-text field.
|
|
-> {"ok": true, "models": ["gemma3:12b", ...]}
|
|
|
|
dynamic_subtitle_config {}
|
|
Compat: style of the FIRST ACTIVE registered layout (no id/name/
|
|
active). Prefer list_dynamic_subtitle_layouts for the app's UI.
|
|
-> {"ok": true, "band_height", "block_center_y", "line_gap", "font",
|
|
"font_size", "emphasis_font", "emphasis_face", "emphasis_size",
|
|
"active_color", "emphasis_color", "text_scale"}
|
|
|
|
set_dynamic_subtitle_config {<any of the fields above>}
|
|
Compat: persists style fields onto the first active layout. Prefer
|
|
update_dynamic_subtitle_layout for the app's UI.
|
|
-> {"ok": true, <same shape as dynamic_subtitle_config>}
|
|
|
|
list_dynamic_subtitle_layouts {}
|
|
All registered "Legendas Dinâmicas" layouts. A single global config
|
|
used to hold ONE style; it's now a list of named, independently
|
|
toggleable layouts. With 2+ marked `active`, generate_dynamic_subtitles
|
|
randomly samples one per subtitle block, alternating styles through
|
|
the video. With 0 active, the first registered layout is used.
|
|
-> {"ok": true, "layouts": [{"id", "name", "active", "band_height",
|
|
"block_center_y", "line_gap", "font", "font_size",
|
|
"emphasis_font", "emphasis_face", "emphasis_size",
|
|
"active_color", "emphasis_color", "text_scale", "role"}, ...]}
|
|
|
|
create_dynamic_subtitle_layout {"name": "...", <any style field above>}
|
|
Registers a new layout, active by default. Omitted style fields fall
|
|
back to the same defaults as the very first layout.
|
|
-> {"ok": true, "layout": {...}}
|
|
|
|
update_dynamic_subtitle_layout {"id": "...", <name/active/style fields>}
|
|
Updates only the given fields of one registered layout.
|
|
-> {"ok": true, "layout": {...}} or {"ok": false, "error": "..."}
|
|
|
|
delete_dynamic_subtitle_layout {"id": "..."}
|
|
Removes a layout. If it was the last one, a "Padrão" layout is
|
|
recreated automatically so there is always at least one registered.
|
|
-> {"ok": true, "layouts": [...]}
|
|
|
|
set_dynamic_subtitle_layout_active {"id": "...", "active": true}
|
|
Toggles one layout's active flag.
|
|
-> {"ok": true, "layout": {...}}
|
|
|
|
silence_config {}
|
|
-> {"ok": true, "noise_db": -30.0, "min_silence": 0.5, "padding": 0.05}
|
|
|
|
set_silence_config {"noise_db": -30.0, "min_silence": 0.5, "padding": 0.05}
|
|
Persists only the given fields. detect_media_silence and
|
|
remove_media_silence read this as their own fallback default.
|
|
-> {"ok": true, <same shape as silence_config>}
|
|
|
|
transcribe {"path": "...", "model": "small", "language": "pt"|null,
|
|
"hf_token": "..."|null, "num_speakers": ""|null,
|
|
"force": true|false}
|
|
`force: true` ignores the existing transcript cache and overwrites it
|
|
with a fresh transcription. The default is false.
|
|
-> JSON-lines:
|
|
{"type":"progress","fraction":0.5,"stage":"Transcrevendo..."}
|
|
{"type":"result","transcripts":[{"media","language","words",
|
|
"duration","preview","saved",
|
|
"speakers"}]}
|
|
{"type":"error","message":"..."}
|
|
|
|
edit_by_transcript {"path": "...", "phrases": ["frase um", "frase dois"],
|
|
"mode": "remove"|"keep_only", "clip_name": "..."|null,
|
|
"padding": 0.0, "model": "small", "language": "pt"|null}
|
|
-> {"ok": true, "path": "..._transcript_edit.fcpxml", "message": "..."}
|
|
or {"ok": false, "error": "..."}
|
|
|
|
remove_filler_words {"path": "...", "fillers": ["um","uh"]|null,
|
|
"clip_name": "..."|null, "padding": 0.02,
|
|
"model": "small", "language": "pt"|null}
|
|
-> {"ok": true, "path": "..._defillered.fcpxml", "message": "..."}
|
|
or {"ok": false, "error": "..."}
|
|
|
|
transcript_markers {"path": "...", "clip_name": "..."|null,
|
|
"marker_type": "chapter", "max_label_length": 50,
|
|
"model": "small", "language": "pt"|null}
|
|
-> {"ok": true, "path": "..._transcript_markers.fcpxml", "message": "..."}
|
|
or {"ok": false, "error": "..."}
|
|
|
|
add_zoom {"path": "...", "clip_id": "...", "start": 10.0, "end": 16.0,
|
|
"scale": 1.3, "ease": 0.3, "position": "0 0"|null}
|
|
-> {"ok": true, "path": "..._zoom.fcpxml", "message": "..."}
|
|
or {"ok": false, "error": "..."}
|
|
|
|
generate_dynamic_subtitles {"path": "...", "clip_name": "..."|null,
|
|
"band_height": 0.22, "block_center_y": -167,
|
|
"font": "Helvetica Neue", "font_size": 128,
|
|
"emphasis_font": "Playfair Display",
|
|
"emphasis_face": "Medium Italic", "emphasis_size": 265,
|
|
"active_color": "1 1 1 1", "emphasis_color": "1 1 1 1",
|
|
"model": "small", "language": "pt"|null}
|
|
-> {"ok": true, "path": "..._dynamic_subtitles.fcpxml", "message": "..."}
|
|
or {"ok": false, "error": "..."}
|
|
|
|
rename_speakers {"path": "/to/media_transcript.json",
|
|
"speakers": {"SPEAKER_01": "Nome"}}
|
|
-> {"ok": true, "speakers": [...]}
|
|
|
|
set_diarization {"token": "hf_...", "num_speakers": ""}
|
|
-> {"ok": true, "diarization": bool, "diarization_message": "...",
|
|
"num_speakers": "..."}
|
|
|
|
acoustics_capability
|
|
Whether librosa (pitch/energy for voice analysis) is installed.
|
|
-> {"ok": true, "available": bool, "message": "..."}
|
|
|
|
voice_analysis
|
|
-> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85,
|
|
"emphasis_weights": {...}, "emotion_enabled": false,
|
|
"emotion_sensitivity": 0.5}
|
|
|
|
set_voice_analysis {"energy_threshold": 0.6, "emphasis_threshold": 0.9,
|
|
"emphasis_weights": {"energy": 0.4}|null,
|
|
"emotion_enabled": true, "emotion_sensitivity": 0.5}
|
|
-> same shape as voice_analysis (only given fields change)
|
|
|
|
Exit code 0 on success, 1 on error.
|
|
"""
|
|
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
# `admin/` não é um pacote instalado — a raiz do repositório precisa estar no
|
|
# caminho para `admin.api` resolver quando o app roda este arquivo por path.
|
|
_REPO_ROOT = str(Path(__file__).resolve().parent.parent)
|
|
if _REPO_ROOT not in sys.path:
|
|
sys.path.insert(0, _REPO_ROOT)
|
|
|
|
from admin.api import ( # noqa: E402
|
|
editing,
|
|
models,
|
|
project,
|
|
review,
|
|
subtitles,
|
|
transcription,
|
|
voice,
|
|
zoom,
|
|
)
|
|
from admin.api.shared import emit # noqa: E402,F401
|
|
|
|
|
|
def main() -> int:
|
|
args = sys.argv[1:]
|
|
if not args:
|
|
print("usage: models_api.py <command> [json_args]", file=sys.stderr)
|
|
return 1
|
|
command = args[0]
|
|
try:
|
|
data: dict = json.loads(args[1]) if len(args) > 1 else {}
|
|
except json.JSONDecodeError:
|
|
print("invalid JSON args", file=sys.stderr)
|
|
return 1
|
|
|
|
handlers = {
|
|
"catalog": models.cmd_catalog,
|
|
"download": models.cmd_download,
|
|
"cancel": models.cmd_cancel,
|
|
"select": models.cmd_select,
|
|
"set_language": models.cmd_set_language,
|
|
"delete": models.cmd_delete,
|
|
"open_finder": models.cmd_open_finder,
|
|
"set_models_dir": models.cmd_set_models_dir,
|
|
"inspect": project.cmd_inspect,
|
|
"transcribe": transcription.cmd_transcribe,
|
|
"export_srt": subtitles.cmd_export_srt,
|
|
"remove_silences": editing.cmd_remove_silences,
|
|
"edit_by_transcript": editing.cmd_edit_by_transcript,
|
|
"remove_filler_words": editing.cmd_remove_filler_words,
|
|
"transcript_markers": editing.cmd_transcript_markers,
|
|
"generate_dynamic_subtitles": subtitles.cmd_generate_dynamic_subtitles,
|
|
"generate_plain_subtitles": subtitles.cmd_generate_plain_subtitles,
|
|
"add_zoom": zoom.cmd_add_zoom,
|
|
"zoom_clips": zoom.cmd_zoom_clips,
|
|
"zoom_segments": zoom.cmd_zoom_segments,
|
|
"rename_speakers": transcription.cmd_rename_speakers,
|
|
"set_diarization": transcription.cmd_set_diarization,
|
|
"acoustics_capability": voice.cmd_acoustics_capability,
|
|
"voice_analysis": voice.cmd_voice_analysis,
|
|
"set_voice_analysis": voice.cmd_set_voice_analysis,
|
|
"analyze_voice": voice.cmd_analyze_voice,
|
|
"dynamic_subtitle_config": subtitles.cmd_dynamic_subtitle_config,
|
|
"set_dynamic_subtitle_config": subtitles.cmd_set_dynamic_subtitle_config,
|
|
"list_dynamic_subtitle_layouts": subtitles.cmd_list_dynamic_subtitle_layouts,
|
|
"create_dynamic_subtitle_layout": subtitles.cmd_create_dynamic_subtitle_layout,
|
|
"update_dynamic_subtitle_layout": subtitles.cmd_update_dynamic_subtitle_layout,
|
|
"delete_dynamic_subtitle_layout": subtitles.cmd_delete_dynamic_subtitle_layout,
|
|
"set_dynamic_subtitle_layout_active": subtitles.cmd_set_dynamic_subtitle_layout_active,
|
|
"plain_subtitle_config": subtitles.cmd_plain_subtitle_config,
|
|
"set_plain_subtitle_config": subtitles.cmd_set_plain_subtitle_config,
|
|
"apply_voice_actions": voice.cmd_apply_voice_actions,
|
|
"generate_voice_script": voice.cmd_generate_voice_script,
|
|
"list_ollama_models": voice.cmd_list_ollama_models,
|
|
"build_phrase_review": review.cmd_build_phrase_review,
|
|
"save_phrase_review": review.cmd_save_phrase_review,
|
|
"build_speaker_review": review.cmd_build_speaker_review,
|
|
"recalc_speaker_review": review.cmd_recalc_speaker_review,
|
|
"save_speaker_review": review.cmd_save_speaker_review,
|
|
"project_config": project.cmd_project_config,
|
|
"set_project_config": project.cmd_set_project_config,
|
|
"silence_config": editing.cmd_silence_config,
|
|
"set_silence_config": editing.cmd_set_silence_config,
|
|
}
|
|
handler = handlers.get(command)
|
|
if handler is None:
|
|
print(f"unknown command: {command}", file=sys.stderr)
|
|
return 1
|
|
try:
|
|
result = handler(data)
|
|
except TypeError:
|
|
result = handler()
|
|
return result or 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|