Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
301 lines
12 KiB
Python
301 lines
12 KiB
Python
"""Análise de voz e aplicação das decisões de edição.
|
|
|
|
Extraído de models_api.py — a tabela de comandos segue lá.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import json
|
|
import re
|
|
from pathlib import Path
|
|
|
|
from fcpxml.model_manager import (
|
|
load_hf_token,
|
|
load_num_speakers,
|
|
load_selected_model,
|
|
load_transcript_language,
|
|
load_voice_analysis_config,
|
|
save_voice_analysis_config,
|
|
)
|
|
|
|
from . import shared
|
|
from .shared import (
|
|
_load_cached_transcript,
|
|
_load_cached_voice_timeline,
|
|
_project_media_paths,
|
|
_project_media_rotations,
|
|
_transcript_json_path,
|
|
_voice_timeline_json_path,
|
|
)
|
|
|
|
|
|
def cmd_analyze_voice(args: dict) -> int:
|
|
"""Build the voice timeline (transcript+diarization+acoustics -> emphasis)
|
|
for every unique source media in the project, so `refine_voice_timeline`
|
|
and friends have something to read without ever reopening the audio.
|
|
|
|
Analysis only — writes _voice_timeline.json next to each media, doesn't
|
|
touch the project XML. `path` passes through unchanged so it composes
|
|
with the other batch steps (silence removal, captions) regardless of
|
|
where in the list it runs.
|
|
"""
|
|
path = str(args.get("path", ""))
|
|
if not path or not Path(path).exists():
|
|
shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
|
|
return 1
|
|
|
|
model = str(args.get("model", "") or load_selected_model() or "")
|
|
language = args.get("language")
|
|
if language is None:
|
|
language = load_transcript_language()
|
|
if language == "auto":
|
|
language = None
|
|
token = str(args.get("hf_token") or load_hf_token() or "")
|
|
num_speakers = str(args.get("num_speakers") or load_num_speakers() or "")
|
|
|
|
try:
|
|
media_paths = _project_media_paths(path)
|
|
rotations = _project_media_rotations(path)
|
|
except Exception as exc:
|
|
shared.emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
|
|
return 1
|
|
if not media_paths:
|
|
shared.emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."})
|
|
return 1
|
|
|
|
from server import handle_build_voice_timeline
|
|
|
|
messages: list[str] = []
|
|
output_dir = str(args.get("output_dir") or "").strip()
|
|
existing: list[Path] = []
|
|
for mp in media_paths:
|
|
timeline_path = _voice_timeline_json_path(mp, output_dir)
|
|
if _load_cached_voice_timeline(timeline_path, mp) is not None:
|
|
existing.append(timeline_path)
|
|
if existing and len(existing) == len(media_paths) and not bool(args.get("force_reprocess", False)):
|
|
message = "# Voice Timeline Cache\n\n"
|
|
message += "Reaproveitando análise de voz existente. Nada foi reprocessado.\n\n"
|
|
for timeline_path in existing:
|
|
message += f"- **Timeline JSON**: {timeline_path}\n"
|
|
shared.emit({
|
|
"ok": True,
|
|
"path": path,
|
|
"reused": True,
|
|
"timelines": [str(p) for p in existing],
|
|
"message": message,
|
|
})
|
|
return 0
|
|
|
|
for mp in media_paths:
|
|
transcript_path = _transcript_json_path(mp, output_dir)
|
|
reused_prefix = ""
|
|
if _load_cached_transcript(transcript_path) is not None:
|
|
reused_prefix = f"# Cache\n\nReaproveitando transcrição existente: `{transcript_path}`\n\n"
|
|
try:
|
|
contents = asyncio.run(handle_build_voice_timeline({
|
|
"media_path": mp, "model": model, "language": language,
|
|
"hf_token": token, "num_speakers": num_speakers,
|
|
"output_dir": output_dir, "rotation": rotations.get(mp, 0.0),
|
|
}))
|
|
except Exception as exc:
|
|
shared.emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"})
|
|
return 1
|
|
messages.append(reused_prefix + "\n".join(getattr(c, "text", str(c)) for c in contents))
|
|
|
|
shared.emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)})
|
|
return 0
|
|
|
|
def cmd_acoustics_capability(args: dict) -> int:
|
|
"""Whether librosa (pitch/energy extraction) is installed in this venv.
|
|
|
|
Surfaces `features_capability()` — previously computed but never
|
|
exposed to the app, so `layers.acoustics: false` in a voice timeline
|
|
had no explanation the user could act on.
|
|
"""
|
|
from fcpxml.voice_features import features_capability
|
|
ok, msg = features_capability()
|
|
shared.emit({"ok": True, "available": ok, "message": msg})
|
|
return 0
|
|
|
|
def cmd_voice_analysis(args: dict) -> int:
|
|
"""Read the persisted voice-analysis settings (energy/emphasis/emotion)."""
|
|
config = load_voice_analysis_config()
|
|
shared.emit({"ok": True, **config, "emphasis_threshold": config["emphasis_floor"]})
|
|
return 0
|
|
|
|
def cmd_set_voice_analysis(args: dict) -> int:
|
|
"""Persist voice-analysis settings. Only the given fields change."""
|
|
weights = args.get("emphasis_weights")
|
|
config = save_voice_analysis_config(
|
|
energy_threshold=args.get("energy_threshold"),
|
|
emphasis_weights=weights if isinstance(weights, dict) else None,
|
|
emphasis_floor=args.get("emphasis_threshold"),
|
|
emotion_enabled=args.get("emotion_enabled"),
|
|
emotion_sensitivity=args.get("emotion_sensitivity"),
|
|
zoom_scale=args.get("zoom_scale"),
|
|
zoom_mode=args.get("zoom_mode"),
|
|
zoom_ease_in=args.get("zoom_ease_in"),
|
|
zoom_ease_out=args.get("zoom_ease_out"),
|
|
)
|
|
shared.emit({"ok": True, **config})
|
|
return 0
|
|
|
|
def cmd_apply_voice_actions(args: dict) -> int:
|
|
"""Apply a decision list (cuts/zooms/texts/markers) to the project XML.
|
|
|
|
The list is produced by a model reading the _voice_timeline.json — this
|
|
is the step that turns those decisions into an edit, and the one the
|
|
batch chain was missing: without it the app could measure the voice and
|
|
caption the result, but never cut by it.
|
|
|
|
`actions_path` points at the JSON; either a bare list or the
|
|
``{"actions": [...]}`` wrapper the skill emits is accepted. Times stay in
|
|
ORIGINAL source seconds — the handler resolves cuts first and shifts
|
|
everything else itself.
|
|
"""
|
|
path = str(args.get("path", ""))
|
|
if not path or not Path(path).exists():
|
|
shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
|
|
return 1
|
|
|
|
actions = args.get("actions")
|
|
if actions is None:
|
|
actions_path = str(args.get("actions_path", ""))
|
|
if not actions_path or not Path(actions_path).exists():
|
|
shared.emit({"ok": False, "error": "Arquivo de decisões (JSON) não encontrado."})
|
|
return 1
|
|
try:
|
|
with open(actions_path, encoding="utf-8") as fh:
|
|
loaded = json.load(fh)
|
|
except (OSError, ValueError) as exc:
|
|
shared.emit({"ok": False, "error": f"Erro ao ler as decisões: {exc}"})
|
|
return 1
|
|
actions = loaded.get("actions") if isinstance(loaded, dict) else loaded
|
|
|
|
# The documented output format is {"source": ..., "actions": [...]} —
|
|
# callers passing that whole object inline (e.g. the wizard pasting the
|
|
# skill's JSON verbatim) need the same unwrap the actions_path branch
|
|
# above already does, or a well-formed payload gets rejected as
|
|
# "malformed" for having one extra layer of nesting.
|
|
if isinstance(actions, dict):
|
|
actions = actions.get("actions")
|
|
|
|
if not isinstance(actions, list) or not actions:
|
|
shared.emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."})
|
|
return 1
|
|
|
|
from server import handle_apply_voice_actions
|
|
|
|
try:
|
|
contents = asyncio.run(handle_apply_voice_actions({
|
|
"filepath": path,
|
|
"actions": actions,
|
|
"output_dir": args.get("output_dir"),
|
|
}))
|
|
except Exception as exc:
|
|
shared.emit({"ok": False, "error": f"Falha ao aplicar as decisões: {exc}"})
|
|
return 1
|
|
|
|
message = "\n".join(getattr(c, "text", str(c)) for c in contents)
|
|
# The handler reports dropped/rejected actions individually; hand the
|
|
# whole report back so the app can surface them instead of only the count.
|
|
out_path = path
|
|
for line in message.splitlines():
|
|
if line.startswith("- **Saved to**:"):
|
|
out_path = line.split("`")[1] if "`" in line else path
|
|
break
|
|
shared.emit({"ok": True, "path": out_path, "message": message})
|
|
return 0
|
|
|
|
|
|
def cmd_generate_voice_script(args: dict) -> int:
|
|
"""Run the ENTIRE voice-edit pass against a LOCAL model, inside the engine.
|
|
|
|
Transcribe (cached) -> build the voice timeline -> hand it to a local
|
|
Ollama model (Gemma 3 / Llama) that directs the edit -> return the readable
|
|
script (roteiro) and the action JSON, and optionally apply to a FCPXML. No
|
|
wizard, no copy-paste: the model's decisions are validated and applied by
|
|
the same pipeline the rules engine uses.
|
|
|
|
Args (all optional except one of ``media_path`` / ``voice_timeline``):
|
|
media_path audio/video to analyze and direct (required when there is
|
|
no voice_timeline yet)
|
|
voice_timeline path to an existing _voice_timeline.json; when given the
|
|
analysis is reused and media_path is not required
|
|
filepath optional FCPXML to apply the decisions to (non-destructive)
|
|
model local model Ollama serves (default gemma3:12b)
|
|
base_url Ollama base URL (default http://localhost:11434)
|
|
model_size whisper size if transcription is needed
|
|
language ISO language hint for transcription
|
|
hf_token HuggingFace token for diarization
|
|
num_speakers known speaker count, if any
|
|
output_dir folder for the timeline/review/actions JSON
|
|
apply_to_fcpxml apply to filepath when given (default true)
|
|
-> {"ok": true, "message": "...", "roteiro_path", "actions_path",
|
|
"applied_path"} or {"ok": false, "error": "..."}
|
|
"""
|
|
media_path = str(args.get("media_path", ""))
|
|
voice_timeline = str(args.get("voice_timeline", ""))
|
|
if not voice_timeline and (not media_path or not Path(media_path).exists()):
|
|
shared.emit({"ok": False, "error": "Arquivo de mídia não encontrado (informe media_path ou voice_timeline)."})
|
|
return 1
|
|
|
|
from server import handle_generate_voice_script
|
|
|
|
try:
|
|
contents = asyncio.run(handle_generate_voice_script({
|
|
"media_path": media_path,
|
|
"voice_timeline": args.get("voice_timeline"),
|
|
"filepath": args.get("filepath"),
|
|
"model": args.get("model"),
|
|
"base_url": args.get("base_url"),
|
|
"model_size": args.get("model_size"),
|
|
"language": args.get("language"),
|
|
"hf_token": args.get("hf_token"),
|
|
"num_speakers": args.get("num_speakers"),
|
|
"output_dir": args.get("output_dir"),
|
|
"apply_to_fcpxml": args.get("apply_to_fcpxml", True),
|
|
}))
|
|
except Exception as exc:
|
|
shared.emit({"ok": False, "error": f"Falha ao gerar roteiro por IA local: {exc}"})
|
|
return 1
|
|
|
|
message = "\n".join(getattr(c, "text", str(c)) for c in contents)
|
|
|
|
def _path_after(label: str) -> str:
|
|
m = re.search(rf"\*\*{label}\*\*: (.+)", message)
|
|
return m.group(1).strip() if m else ""
|
|
|
|
roteiro_path = _path_after(r"Roteiro \(legível\)")
|
|
actions_path = _path_after("Ações JSON")
|
|
applied_path = ""
|
|
for line in message.splitlines():
|
|
if line.startswith("- **Saved to**:"):
|
|
applied_path = line.split("`")[1] if "`" in line else ""
|
|
break
|
|
shared.emit({
|
|
"ok": True,
|
|
"message": message,
|
|
"roteiro_path": roteiro_path,
|
|
"actions_path": actions_path,
|
|
"applied_path": applied_path,
|
|
})
|
|
return 0
|
|
|
|
|
|
def cmd_list_ollama_models(args: dict) -> int:
|
|
"""List the models Ollama currently serves, for the app's model picker.
|
|
|
|
Args:
|
|
base_url Ollama base URL (default http://localhost:11434)
|
|
-> {"ok": true, "models": ["gemma3:12b", ...]} (empty list if Ollama
|
|
is unreachable, so the UI can fall back to a text field)
|
|
"""
|
|
from fcpxml.llm_local import list_ollama_models
|
|
|
|
base_url = str(args.get("base_url") or "http://localhost:11434")
|
|
models = list_ollama_models(base_url=base_url)
|
|
shared.emit({"ok": True, "models": models})
|
|
return 0
|