feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
711c397dfe
commit
7b5aed79ee
@@ -105,6 +105,20 @@ def _project_media_paths(path: str) -> list[str]:
|
||||
media_paths.append(mp)
|
||||
return media_paths
|
||||
|
||||
def _project_media_rotations(path: str) -> dict[str, float]:
|
||||
"""Degrees each source media was rotated by via a Transform filter on its
|
||||
clip in the FCPXML — keyed by the same resolved media path
|
||||
``_project_media_paths`` returns, so the two can be joined by media_path."""
|
||||
proj = parse_fcpxml(path)
|
||||
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
|
||||
rotations: dict[str, float] = {}
|
||||
if tl is not None:
|
||||
for clip in getattr(tl, "clips", []):
|
||||
mp = media_src_to_path(clip.media_path or "")
|
||||
if mp and clip.rotation:
|
||||
rotations[mp] = clip.rotation
|
||||
return rotations
|
||||
|
||||
def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path:
|
||||
p = Path(media_path)
|
||||
if output_dir:
|
||||
|
||||
+95
-1
@@ -7,6 +7,7 @@ from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from fcpxml.model_manager import (
|
||||
@@ -23,6 +24,7 @@ from .shared import (
|
||||
_load_cached_transcript,
|
||||
_load_cached_voice_timeline,
|
||||
_project_media_paths,
|
||||
_project_media_rotations,
|
||||
_transcript_json_path,
|
||||
_voice_timeline_json_path,
|
||||
)
|
||||
@@ -54,6 +56,7 @@ def cmd_analyze_voice(args: dict) -> int:
|
||||
|
||||
try:
|
||||
media_paths = _project_media_paths(path)
|
||||
rotations = _project_media_rotations(path)
|
||||
except Exception as exc:
|
||||
shared.emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
|
||||
return 1
|
||||
@@ -93,7 +96,7 @@ def cmd_analyze_voice(args: dict) -> int:
|
||||
contents = asyncio.run(handle_build_voice_timeline({
|
||||
"media_path": mp, "model": model, "language": language,
|
||||
"hf_token": token, "num_speakers": num_speakers,
|
||||
"output_dir": output_dir,
|
||||
"output_dir": output_dir, "rotation": rotations.get(mp, 0.0),
|
||||
}))
|
||||
except Exception as exc:
|
||||
shared.emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"})
|
||||
@@ -204,3 +207,94 @@ def cmd_apply_voice_actions(args: dict) -> int:
|
||||
break
|
||||
shared.emit({"ok": True, "path": out_path, "message": message})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_generate_voice_script(args: dict) -> int:
|
||||
"""Run the ENTIRE voice-edit pass against a LOCAL model, inside the engine.
|
||||
|
||||
Transcribe (cached) -> build the voice timeline -> hand it to a local
|
||||
Ollama model (Gemma 3 / Llama) that directs the edit -> return the readable
|
||||
script (roteiro) and the action JSON, and optionally apply to a FCPXML. No
|
||||
wizard, no copy-paste: the model's decisions are validated and applied by
|
||||
the same pipeline the rules engine uses.
|
||||
|
||||
Args (all optional except one of ``media_path`` / ``voice_timeline``):
|
||||
media_path audio/video to analyze and direct (required when there is
|
||||
no voice_timeline yet)
|
||||
voice_timeline path to an existing _voice_timeline.json; when given the
|
||||
analysis is reused and media_path is not required
|
||||
filepath optional FCPXML to apply the decisions to (non-destructive)
|
||||
model local model Ollama serves (default gemma3:12b)
|
||||
base_url Ollama base URL (default http://localhost:11434)
|
||||
model_size whisper size if transcription is needed
|
||||
language ISO language hint for transcription
|
||||
hf_token HuggingFace token for diarization
|
||||
num_speakers known speaker count, if any
|
||||
output_dir folder for the timeline/review/actions JSON
|
||||
apply_to_fcpxml apply to filepath when given (default true)
|
||||
-> {"ok": true, "message": "...", "roteiro_path", "actions_path",
|
||||
"applied_path"} or {"ok": false, "error": "..."}
|
||||
"""
|
||||
media_path = str(args.get("media_path", ""))
|
||||
voice_timeline = str(args.get("voice_timeline", ""))
|
||||
if not voice_timeline and (not media_path or not Path(media_path).exists()):
|
||||
shared.emit({"ok": False, "error": "Arquivo de mídia não encontrado (informe media_path ou voice_timeline)."})
|
||||
return 1
|
||||
|
||||
from server import handle_generate_voice_script
|
||||
|
||||
try:
|
||||
contents = asyncio.run(handle_generate_voice_script({
|
||||
"media_path": media_path,
|
||||
"voice_timeline": args.get("voice_timeline"),
|
||||
"filepath": args.get("filepath"),
|
||||
"model": args.get("model"),
|
||||
"base_url": args.get("base_url"),
|
||||
"model_size": args.get("model_size"),
|
||||
"language": args.get("language"),
|
||||
"hf_token": args.get("hf_token"),
|
||||
"num_speakers": args.get("num_speakers"),
|
||||
"output_dir": args.get("output_dir"),
|
||||
"apply_to_fcpxml": args.get("apply_to_fcpxml", True),
|
||||
}))
|
||||
except Exception as exc:
|
||||
shared.emit({"ok": False, "error": f"Falha ao gerar roteiro por IA local: {exc}"})
|
||||
return 1
|
||||
|
||||
message = "\n".join(getattr(c, "text", str(c)) for c in contents)
|
||||
|
||||
def _path_after(label: str) -> str:
|
||||
m = re.search(rf"\*\*{label}\*\*: (.+)", message)
|
||||
return m.group(1).strip() if m else ""
|
||||
|
||||
roteiro_path = _path_after(r"Roteiro \(legível\)")
|
||||
actions_path = _path_after("Ações JSON")
|
||||
applied_path = ""
|
||||
for line in message.splitlines():
|
||||
if line.startswith("- **Saved to**:"):
|
||||
applied_path = line.split("`")[1] if "`" in line else ""
|
||||
break
|
||||
shared.emit({
|
||||
"ok": True,
|
||||
"message": message,
|
||||
"roteiro_path": roteiro_path,
|
||||
"actions_path": actions_path,
|
||||
"applied_path": applied_path,
|
||||
})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_list_ollama_models(args: dict) -> int:
|
||||
"""List the models Ollama currently serves, for the app's model picker.
|
||||
|
||||
Args:
|
||||
base_url Ollama base URL (default http://localhost:11434)
|
||||
-> {"ok": true, "models": ["gemma3:12b", ...]} (empty list if Ollama
|
||||
is unreachable, so the UI can fall back to a text field)
|
||||
"""
|
||||
from fcpxml.llm_local import list_ollama_models
|
||||
|
||||
base_url = str(args.get("base_url") or "http://localhost:11434")
|
||||
models = list_ollama_models(base_url=base_url)
|
||||
shared.emit({"ok": True, "models": models})
|
||||
return 0
|
||||
|
||||
@@ -68,6 +68,27 @@ Commands:
|
||||
-> {"ok": true, "review_path", "actions_path", "emphasis_count",
|
||||
"removed_count"}
|
||||
|
||||
generate_voice_script {"media_path": "...", "voice_timeline": "...", "filepath": "...",
|
||||
"model": "gemma3:12b", "base_url": "http://localhost:11434",
|
||||
"model_size": "base", "language": "pt"|"auto"|null,
|
||||
"hf_token": "..."|null, "num_speakers": ""|null,
|
||||
"output_dir": "...", "apply_to_fcpxml": true}
|
||||
The WHOLE voice-edit pass run INSIDE the engine against a LOCAL model
|
||||
(Ollama running Gemma 3 / Llama) — no wizard, no copy-paste. If
|
||||
`voice_timeline` is given (the analysis from an earlier step), it is
|
||||
reused and the transcription/acoustics are skipped; otherwise
|
||||
`media_path` is transcribed and analyzed. Then the local model directs
|
||||
the edit -> returns the readable script (roteiro) + the action JSON,
|
||||
and optionally applies to `filepath` (FCPXML, non-destructive).
|
||||
-> {"ok": true, "message", "roteiro_path", "actions_path", "applied_path"}
|
||||
or {"ok": false, "error": "..."}
|
||||
|
||||
list_ollama_models {"base_url": "http://localhost:11434"}
|
||||
Lists the models Ollama currently serves, for the app's model picker
|
||||
in step 4 (generate script by local AI). Empty list if Ollama is
|
||||
unreachable, so the UI falls back to a free-text field.
|
||||
-> {"ok": true, "models": ["gemma3:12b", ...]}
|
||||
|
||||
dynamic_subtitle_config {}
|
||||
-> {"ok": true, "band_height", "block_center_y", "line_gap", "font",
|
||||
"font_size", "emphasis_font", "emphasis_face", "emphasis_size",
|
||||
@@ -220,6 +241,8 @@ def main() -> int:
|
||||
"plain_subtitle_config": subtitles.cmd_plain_subtitle_config,
|
||||
"set_plain_subtitle_config": subtitles.cmd_set_plain_subtitle_config,
|
||||
"apply_voice_actions": voice.cmd_apply_voice_actions,
|
||||
"generate_voice_script": voice.cmd_generate_voice_script,
|
||||
"list_ollama_models": voice.cmd_list_ollama_models,
|
||||
"build_phrase_review": review.cmd_build_phrase_review,
|
||||
"save_phrase_review": review.cmd_save_phrase_review,
|
||||
"project_config": project.cmd_project_config,
|
||||
|
||||
Reference in New Issue
Block a user