feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão

Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-21 18:26:04 -04:00
co-authored by Claude Sonnet 5
parent 711c397dfe
commit 7b5aed79ee
36 changed files with 2922 additions and 624 deletions
+259
View File
@@ -0,0 +1,259 @@
"""Tests for the local LLM integration (Ollama): prompt building, JSON
extraction, and turning a model response into validated voice actions.
The model itself is mocked — these tests cover the client contract
(extraction, validation, message shape) without a running Ollama. A real
end-to-end run lives in the manual test harness (see ENGINE notes) because it
needs the local model server.
"""
import json
import pytest
from fcpxml import llm_local
from fcpxml.llm_local import (
_extract_json,
build_edit_messages,
generate_voice_actions,
)
from fcpxml.voice_actions import parse_actions
def _fake_timeline() -> dict:
"""A minimal but well-formed voice timeline (matches build_voice_timeline)."""
return {
"version": "1.0",
"source": "demo.mp4",
"rotation": 0.0,
"language": "pt",
"layers": {
"transcript": True,
"acoustics": True,
"speakers": False,
"emotion": False,
"alignment": False,
},
"scales": {},
"summary": {
"duration": 12.0,
"speaker_count": 1,
"segment_count": 2,
"word_count": 4,
"avg_emphasis": 0.4,
"peak_selection": "top 2%",
"peak_count": 1,
"peak_moments": [],
},
"speakers": [
{
"id": "SPEAKER_00",
"name": "Speaker 1",
"speaking_seconds": 12.0,
"share": 1.0,
"segment_count": 2,
"avg_segment": 6.0,
"word_count": 4,
"samples": [],
}
],
"segments": [
{
"start": 0.0,
"end": 6.0,
"speaker": "SPEAKER_00",
"text": "Hoje vamos falar de mastopexia.",
"gap_before": 0.0,
"take_boundary": False,
"avg_energy": 0.5,
"peak_emphasis": 0.6,
"emotion": "neutral",
"emotion_confidence": 0.0,
"arousal": 0.5,
"valence": 0.6,
"words": [
{"text": "Hoje", "start": 0.1, "end": 0.5, "speaker": "SPEAKER_00",
"energy": 0.3, "pitch_delta": 0.2, "rate_delta": 0.1, "pause_before": 0.1,
"emphasis": 0.3, "emotion": "neutral", "emotion_confidence": 0.0,
"arousal": 0.4, "valence": 0.6, "energy_raw": 0.3, "pitch_hz": 120.0},
{"text": "vamos", "start": 0.6, "end": 0.9, "speaker": "SPEAKER_00",
"energy": 0.4, "pitch_delta": 0.3, "rate_delta": 0.2, "pause_before": 0.0,
"emphasis": 0.5, "emotion": "neutral", "emotion_confidence": 0.0,
"arousal": 0.5, "valence": 0.6, "energy_raw": 0.4, "pitch_hz": 125.0},
],
},
{
"start": 6.0,
"end": 12.0,
"speaker": "SPEAKER_00",
"text": "E aí cara, tá gravando?",
"gap_before": 0.2,
"take_boundary": False,
"avg_energy": 0.4,
"peak_emphasis": 0.3,
"emotion": "neutral",
"emotion_confidence": 0.0,
"arousal": 0.4,
"valence": 0.5,
"words": [
{"text": "E", "start": 6.1, "end": 6.3, "speaker": "SPEAKER_00",
"energy": 0.3, "pitch_delta": 0.2, "rate_delta": 0.1, "pause_before": 0.1,
"emphasis": 0.3, "emotion": "neutral", "emotion_confidence": 0.0,
"arousal": 0.4, "valence": 0.5, "energy_raw": 0.3, "pitch_hz": 120.0},
{"text": "cara", "start": 6.4, "end": 6.7, "speaker": "SPEAKER_00",
"energy": 0.4, "pitch_delta": 0.3, "rate_delta": 0.2, "pause_before": 0.0,
"emphasis": 0.4, "emotion": "neutral", "emotion_confidence": 0.0,
"arousal": 0.4, "valence": 0.5, "energy_raw": 0.4, "pitch_hz": 122.0},
],
},
],
}
def test_extract_json_unfenced():
text = '{"source": "x", "actions": []}'
assert _extract_json(text) == {"source": "x", "actions": []}
def test_extract_json_with_fence():
text = '```json\n{"source": "x", "actions": [{"kind": "cut", "start": 1, "end": 2}]}\n```'
data = _extract_json(text)
assert data["source"] == "x"
assert data["actions"][0]["kind"] == "cut"
def test_extract_json_with_prose_around():
text = 'Aqui está:\n{"source": "x", "actions": []}\nfim.'
assert _extract_json(text) == {"source": "x", "actions": []}
def test_extract_json_invalid_returns_none():
assert _extract_json("no json here") is None
assert _extract_json("") is None
def test_extract_json_unwraps_wrapped_list():
# Some models wrap the expected {"source","actions"} object in a
# single-element list; the actions must still be found.
text = '```json\n[{"source": "x", "actions": [{"kind": "cut", "start": 1, "end": 2}]}]\n```'
data = _extract_json(text)
assert isinstance(data, dict)
assert data["source"] == "x"
assert data["actions"][0]["kind"] == "cut"
def test_build_edit_messages_shape():
system, user = build_edit_messages(_fake_timeline())
assert "editor" in system.lower()
assert "demo.mp4" in user
assert "segments" in user
assert "SOMENTE" in user
def test_generate_voice_actions_parses_model_json(monkeypatch):
canned = json.dumps({
"source": "demo.mp4",
"actions": [
{"kind": "cut", "start": 6.0, "end": 12.0,
"reason": "papo casual com a equipe"},
{"kind": "zoom", "start": 0.1, "end": 0.9, "params": {"scale": 1.3},
"reason": "ênfase em 'vamos'"},
],
})
monkeypatch.setattr(llm_local, "ollama_chat", lambda **kwargs: canned)
result = generate_voice_actions(_fake_timeline(), model="test")
actions, errors = parse_actions([a.as_dict() for a in result["actions"]])
assert errors == []
assert len(actions) == 2
assert {a.kind for a in actions} == {"cut", "zoom"}
assert result["errors"] == []
def test_generate_voice_actions_reports_bad_rows(monkeypatch):
canned = json.dumps({
"source": "demo.mp4",
"actions": [
{"kind": "cut", "start": 6.0, "end": 12.0, "reason": "ok"},
{"kind": "bogus", "start": 1, "end": 2},
{"kind": "zoom", "start": 0.1, "end": 0.05},
],
})
monkeypatch.setattr(llm_local, "ollama_chat", lambda **kwargs: canned)
result = generate_voice_actions(_fake_timeline(), model="test")
assert len(result["actions"]) == 1
assert result["actions"][0].kind == "cut"
assert len(result["errors"]) >= 2
def test_generate_voice_actions_handles_transport_error(monkeypatch):
def _boom(**kwargs):
raise RuntimeError("ollama down")
monkeypatch.setattr(llm_local, "ollama_chat", _boom)
result = generate_voice_actions(_fake_timeline(), model="test")
assert result["actions"] == []
assert any("ollama" in e.lower() for e in result["errors"])
def test_build_edit_messages_is_compact():
"""The prompt must drop the heavy per-word audio features so a real
recording fits in the model context (the 47k-token dump made Ollama drop
the connection)."""
import json as _json
timeline = _fake_timeline()
# pad with the kind of audio detail a real timeline carries
timeline["segments"][0]["words"][0]["samples"] = [0.1, 0.2, 0.3]
timeline["speakers"][0]["samples"] = [1, 2, 3]
raw = _json.dumps(timeline, ensure_ascii=False)
system, user = build_edit_messages(timeline)
assert "demo.mp4" in user and "segments" in user
# projected prompt is materially smaller than the raw timeline
assert len(user) < len(raw)
# heavy fields we deliberately drop never reach the model
assert '"samples"' not in user
assert '"energy_raw"' not in user
assert '"pitch_hz"' not in user
def test_ollama_chat_wraps_empty_response(monkeypatch):
"""A dropped connection that yields an empty body must surface as a clear
RuntimeError (not an unhandled JSONDecodeError crashing the pipeline)."""
class _Resp:
def raise_for_status(self):
pass
def json(self):
raise ValueError("Expecting value: line 1 column 1")
monkeypatch.setattr(llm_local.httpx, "post", lambda *a, **k: _Resp())
try:
llm_local.ollama_chat(model="test", messages=[{"role": "user", "content": "x"}])
except RuntimeError as exc:
assert "Falha ao falar com o modelo local" in str(exc)
else:
raise AssertionError("expected RuntimeError")
def test_list_ollama_models_parses_tags(monkeypatch):
class _Resp:
def raise_for_status(self):
pass
def json(self):
return {"models": [{"name": "gemma3:4b"}, {"name": "gemma3:12b"}]}
monkeypatch.setattr(llm_local.httpx, "get", lambda *a, **k: _Resp())
assert llm_local.list_ollama_models() == ["gemma3:12b", "gemma3:4b"]
def test_list_ollama_models_returns_empty_on_error(monkeypatch):
def _boom(*a, **k):
raise RuntimeError("ollama down")
monkeypatch.setattr(llm_local.httpx, "get", _boom)
assert llm_local.list_ollama_models() == []