feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
711c397dfe
commit
7b5aed79ee
@@ -0,0 +1,259 @@
|
||||
"""Tests for the local LLM integration (Ollama): prompt building, JSON
|
||||
extraction, and turning a model response into validated voice actions.
|
||||
|
||||
The model itself is mocked — these tests cover the client contract
|
||||
(extraction, validation, message shape) without a running Ollama. A real
|
||||
end-to-end run lives in the manual test harness (see ENGINE notes) because it
|
||||
needs the local model server.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml import llm_local
|
||||
from fcpxml.llm_local import (
|
||||
_extract_json,
|
||||
build_edit_messages,
|
||||
generate_voice_actions,
|
||||
)
|
||||
from fcpxml.voice_actions import parse_actions
|
||||
|
||||
|
||||
def _fake_timeline() -> dict:
|
||||
"""A minimal but well-formed voice timeline (matches build_voice_timeline)."""
|
||||
return {
|
||||
"version": "1.0",
|
||||
"source": "demo.mp4",
|
||||
"rotation": 0.0,
|
||||
"language": "pt",
|
||||
"layers": {
|
||||
"transcript": True,
|
||||
"acoustics": True,
|
||||
"speakers": False,
|
||||
"emotion": False,
|
||||
"alignment": False,
|
||||
},
|
||||
"scales": {},
|
||||
"summary": {
|
||||
"duration": 12.0,
|
||||
"speaker_count": 1,
|
||||
"segment_count": 2,
|
||||
"word_count": 4,
|
||||
"avg_emphasis": 0.4,
|
||||
"peak_selection": "top 2%",
|
||||
"peak_count": 1,
|
||||
"peak_moments": [],
|
||||
},
|
||||
"speakers": [
|
||||
{
|
||||
"id": "SPEAKER_00",
|
||||
"name": "Speaker 1",
|
||||
"speaking_seconds": 12.0,
|
||||
"share": 1.0,
|
||||
"segment_count": 2,
|
||||
"avg_segment": 6.0,
|
||||
"word_count": 4,
|
||||
"samples": [],
|
||||
}
|
||||
],
|
||||
"segments": [
|
||||
{
|
||||
"start": 0.0,
|
||||
"end": 6.0,
|
||||
"speaker": "SPEAKER_00",
|
||||
"text": "Hoje vamos falar de mastopexia.",
|
||||
"gap_before": 0.0,
|
||||
"take_boundary": False,
|
||||
"avg_energy": 0.5,
|
||||
"peak_emphasis": 0.6,
|
||||
"emotion": "neutral",
|
||||
"emotion_confidence": 0.0,
|
||||
"arousal": 0.5,
|
||||
"valence": 0.6,
|
||||
"words": [
|
||||
{"text": "Hoje", "start": 0.1, "end": 0.5, "speaker": "SPEAKER_00",
|
||||
"energy": 0.3, "pitch_delta": 0.2, "rate_delta": 0.1, "pause_before": 0.1,
|
||||
"emphasis": 0.3, "emotion": "neutral", "emotion_confidence": 0.0,
|
||||
"arousal": 0.4, "valence": 0.6, "energy_raw": 0.3, "pitch_hz": 120.0},
|
||||
{"text": "vamos", "start": 0.6, "end": 0.9, "speaker": "SPEAKER_00",
|
||||
"energy": 0.4, "pitch_delta": 0.3, "rate_delta": 0.2, "pause_before": 0.0,
|
||||
"emphasis": 0.5, "emotion": "neutral", "emotion_confidence": 0.0,
|
||||
"arousal": 0.5, "valence": 0.6, "energy_raw": 0.4, "pitch_hz": 125.0},
|
||||
],
|
||||
},
|
||||
{
|
||||
"start": 6.0,
|
||||
"end": 12.0,
|
||||
"speaker": "SPEAKER_00",
|
||||
"text": "E aí cara, tá gravando?",
|
||||
"gap_before": 0.2,
|
||||
"take_boundary": False,
|
||||
"avg_energy": 0.4,
|
||||
"peak_emphasis": 0.3,
|
||||
"emotion": "neutral",
|
||||
"emotion_confidence": 0.0,
|
||||
"arousal": 0.4,
|
||||
"valence": 0.5,
|
||||
"words": [
|
||||
{"text": "E", "start": 6.1, "end": 6.3, "speaker": "SPEAKER_00",
|
||||
"energy": 0.3, "pitch_delta": 0.2, "rate_delta": 0.1, "pause_before": 0.1,
|
||||
"emphasis": 0.3, "emotion": "neutral", "emotion_confidence": 0.0,
|
||||
"arousal": 0.4, "valence": 0.5, "energy_raw": 0.3, "pitch_hz": 120.0},
|
||||
{"text": "cara", "start": 6.4, "end": 6.7, "speaker": "SPEAKER_00",
|
||||
"energy": 0.4, "pitch_delta": 0.3, "rate_delta": 0.2, "pause_before": 0.0,
|
||||
"emphasis": 0.4, "emotion": "neutral", "emotion_confidence": 0.0,
|
||||
"arousal": 0.4, "valence": 0.5, "energy_raw": 0.4, "pitch_hz": 122.0},
|
||||
],
|
||||
},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def test_extract_json_unfenced():
|
||||
text = '{"source": "x", "actions": []}'
|
||||
assert _extract_json(text) == {"source": "x", "actions": []}
|
||||
|
||||
|
||||
def test_extract_json_with_fence():
|
||||
text = '```json\n{"source": "x", "actions": [{"kind": "cut", "start": 1, "end": 2}]}\n```'
|
||||
data = _extract_json(text)
|
||||
assert data["source"] == "x"
|
||||
assert data["actions"][0]["kind"] == "cut"
|
||||
|
||||
|
||||
def test_extract_json_with_prose_around():
|
||||
text = 'Aqui está:\n{"source": "x", "actions": []}\nfim.'
|
||||
assert _extract_json(text) == {"source": "x", "actions": []}
|
||||
|
||||
|
||||
def test_extract_json_invalid_returns_none():
|
||||
assert _extract_json("no json here") is None
|
||||
assert _extract_json("") is None
|
||||
|
||||
|
||||
def test_extract_json_unwraps_wrapped_list():
|
||||
# Some models wrap the expected {"source","actions"} object in a
|
||||
# single-element list; the actions must still be found.
|
||||
text = '```json\n[{"source": "x", "actions": [{"kind": "cut", "start": 1, "end": 2}]}]\n```'
|
||||
data = _extract_json(text)
|
||||
assert isinstance(data, dict)
|
||||
assert data["source"] == "x"
|
||||
assert data["actions"][0]["kind"] == "cut"
|
||||
|
||||
|
||||
def test_build_edit_messages_shape():
|
||||
system, user = build_edit_messages(_fake_timeline())
|
||||
assert "editor" in system.lower()
|
||||
assert "demo.mp4" in user
|
||||
assert "segments" in user
|
||||
assert "SOMENTE" in user
|
||||
|
||||
|
||||
def test_generate_voice_actions_parses_model_json(monkeypatch):
|
||||
canned = json.dumps({
|
||||
"source": "demo.mp4",
|
||||
"actions": [
|
||||
{"kind": "cut", "start": 6.0, "end": 12.0,
|
||||
"reason": "papo casual com a equipe"},
|
||||
{"kind": "zoom", "start": 0.1, "end": 0.9, "params": {"scale": 1.3},
|
||||
"reason": "ênfase em 'vamos'"},
|
||||
],
|
||||
})
|
||||
monkeypatch.setattr(llm_local, "ollama_chat", lambda **kwargs: canned)
|
||||
|
||||
result = generate_voice_actions(_fake_timeline(), model="test")
|
||||
actions, errors = parse_actions([a.as_dict() for a in result["actions"]])
|
||||
assert errors == []
|
||||
assert len(actions) == 2
|
||||
assert {a.kind for a in actions} == {"cut", "zoom"}
|
||||
assert result["errors"] == []
|
||||
|
||||
|
||||
def test_generate_voice_actions_reports_bad_rows(monkeypatch):
|
||||
canned = json.dumps({
|
||||
"source": "demo.mp4",
|
||||
"actions": [
|
||||
{"kind": "cut", "start": 6.0, "end": 12.0, "reason": "ok"},
|
||||
{"kind": "bogus", "start": 1, "end": 2},
|
||||
{"kind": "zoom", "start": 0.1, "end": 0.05},
|
||||
],
|
||||
})
|
||||
monkeypatch.setattr(llm_local, "ollama_chat", lambda **kwargs: canned)
|
||||
|
||||
result = generate_voice_actions(_fake_timeline(), model="test")
|
||||
assert len(result["actions"]) == 1
|
||||
assert result["actions"][0].kind == "cut"
|
||||
assert len(result["errors"]) >= 2
|
||||
|
||||
|
||||
def test_generate_voice_actions_handles_transport_error(monkeypatch):
|
||||
def _boom(**kwargs):
|
||||
raise RuntimeError("ollama down")
|
||||
monkeypatch.setattr(llm_local, "ollama_chat", _boom)
|
||||
|
||||
result = generate_voice_actions(_fake_timeline(), model="test")
|
||||
assert result["actions"] == []
|
||||
assert any("ollama" in e.lower() for e in result["errors"])
|
||||
|
||||
|
||||
def test_build_edit_messages_is_compact():
|
||||
"""The prompt must drop the heavy per-word audio features so a real
|
||||
recording fits in the model context (the 47k-token dump made Ollama drop
|
||||
the connection)."""
|
||||
import json as _json
|
||||
|
||||
timeline = _fake_timeline()
|
||||
# pad with the kind of audio detail a real timeline carries
|
||||
timeline["segments"][0]["words"][0]["samples"] = [0.1, 0.2, 0.3]
|
||||
timeline["speakers"][0]["samples"] = [1, 2, 3]
|
||||
raw = _json.dumps(timeline, ensure_ascii=False)
|
||||
system, user = build_edit_messages(timeline)
|
||||
assert "demo.mp4" in user and "segments" in user
|
||||
# projected prompt is materially smaller than the raw timeline
|
||||
assert len(user) < len(raw)
|
||||
# heavy fields we deliberately drop never reach the model
|
||||
assert '"samples"' not in user
|
||||
assert '"energy_raw"' not in user
|
||||
assert '"pitch_hz"' not in user
|
||||
|
||||
|
||||
def test_ollama_chat_wraps_empty_response(monkeypatch):
|
||||
"""A dropped connection that yields an empty body must surface as a clear
|
||||
RuntimeError (not an unhandled JSONDecodeError crashing the pipeline)."""
|
||||
|
||||
class _Resp:
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
def json(self):
|
||||
raise ValueError("Expecting value: line 1 column 1")
|
||||
|
||||
monkeypatch.setattr(llm_local.httpx, "post", lambda *a, **k: _Resp())
|
||||
try:
|
||||
llm_local.ollama_chat(model="test", messages=[{"role": "user", "content": "x"}])
|
||||
except RuntimeError as exc:
|
||||
assert "Falha ao falar com o modelo local" in str(exc)
|
||||
else:
|
||||
raise AssertionError("expected RuntimeError")
|
||||
|
||||
|
||||
def test_list_ollama_models_parses_tags(monkeypatch):
|
||||
class _Resp:
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
def json(self):
|
||||
return {"models": [{"name": "gemma3:4b"}, {"name": "gemma3:12b"}]}
|
||||
|
||||
monkeypatch.setattr(llm_local.httpx, "get", lambda *a, **k: _Resp())
|
||||
assert llm_local.list_ollama_models() == ["gemma3:12b", "gemma3:4b"]
|
||||
|
||||
|
||||
def test_list_ollama_models_returns_empty_on_error(monkeypatch):
|
||||
def _boom(*a, **k):
|
||||
raise RuntimeError("ollama down")
|
||||
|
||||
monkeypatch.setattr(llm_local.httpx, "get", _boom)
|
||||
assert llm_local.list_ollama_models() == []
|
||||
|
||||
Reference in New Issue
Block a user