Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
260 lines
9.3 KiB
Python
260 lines
9.3 KiB
Python
"""Tests for the local LLM integration (Ollama): prompt building, JSON
|
|
extraction, and turning a model response into validated voice actions.
|
|
|
|
The model itself is mocked — these tests cover the client contract
|
|
(extraction, validation, message shape) without a running Ollama. A real
|
|
end-to-end run lives in the manual test harness (see ENGINE notes) because it
|
|
needs the local model server.
|
|
"""
|
|
|
|
import json
|
|
|
|
import pytest
|
|
|
|
from fcpxml import llm_local
|
|
from fcpxml.llm_local import (
|
|
_extract_json,
|
|
build_edit_messages,
|
|
generate_voice_actions,
|
|
)
|
|
from fcpxml.voice_actions import parse_actions
|
|
|
|
|
|
def _fake_timeline() -> dict:
|
|
"""A minimal but well-formed voice timeline (matches build_voice_timeline)."""
|
|
return {
|
|
"version": "1.0",
|
|
"source": "demo.mp4",
|
|
"rotation": 0.0,
|
|
"language": "pt",
|
|
"layers": {
|
|
"transcript": True,
|
|
"acoustics": True,
|
|
"speakers": False,
|
|
"emotion": False,
|
|
"alignment": False,
|
|
},
|
|
"scales": {},
|
|
"summary": {
|
|
"duration": 12.0,
|
|
"speaker_count": 1,
|
|
"segment_count": 2,
|
|
"word_count": 4,
|
|
"avg_emphasis": 0.4,
|
|
"peak_selection": "top 2%",
|
|
"peak_count": 1,
|
|
"peak_moments": [],
|
|
},
|
|
"speakers": [
|
|
{
|
|
"id": "SPEAKER_00",
|
|
"name": "Speaker 1",
|
|
"speaking_seconds": 12.0,
|
|
"share": 1.0,
|
|
"segment_count": 2,
|
|
"avg_segment": 6.0,
|
|
"word_count": 4,
|
|
"samples": [],
|
|
}
|
|
],
|
|
"segments": [
|
|
{
|
|
"start": 0.0,
|
|
"end": 6.0,
|
|
"speaker": "SPEAKER_00",
|
|
"text": "Hoje vamos falar de mastopexia.",
|
|
"gap_before": 0.0,
|
|
"take_boundary": False,
|
|
"avg_energy": 0.5,
|
|
"peak_emphasis": 0.6,
|
|
"emotion": "neutral",
|
|
"emotion_confidence": 0.0,
|
|
"arousal": 0.5,
|
|
"valence": 0.6,
|
|
"words": [
|
|
{"text": "Hoje", "start": 0.1, "end": 0.5, "speaker": "SPEAKER_00",
|
|
"energy": 0.3, "pitch_delta": 0.2, "rate_delta": 0.1, "pause_before": 0.1,
|
|
"emphasis": 0.3, "emotion": "neutral", "emotion_confidence": 0.0,
|
|
"arousal": 0.4, "valence": 0.6, "energy_raw": 0.3, "pitch_hz": 120.0},
|
|
{"text": "vamos", "start": 0.6, "end": 0.9, "speaker": "SPEAKER_00",
|
|
"energy": 0.4, "pitch_delta": 0.3, "rate_delta": 0.2, "pause_before": 0.0,
|
|
"emphasis": 0.5, "emotion": "neutral", "emotion_confidence": 0.0,
|
|
"arousal": 0.5, "valence": 0.6, "energy_raw": 0.4, "pitch_hz": 125.0},
|
|
],
|
|
},
|
|
{
|
|
"start": 6.0,
|
|
"end": 12.0,
|
|
"speaker": "SPEAKER_00",
|
|
"text": "E aí cara, tá gravando?",
|
|
"gap_before": 0.2,
|
|
"take_boundary": False,
|
|
"avg_energy": 0.4,
|
|
"peak_emphasis": 0.3,
|
|
"emotion": "neutral",
|
|
"emotion_confidence": 0.0,
|
|
"arousal": 0.4,
|
|
"valence": 0.5,
|
|
"words": [
|
|
{"text": "E", "start": 6.1, "end": 6.3, "speaker": "SPEAKER_00",
|
|
"energy": 0.3, "pitch_delta": 0.2, "rate_delta": 0.1, "pause_before": 0.1,
|
|
"emphasis": 0.3, "emotion": "neutral", "emotion_confidence": 0.0,
|
|
"arousal": 0.4, "valence": 0.5, "energy_raw": 0.3, "pitch_hz": 120.0},
|
|
{"text": "cara", "start": 6.4, "end": 6.7, "speaker": "SPEAKER_00",
|
|
"energy": 0.4, "pitch_delta": 0.3, "rate_delta": 0.2, "pause_before": 0.0,
|
|
"emphasis": 0.4, "emotion": "neutral", "emotion_confidence": 0.0,
|
|
"arousal": 0.4, "valence": 0.5, "energy_raw": 0.4, "pitch_hz": 122.0},
|
|
],
|
|
},
|
|
],
|
|
}
|
|
|
|
|
|
def test_extract_json_unfenced():
|
|
text = '{"source": "x", "actions": []}'
|
|
assert _extract_json(text) == {"source": "x", "actions": []}
|
|
|
|
|
|
def test_extract_json_with_fence():
|
|
text = '```json\n{"source": "x", "actions": [{"kind": "cut", "start": 1, "end": 2}]}\n```'
|
|
data = _extract_json(text)
|
|
assert data["source"] == "x"
|
|
assert data["actions"][0]["kind"] == "cut"
|
|
|
|
|
|
def test_extract_json_with_prose_around():
|
|
text = 'Aqui está:\n{"source": "x", "actions": []}\nfim.'
|
|
assert _extract_json(text) == {"source": "x", "actions": []}
|
|
|
|
|
|
def test_extract_json_invalid_returns_none():
|
|
assert _extract_json("no json here") is None
|
|
assert _extract_json("") is None
|
|
|
|
|
|
def test_extract_json_unwraps_wrapped_list():
|
|
# Some models wrap the expected {"source","actions"} object in a
|
|
# single-element list; the actions must still be found.
|
|
text = '```json\n[{"source": "x", "actions": [{"kind": "cut", "start": 1, "end": 2}]}]\n```'
|
|
data = _extract_json(text)
|
|
assert isinstance(data, dict)
|
|
assert data["source"] == "x"
|
|
assert data["actions"][0]["kind"] == "cut"
|
|
|
|
|
|
def test_build_edit_messages_shape():
|
|
system, user = build_edit_messages(_fake_timeline())
|
|
assert "editor" in system.lower()
|
|
assert "demo.mp4" in user
|
|
assert "segments" in user
|
|
assert "SOMENTE" in user
|
|
|
|
|
|
def test_generate_voice_actions_parses_model_json(monkeypatch):
|
|
canned = json.dumps({
|
|
"source": "demo.mp4",
|
|
"actions": [
|
|
{"kind": "cut", "start": 6.0, "end": 12.0,
|
|
"reason": "papo casual com a equipe"},
|
|
{"kind": "zoom", "start": 0.1, "end": 0.9, "params": {"scale": 1.3},
|
|
"reason": "ênfase em 'vamos'"},
|
|
],
|
|
})
|
|
monkeypatch.setattr(llm_local, "ollama_chat", lambda **kwargs: canned)
|
|
|
|
result = generate_voice_actions(_fake_timeline(), model="test")
|
|
actions, errors = parse_actions([a.as_dict() for a in result["actions"]])
|
|
assert errors == []
|
|
assert len(actions) == 2
|
|
assert {a.kind for a in actions} == {"cut", "zoom"}
|
|
assert result["errors"] == []
|
|
|
|
|
|
def test_generate_voice_actions_reports_bad_rows(monkeypatch):
|
|
canned = json.dumps({
|
|
"source": "demo.mp4",
|
|
"actions": [
|
|
{"kind": "cut", "start": 6.0, "end": 12.0, "reason": "ok"},
|
|
{"kind": "bogus", "start": 1, "end": 2},
|
|
{"kind": "zoom", "start": 0.1, "end": 0.05},
|
|
],
|
|
})
|
|
monkeypatch.setattr(llm_local, "ollama_chat", lambda **kwargs: canned)
|
|
|
|
result = generate_voice_actions(_fake_timeline(), model="test")
|
|
assert len(result["actions"]) == 1
|
|
assert result["actions"][0].kind == "cut"
|
|
assert len(result["errors"]) >= 2
|
|
|
|
|
|
def test_generate_voice_actions_handles_transport_error(monkeypatch):
|
|
def _boom(**kwargs):
|
|
raise RuntimeError("ollama down")
|
|
monkeypatch.setattr(llm_local, "ollama_chat", _boom)
|
|
|
|
result = generate_voice_actions(_fake_timeline(), model="test")
|
|
assert result["actions"] == []
|
|
assert any("ollama" in e.lower() for e in result["errors"])
|
|
|
|
|
|
def test_build_edit_messages_is_compact():
|
|
"""The prompt must drop the heavy per-word audio features so a real
|
|
recording fits in the model context (the 47k-token dump made Ollama drop
|
|
the connection)."""
|
|
import json as _json
|
|
|
|
timeline = _fake_timeline()
|
|
# pad with the kind of audio detail a real timeline carries
|
|
timeline["segments"][0]["words"][0]["samples"] = [0.1, 0.2, 0.3]
|
|
timeline["speakers"][0]["samples"] = [1, 2, 3]
|
|
raw = _json.dumps(timeline, ensure_ascii=False)
|
|
system, user = build_edit_messages(timeline)
|
|
assert "demo.mp4" in user and "segments" in user
|
|
# projected prompt is materially smaller than the raw timeline
|
|
assert len(user) < len(raw)
|
|
# heavy fields we deliberately drop never reach the model
|
|
assert '"samples"' not in user
|
|
assert '"energy_raw"' not in user
|
|
assert '"pitch_hz"' not in user
|
|
|
|
|
|
def test_ollama_chat_wraps_empty_response(monkeypatch):
|
|
"""A dropped connection that yields an empty body must surface as a clear
|
|
RuntimeError (not an unhandled JSONDecodeError crashing the pipeline)."""
|
|
|
|
class _Resp:
|
|
def raise_for_status(self):
|
|
pass
|
|
|
|
def json(self):
|
|
raise ValueError("Expecting value: line 1 column 1")
|
|
|
|
monkeypatch.setattr(llm_local.httpx, "post", lambda *a, **k: _Resp())
|
|
try:
|
|
llm_local.ollama_chat(model="test", messages=[{"role": "user", "content": "x"}])
|
|
except RuntimeError as exc:
|
|
assert "Falha ao falar com o modelo local" in str(exc)
|
|
else:
|
|
raise AssertionError("expected RuntimeError")
|
|
|
|
|
|
def test_list_ollama_models_parses_tags(monkeypatch):
|
|
class _Resp:
|
|
def raise_for_status(self):
|
|
pass
|
|
|
|
def json(self):
|
|
return {"models": [{"name": "gemma3:4b"}, {"name": "gemma3:12b"}]}
|
|
|
|
monkeypatch.setattr(llm_local.httpx, "get", lambda *a, **k: _Resp())
|
|
assert llm_local.list_ollama_models() == ["gemma3:12b", "gemma3:4b"]
|
|
|
|
|
|
def test_list_ollama_models_returns_empty_on_error(monkeypatch):
|
|
def _boom(*a, **k):
|
|
raise RuntimeError("ollama down")
|
|
|
|
monkeypatch.setattr(llm_local.httpx, "get", _boom)
|
|
assert llm_local.list_ollama_models() == []
|
|
|