feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão

Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-21 18:26:04 -04:00
co-authored by Claude Sonnet 5
parent 711c397dfe
commit 7b5aed79ee
36 changed files with 2922 additions and 624 deletions
+23
View File
@@ -264,3 +264,26 @@ class TestIntegration:
report = modifier.validate_subtitle_layout()
assert report["summary"]["spatial_collision"] >= 1
assert blocking(report["severity"])
def test_disabled_title_is_excluded_from_validation(self, temp_fcpxml):
"""A title with enabled="0" never renders in Final Cut
(generate_subtitles_by_emphasis disables plain titles under an
emphasis phrase instead of never creating them) — it must not count
as a collision, or as outside-frame/outside-safe-area, against the
title actually drawn in its place."""
modifier = FCPXMLModifier(temp_fcpxml)
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
def position(el):
for p in el.findall("param"):
if p.get("name") == "Position":
return p
return None
p0 = position(titles[0])
position(titles[1]).set("value", p0.get("value"))
titles[1].set("enabled", "0")
report = modifier.validate_subtitle_layout()
assert report["summary"]["spatial_collision"] == 0
assert not blocking(report["severity"])
+232
View File
@@ -0,0 +1,232 @@
"""Tests for fcpxml/forced_align.py — optional phonetic forced alignment.
The dependency (whisperx) is not installed in CI, so the core contract under
test is graceful degradation: when whisperx is unavailable the aligner returns
the words unchanged. A second group injects a fake whisperx module to verify
the refined times are written back in order and that malformed results are
skipped rather than clobbering good timestamps.
"""
import sys
import types
from pathlib import Path
import pytest
from fcpxml.forced_align import ForcedAligner
def _words():
return [
{"word": "Um,", "start": 0.0, "end": 0.5, "confidence": 0.9},
{"word": "welcome", "start": 0.5, "end": 1.0, "confidence": 0.9},
{"word": "show.", "start": 1.5, "end": 2.5, "confidence": 0.8},
]
def _raw_segments():
return [
{
"text": "Um, welcome",
"start": 0.0,
"end": 1.0,
"words": _words()[:2],
},
{
"text": "show.",
"start": 1.5,
"end": 2.5,
"words": _words()[2:],
},
]
def _fake_whisperx(shift=0.4):
"""A stand-in whisperx module that "corrects" word starts by ``shift``."""
mod = types.SimpleNamespace()
def load_audio(path):
return [0.0]
def load_align_model(language_code, device, model_dir=None):
return ("MODEL", {"language": language_code})
def align(align_input, align_model, metadata, audio, device,
return_char_alignments=False, chunk_size=30):
segments = []
for seg in align_input:
new_words = []
for w in seg["words"]:
new_words.append(
{
"word": w["word"],
"start": w["start"] + shift,
"end": w["end"] + shift,
"score": w["score"],
}
)
segments.append({**seg, "words": new_words})
return {"segments": segments}
mod.load_audio = load_audio
mod.load_align_model = load_align_model
mod.align = align
return mod
class TestForcedAlignerDegradation:
def test_unavailable_when_whisperx_missing(self):
assert ForcedAligner.available() is False
def test_returns_words_unchanged_when_whisperx_missing(self, monkeypatch):
import builtins
real_import = builtins.__import__
def block(name, *a, **k):
if name == "whisperx":
raise ImportError("blocked")
return real_import(name, *a, **k)
monkeypatch.setattr(builtins, "__import__", block)
result = ForcedAligner().align(_words(), _raw_segments(), "x.wav", "en")
assert result == _words()
def test_skips_when_no_words(self):
assert ForcedAligner().align([], [], "x.wav", "en") == []
class TestForcedAlignerWithWhisperX:
@pytest.fixture
def whisperx(self, monkeypatch):
fake = _fake_whisperx(shift=0.4)
monkeypatch.setitem(sys.modules, "whisperx", fake)
return fake
def test_refines_timestamps_in_order(self, whisperx):
words = _words()
result = ForcedAligner().align(words, _raw_segments(), "x.wav", "en")
assert [w["start"] for w in result] == [0.4, 0.9, 1.9]
assert [w["end"] for w in result] == [0.9, 1.4, 2.9]
# The same dict objects are returned with times overwritten in place.
assert result[0]["start"] == 0.4
assert words[0]["start"] == 0.4
def test_caches_align_model_per_language(self, whisperx, monkeypatch):
calls = {"n": 0}
orig = whisperx.load_align_model
def counting(*a, **k):
calls["n"] += 1
return orig(*a, **k)
whisperx.load_align_model = counting
aligner = ForcedAligner()
aligner.align(_words(), _raw_segments(), "a.wav", "en")
aligner.align(_words(), _raw_segments(), "b.wav", "en")
assert calls["n"] == 1
def test_skips_unusable_word_times(self, monkeypatch):
fake = _fake_whisperx()
# Force one word to come back with None start (alignment failed).
real_align = fake.align
def broken(align_input, *a, **k):
out = real_align(align_input, *a, **k)
out["segments"][0]["words"][0]["start"] = None
return out
fake.align = broken
monkeypatch.setitem(sys.modules, "whisperx", fake)
words = _words()
result = ForcedAligner().align(words, _raw_segments(), "x.wav", "en")
# First word time untouched (None skipped), rest corrected.
assert result[0]["start"] == 0.0
assert result[1]["start"] == 0.9
def test_unexpected_exception_returns_original(self, monkeypatch):
fake = types.SimpleNamespace()
fake.load_audio = lambda p: [0.0]
fake.load_align_model = lambda *a, **k: ("M", {})
fake.align = lambda *a, **k: 1 / 0 # boom
monkeypatch.setitem(sys.modules, "whisperx", fake)
words = _words()
result = ForcedAligner().align(words, _raw_segments(), "x.wav", "en")
assert result == words
class TestTranscribeAlignmentFlag:
"""Wire-up: transcribe() reports whether forced alignment ran."""
def _install_fakes(self, monkeypatch, align_shift=0.4):
# faster_whisper
fw = types.SimpleNamespace()
class _Word:
def __init__(self, word, start, end, prob):
self.word = word
self.start = start
self.end = end
self.probability = prob
class _Seg:
def __init__(self, text, start, end, words):
self.text = text
self.start = start
self.end = end
self.words = words
class _Info:
language = "en"
duration = 2.5
class _Model:
def transcribe(self, path, language=None, word_timestamps=False, vad_filter=False):
seg = _Seg(
"Um, welcome show.",
0.0,
2.5,
[
_Word("Um,", 0.0, 0.5, 0.9),
_Word("welcome", 0.5, 1.0, 0.9),
_Word("show.", 1.5, 2.5, 0.8),
],
)
return iter([seg]), _Info()
fw.WhisperModel = lambda *a, **k: _Model()
monkeypatch.setitem(sys.modules, "faster_whisper", fw)
# whisperx (only needed when align=True)
wx = _fake_whisperx(shift=align_shift)
monkeypatch.setitem(sys.modules, "whisperx", wx)
# model_manager.get_models_dir
import fcpxml.model_manager as mm
monkeypatch.setattr(mm, "get_models_dir", lambda: Path("/tmp"))
def test_alignment_true_when_whisperx_present(self, monkeypatch, tmp_path):
self._install_fakes(monkeypatch)
f = tmp_path / "a.wav"
f.write_bytes(b"RIFF0000WAVE")
from fcpxml.transcribe import transcribe
result = transcribe(str(f), model_size="base", align=True)
assert result is not None
assert result["alignment"] is True
assert result["words"][0]["start"] == pytest.approx(0.4)
def test_alignment_false_when_disabled(self, monkeypatch, tmp_path):
self._install_fakes(monkeypatch)
f = tmp_path / "a.wav"
f.write_bytes(b"RIFF0000WAVE")
from fcpxml.transcribe import transcribe
result = transcribe(str(f), model_size="base", align=False)
assert result is not None
assert result["alignment"] is False
assert result["words"][0]["start"] == 0.0
+259
View File
@@ -0,0 +1,259 @@
"""Tests for the local LLM integration (Ollama): prompt building, JSON
extraction, and turning a model response into validated voice actions.
The model itself is mocked — these tests cover the client contract
(extraction, validation, message shape) without a running Ollama. A real
end-to-end run lives in the manual test harness (see ENGINE notes) because it
needs the local model server.
"""
import json
import pytest
from fcpxml import llm_local
from fcpxml.llm_local import (
_extract_json,
build_edit_messages,
generate_voice_actions,
)
from fcpxml.voice_actions import parse_actions
def _fake_timeline() -> dict:
"""A minimal but well-formed voice timeline (matches build_voice_timeline)."""
return {
"version": "1.0",
"source": "demo.mp4",
"rotation": 0.0,
"language": "pt",
"layers": {
"transcript": True,
"acoustics": True,
"speakers": False,
"emotion": False,
"alignment": False,
},
"scales": {},
"summary": {
"duration": 12.0,
"speaker_count": 1,
"segment_count": 2,
"word_count": 4,
"avg_emphasis": 0.4,
"peak_selection": "top 2%",
"peak_count": 1,
"peak_moments": [],
},
"speakers": [
{
"id": "SPEAKER_00",
"name": "Speaker 1",
"speaking_seconds": 12.0,
"share": 1.0,
"segment_count": 2,
"avg_segment": 6.0,
"word_count": 4,
"samples": [],
}
],
"segments": [
{
"start": 0.0,
"end": 6.0,
"speaker": "SPEAKER_00",
"text": "Hoje vamos falar de mastopexia.",
"gap_before": 0.0,
"take_boundary": False,
"avg_energy": 0.5,
"peak_emphasis": 0.6,
"emotion": "neutral",
"emotion_confidence": 0.0,
"arousal": 0.5,
"valence": 0.6,
"words": [
{"text": "Hoje", "start": 0.1, "end": 0.5, "speaker": "SPEAKER_00",
"energy": 0.3, "pitch_delta": 0.2, "rate_delta": 0.1, "pause_before": 0.1,
"emphasis": 0.3, "emotion": "neutral", "emotion_confidence": 0.0,
"arousal": 0.4, "valence": 0.6, "energy_raw": 0.3, "pitch_hz": 120.0},
{"text": "vamos", "start": 0.6, "end": 0.9, "speaker": "SPEAKER_00",
"energy": 0.4, "pitch_delta": 0.3, "rate_delta": 0.2, "pause_before": 0.0,
"emphasis": 0.5, "emotion": "neutral", "emotion_confidence": 0.0,
"arousal": 0.5, "valence": 0.6, "energy_raw": 0.4, "pitch_hz": 125.0},
],
},
{
"start": 6.0,
"end": 12.0,
"speaker": "SPEAKER_00",
"text": "E aí cara, tá gravando?",
"gap_before": 0.2,
"take_boundary": False,
"avg_energy": 0.4,
"peak_emphasis": 0.3,
"emotion": "neutral",
"emotion_confidence": 0.0,
"arousal": 0.4,
"valence": 0.5,
"words": [
{"text": "E", "start": 6.1, "end": 6.3, "speaker": "SPEAKER_00",
"energy": 0.3, "pitch_delta": 0.2, "rate_delta": 0.1, "pause_before": 0.1,
"emphasis": 0.3, "emotion": "neutral", "emotion_confidence": 0.0,
"arousal": 0.4, "valence": 0.5, "energy_raw": 0.3, "pitch_hz": 120.0},
{"text": "cara", "start": 6.4, "end": 6.7, "speaker": "SPEAKER_00",
"energy": 0.4, "pitch_delta": 0.3, "rate_delta": 0.2, "pause_before": 0.0,
"emphasis": 0.4, "emotion": "neutral", "emotion_confidence": 0.0,
"arousal": 0.4, "valence": 0.5, "energy_raw": 0.4, "pitch_hz": 122.0},
],
},
],
}
def test_extract_json_unfenced():
text = '{"source": "x", "actions": []}'
assert _extract_json(text) == {"source": "x", "actions": []}
def test_extract_json_with_fence():
text = '```json\n{"source": "x", "actions": [{"kind": "cut", "start": 1, "end": 2}]}\n```'
data = _extract_json(text)
assert data["source"] == "x"
assert data["actions"][0]["kind"] == "cut"
def test_extract_json_with_prose_around():
text = 'Aqui está:\n{"source": "x", "actions": []}\nfim.'
assert _extract_json(text) == {"source": "x", "actions": []}
def test_extract_json_invalid_returns_none():
assert _extract_json("no json here") is None
assert _extract_json("") is None
def test_extract_json_unwraps_wrapped_list():
# Some models wrap the expected {"source","actions"} object in a
# single-element list; the actions must still be found.
text = '```json\n[{"source": "x", "actions": [{"kind": "cut", "start": 1, "end": 2}]}]\n```'
data = _extract_json(text)
assert isinstance(data, dict)
assert data["source"] == "x"
assert data["actions"][0]["kind"] == "cut"
def test_build_edit_messages_shape():
system, user = build_edit_messages(_fake_timeline())
assert "editor" in system.lower()
assert "demo.mp4" in user
assert "segments" in user
assert "SOMENTE" in user
def test_generate_voice_actions_parses_model_json(monkeypatch):
canned = json.dumps({
"source": "demo.mp4",
"actions": [
{"kind": "cut", "start": 6.0, "end": 12.0,
"reason": "papo casual com a equipe"},
{"kind": "zoom", "start": 0.1, "end": 0.9, "params": {"scale": 1.3},
"reason": "ênfase em 'vamos'"},
],
})
monkeypatch.setattr(llm_local, "ollama_chat", lambda **kwargs: canned)
result = generate_voice_actions(_fake_timeline(), model="test")
actions, errors = parse_actions([a.as_dict() for a in result["actions"]])
assert errors == []
assert len(actions) == 2
assert {a.kind for a in actions} == {"cut", "zoom"}
assert result["errors"] == []
def test_generate_voice_actions_reports_bad_rows(monkeypatch):
canned = json.dumps({
"source": "demo.mp4",
"actions": [
{"kind": "cut", "start": 6.0, "end": 12.0, "reason": "ok"},
{"kind": "bogus", "start": 1, "end": 2},
{"kind": "zoom", "start": 0.1, "end": 0.05},
],
})
monkeypatch.setattr(llm_local, "ollama_chat", lambda **kwargs: canned)
result = generate_voice_actions(_fake_timeline(), model="test")
assert len(result["actions"]) == 1
assert result["actions"][0].kind == "cut"
assert len(result["errors"]) >= 2
def test_generate_voice_actions_handles_transport_error(monkeypatch):
def _boom(**kwargs):
raise RuntimeError("ollama down")
monkeypatch.setattr(llm_local, "ollama_chat", _boom)
result = generate_voice_actions(_fake_timeline(), model="test")
assert result["actions"] == []
assert any("ollama" in e.lower() for e in result["errors"])
def test_build_edit_messages_is_compact():
"""The prompt must drop the heavy per-word audio features so a real
recording fits in the model context (the 47k-token dump made Ollama drop
the connection)."""
import json as _json
timeline = _fake_timeline()
# pad with the kind of audio detail a real timeline carries
timeline["segments"][0]["words"][0]["samples"] = [0.1, 0.2, 0.3]
timeline["speakers"][0]["samples"] = [1, 2, 3]
raw = _json.dumps(timeline, ensure_ascii=False)
system, user = build_edit_messages(timeline)
assert "demo.mp4" in user and "segments" in user
# projected prompt is materially smaller than the raw timeline
assert len(user) < len(raw)
# heavy fields we deliberately drop never reach the model
assert '"samples"' not in user
assert '"energy_raw"' not in user
assert '"pitch_hz"' not in user
def test_ollama_chat_wraps_empty_response(monkeypatch):
"""A dropped connection that yields an empty body must surface as a clear
RuntimeError (not an unhandled JSONDecodeError crashing the pipeline)."""
class _Resp:
def raise_for_status(self):
pass
def json(self):
raise ValueError("Expecting value: line 1 column 1")
monkeypatch.setattr(llm_local.httpx, "post", lambda *a, **k: _Resp())
try:
llm_local.ollama_chat(model="test", messages=[{"role": "user", "content": "x"}])
except RuntimeError as exc:
assert "Falha ao falar com o modelo local" in str(exc)
else:
raise AssertionError("expected RuntimeError")
def test_list_ollama_models_parses_tags(monkeypatch):
class _Resp:
def raise_for_status(self):
pass
def json(self):
return {"models": [{"name": "gemma3:4b"}, {"name": "gemma3:12b"}]}
monkeypatch.setattr(llm_local.httpx, "get", lambda *a, **k: _Resp())
assert llm_local.list_ollama_models() == ["gemma3:12b", "gemma3:4b"]
def test_list_ollama_models_returns_empty_on_error(monkeypatch):
def _boom(*a, **k):
raise RuntimeError("ollama down")
monkeypatch.setattr(llm_local.httpx, "get", _boom)
assert llm_local.list_ollama_models() == []
+134
View File
@@ -0,0 +1,134 @@
"""Tests for layering subtitle generation by emphasis (etapa 5 review).
Covers the pure helpers in server_tools/subtitles.py that decide which words
also get the dynamic treatment on top of the always-complete plain track, and
which clip-relative windows a plain title must be hidden (enabled="0") under.
"""
import json
from pathlib import Path
from server_tools.subtitles import (
_load_emphasis_spans,
_overlaps_any_span,
_phrase_actions_path,
_segments_in_spans,
_word_in_spans,
_words_in_spans,
)
SPANS = [
{"start": 2.0, "end": 10.7, "level": 1, "text": "abertura"},
{"start": 127.7, "end": 135.0, "level": 2, "text": "mastopexia"},
]
def test_phrase_actions_path_matches_media_stem():
assert _phrase_actions_path("/x/y/0E6A8290.mp4") == Path(
"/x/y/0E6A8290_phrase_actions.json"
)
class TestLoadEmphasisSpans:
def test_missing_file_returns_empty(self, tmp_path):
media = tmp_path / "clip.mp4"
media.write_bytes(b"")
assert _load_emphasis_spans(str(media)) == []
def test_reads_spans_from_sibling_json(self, tmp_path):
media = tmp_path / "clip.mp4"
media.write_bytes(b"")
actions_path = tmp_path / "clip_phrase_actions.json"
actions_path.write_text(
json.dumps({"source": "clip.mp4", "actions": [], "emphasis_spans": SPANS}),
encoding="utf-8",
)
assert _load_emphasis_spans(str(media)) == SPANS
def test_corrupt_json_returns_empty(self, tmp_path):
media = tmp_path / "clip.mp4"
media.write_bytes(b"")
(tmp_path / "clip_phrase_actions.json").write_text("{not json", encoding="utf-8")
assert _load_emphasis_spans(str(media)) == []
def test_missing_emphasis_spans_key_returns_empty(self, tmp_path):
media = tmp_path / "clip.mp4"
media.write_bytes(b"")
(tmp_path / "clip_phrase_actions.json").write_text(
json.dumps({"source": "clip.mp4", "actions": []}), encoding="utf-8"
)
assert _load_emphasis_spans(str(media)) == []
class TestWordInSpans:
def test_word_fully_inside_span(self):
assert _word_in_spans(3.0, 3.4, SPANS) is True
def test_word_fully_outside_every_span(self):
assert _word_in_spans(50.0, 50.4, SPANS) is False
def test_word_straddling_span_boundary_follows_its_midpoint(self):
# midpoint 10.6 -> inside [2.0, 10.7)
assert _word_in_spans(10.4, 10.8, SPANS) is True
# midpoint 10.9 -> outside
assert _word_in_spans(10.7, 11.1, SPANS) is False
def test_span_end_is_exclusive(self):
assert _word_in_spans(10.7, 10.7, SPANS) is False
class TestWordsInSpans:
WORDS = [
{"word": "Aquela", "start": 2.03, "end": 2.69},
{"word": "mama", "start": 2.69, "end": 2.89},
{"word": "fora", "start": 50.0, "end": 50.2},
{"word": "mastopexia", "start": 127.74, "end": 128.58},
]
def test_no_spans_returns_nothing(self):
assert _words_in_spans(self.WORDS, []) == []
def test_keeps_only_words_inside_a_span(self):
kept = _words_in_spans(self.WORDS, SPANS)
assert [w["word"] for w in kept] == ["Aquela", "mama", "mastopexia"]
def test_does_not_remove_words_from_the_source_list(self):
"""This is a filter for the dynamic half, not a partition — the plain
half must still see every word, so this must never mutate `words`."""
before = list(self.WORDS)
_words_in_spans(self.WORDS, SPANS)
assert self.WORDS == before
class TestOverlapsAnySpan:
CLIP_SPANS = [(0.0, 8.64), (60.0, 67.22)]
def test_window_inside_a_span_overlaps(self):
assert _overlaps_any_span(1.0, 2.0, self.CLIP_SPANS) is True
def test_window_outside_every_span_does_not_overlap(self):
assert _overlaps_any_span(20.0, 21.0, self.CLIP_SPANS) is False
def test_window_straddling_a_span_edge_overlaps(self):
assert _overlaps_any_span(8.0, 9.0, self.CLIP_SPANS) is True
def test_touching_but_not_overlapping_is_not_an_overlap(self):
assert _overlaps_any_span(8.64, 9.0, self.CLIP_SPANS) is False
def test_no_spans_never_overlaps(self):
assert _overlaps_any_span(1.0, 2.0, []) is False
class TestSegmentsInSpans:
SEGMENTS = [
{"start": 2.0, "end": 10.67},
{"start": 40.0, "end": 43.0},
{"start": 127.74, "end": 134.96},
]
def test_no_spans_keeps_nothing(self):
assert _segments_in_spans(self.SEGMENTS, []) == []
def test_keeps_only_segments_inside_a_span(self):
kept = _segments_in_spans(self.SEGMENTS, SPANS)
assert kept == [self.SEGMENTS[0], self.SEGMENTS[2]]