Files
gart/code/tests/test_forced_align.py
T
João HenriqueandClaude Sonnet 5 7b5aed79ee feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-21 18:26:04 -04:00

233 lines
7.6 KiB
Python

"""Tests for fcpxml/forced_align.py — optional phonetic forced alignment.
The dependency (whisperx) is not installed in CI, so the core contract under
test is graceful degradation: when whisperx is unavailable the aligner returns
the words unchanged. A second group injects a fake whisperx module to verify
the refined times are written back in order and that malformed results are
skipped rather than clobbering good timestamps.
"""
import sys
import types
from pathlib import Path
import pytest
from fcpxml.forced_align import ForcedAligner
def _words():
return [
{"word": "Um,", "start": 0.0, "end": 0.5, "confidence": 0.9},
{"word": "welcome", "start": 0.5, "end": 1.0, "confidence": 0.9},
{"word": "show.", "start": 1.5, "end": 2.5, "confidence": 0.8},
]
def _raw_segments():
return [
{
"text": "Um, welcome",
"start": 0.0,
"end": 1.0,
"words": _words()[:2],
},
{
"text": "show.",
"start": 1.5,
"end": 2.5,
"words": _words()[2:],
},
]
def _fake_whisperx(shift=0.4):
"""A stand-in whisperx module that "corrects" word starts by ``shift``."""
mod = types.SimpleNamespace()
def load_audio(path):
return [0.0]
def load_align_model(language_code, device, model_dir=None):
return ("MODEL", {"language": language_code})
def align(align_input, align_model, metadata, audio, device,
return_char_alignments=False, chunk_size=30):
segments = []
for seg in align_input:
new_words = []
for w in seg["words"]:
new_words.append(
{
"word": w["word"],
"start": w["start"] + shift,
"end": w["end"] + shift,
"score": w["score"],
}
)
segments.append({**seg, "words": new_words})
return {"segments": segments}
mod.load_audio = load_audio
mod.load_align_model = load_align_model
mod.align = align
return mod
class TestForcedAlignerDegradation:
def test_unavailable_when_whisperx_missing(self):
assert ForcedAligner.available() is False
def test_returns_words_unchanged_when_whisperx_missing(self, monkeypatch):
import builtins
real_import = builtins.__import__
def block(name, *a, **k):
if name == "whisperx":
raise ImportError("blocked")
return real_import(name, *a, **k)
monkeypatch.setattr(builtins, "__import__", block)
result = ForcedAligner().align(_words(), _raw_segments(), "x.wav", "en")
assert result == _words()
def test_skips_when_no_words(self):
assert ForcedAligner().align([], [], "x.wav", "en") == []
class TestForcedAlignerWithWhisperX:
@pytest.fixture
def whisperx(self, monkeypatch):
fake = _fake_whisperx(shift=0.4)
monkeypatch.setitem(sys.modules, "whisperx", fake)
return fake
def test_refines_timestamps_in_order(self, whisperx):
words = _words()
result = ForcedAligner().align(words, _raw_segments(), "x.wav", "en")
assert [w["start"] for w in result] == [0.4, 0.9, 1.9]
assert [w["end"] for w in result] == [0.9, 1.4, 2.9]
# The same dict objects are returned with times overwritten in place.
assert result[0]["start"] == 0.4
assert words[0]["start"] == 0.4
def test_caches_align_model_per_language(self, whisperx, monkeypatch):
calls = {"n": 0}
orig = whisperx.load_align_model
def counting(*a, **k):
calls["n"] += 1
return orig(*a, **k)
whisperx.load_align_model = counting
aligner = ForcedAligner()
aligner.align(_words(), _raw_segments(), "a.wav", "en")
aligner.align(_words(), _raw_segments(), "b.wav", "en")
assert calls["n"] == 1
def test_skips_unusable_word_times(self, monkeypatch):
fake = _fake_whisperx()
# Force one word to come back with None start (alignment failed).
real_align = fake.align
def broken(align_input, *a, **k):
out = real_align(align_input, *a, **k)
out["segments"][0]["words"][0]["start"] = None
return out
fake.align = broken
monkeypatch.setitem(sys.modules, "whisperx", fake)
words = _words()
result = ForcedAligner().align(words, _raw_segments(), "x.wav", "en")
# First word time untouched (None skipped), rest corrected.
assert result[0]["start"] == 0.0
assert result[1]["start"] == 0.9
def test_unexpected_exception_returns_original(self, monkeypatch):
fake = types.SimpleNamespace()
fake.load_audio = lambda p: [0.0]
fake.load_align_model = lambda *a, **k: ("M", {})
fake.align = lambda *a, **k: 1 / 0 # boom
monkeypatch.setitem(sys.modules, "whisperx", fake)
words = _words()
result = ForcedAligner().align(words, _raw_segments(), "x.wav", "en")
assert result == words
class TestTranscribeAlignmentFlag:
"""Wire-up: transcribe() reports whether forced alignment ran."""
def _install_fakes(self, monkeypatch, align_shift=0.4):
# faster_whisper
fw = types.SimpleNamespace()
class _Word:
def __init__(self, word, start, end, prob):
self.word = word
self.start = start
self.end = end
self.probability = prob
class _Seg:
def __init__(self, text, start, end, words):
self.text = text
self.start = start
self.end = end
self.words = words
class _Info:
language = "en"
duration = 2.5
class _Model:
def transcribe(self, path, language=None, word_timestamps=False, vad_filter=False):
seg = _Seg(
"Um, welcome show.",
0.0,
2.5,
[
_Word("Um,", 0.0, 0.5, 0.9),
_Word("welcome", 0.5, 1.0, 0.9),
_Word("show.", 1.5, 2.5, 0.8),
],
)
return iter([seg]), _Info()
fw.WhisperModel = lambda *a, **k: _Model()
monkeypatch.setitem(sys.modules, "faster_whisper", fw)
# whisperx (only needed when align=True)
wx = _fake_whisperx(shift=align_shift)
monkeypatch.setitem(sys.modules, "whisperx", wx)
# model_manager.get_models_dir
import fcpxml.model_manager as mm
monkeypatch.setattr(mm, "get_models_dir", lambda: Path("/tmp"))
def test_alignment_true_when_whisperx_present(self, monkeypatch, tmp_path):
self._install_fakes(monkeypatch)
f = tmp_path / "a.wav"
f.write_bytes(b"RIFF0000WAVE")
from fcpxml.transcribe import transcribe
result = transcribe(str(f), model_size="base", align=True)
assert result is not None
assert result["alignment"] is True
assert result["words"][0]["start"] == pytest.approx(0.4)
def test_alignment_false_when_disabled(self, monkeypatch, tmp_path):
self._install_fakes(monkeypatch)
f = tmp_path / "a.wav"
f.write_bytes(b"RIFF0000WAVE")
from fcpxml.transcribe import transcribe
result = transcribe(str(f), model_size="base", align=False)
assert result is not None
assert result["alignment"] is False
assert result["words"][0]["start"] == 0.0