feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão

Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-21 18:26:04 -04:00
co-authored by Claude Sonnet 5
parent 711c397dfe
commit 7b5aed79ee
36 changed files with 2922 additions and 624 deletions
+134
View File
@@ -0,0 +1,134 @@
"""Tests for layering subtitle generation by emphasis (etapa 5 review).
Covers the pure helpers in server_tools/subtitles.py that decide which words
also get the dynamic treatment on top of the always-complete plain track, and
which clip-relative windows a plain title must be hidden (enabled="0") under.
"""
import json
from pathlib import Path
from server_tools.subtitles import (
_load_emphasis_spans,
_overlaps_any_span,
_phrase_actions_path,
_segments_in_spans,
_word_in_spans,
_words_in_spans,
)
SPANS = [
{"start": 2.0, "end": 10.7, "level": 1, "text": "abertura"},
{"start": 127.7, "end": 135.0, "level": 2, "text": "mastopexia"},
]
def test_phrase_actions_path_matches_media_stem():
assert _phrase_actions_path("/x/y/0E6A8290.mp4") == Path(
"/x/y/0E6A8290_phrase_actions.json"
)
class TestLoadEmphasisSpans:
def test_missing_file_returns_empty(self, tmp_path):
media = tmp_path / "clip.mp4"
media.write_bytes(b"")
assert _load_emphasis_spans(str(media)) == []
def test_reads_spans_from_sibling_json(self, tmp_path):
media = tmp_path / "clip.mp4"
media.write_bytes(b"")
actions_path = tmp_path / "clip_phrase_actions.json"
actions_path.write_text(
json.dumps({"source": "clip.mp4", "actions": [], "emphasis_spans": SPANS}),
encoding="utf-8",
)
assert _load_emphasis_spans(str(media)) == SPANS
def test_corrupt_json_returns_empty(self, tmp_path):
media = tmp_path / "clip.mp4"
media.write_bytes(b"")
(tmp_path / "clip_phrase_actions.json").write_text("{not json", encoding="utf-8")
assert _load_emphasis_spans(str(media)) == []
def test_missing_emphasis_spans_key_returns_empty(self, tmp_path):
media = tmp_path / "clip.mp4"
media.write_bytes(b"")
(tmp_path / "clip_phrase_actions.json").write_text(
json.dumps({"source": "clip.mp4", "actions": []}), encoding="utf-8"
)
assert _load_emphasis_spans(str(media)) == []
class TestWordInSpans:
def test_word_fully_inside_span(self):
assert _word_in_spans(3.0, 3.4, SPANS) is True
def test_word_fully_outside_every_span(self):
assert _word_in_spans(50.0, 50.4, SPANS) is False
def test_word_straddling_span_boundary_follows_its_midpoint(self):
# midpoint 10.6 -> inside [2.0, 10.7)
assert _word_in_spans(10.4, 10.8, SPANS) is True
# midpoint 10.9 -> outside
assert _word_in_spans(10.7, 11.1, SPANS) is False
def test_span_end_is_exclusive(self):
assert _word_in_spans(10.7, 10.7, SPANS) is False
class TestWordsInSpans:
WORDS = [
{"word": "Aquela", "start": 2.03, "end": 2.69},
{"word": "mama", "start": 2.69, "end": 2.89},
{"word": "fora", "start": 50.0, "end": 50.2},
{"word": "mastopexia", "start": 127.74, "end": 128.58},
]
def test_no_spans_returns_nothing(self):
assert _words_in_spans(self.WORDS, []) == []
def test_keeps_only_words_inside_a_span(self):
kept = _words_in_spans(self.WORDS, SPANS)
assert [w["word"] for w in kept] == ["Aquela", "mama", "mastopexia"]
def test_does_not_remove_words_from_the_source_list(self):
"""This is a filter for the dynamic half, not a partition — the plain
half must still see every word, so this must never mutate `words`."""
before = list(self.WORDS)
_words_in_spans(self.WORDS, SPANS)
assert self.WORDS == before
class TestOverlapsAnySpan:
CLIP_SPANS = [(0.0, 8.64), (60.0, 67.22)]
def test_window_inside_a_span_overlaps(self):
assert _overlaps_any_span(1.0, 2.0, self.CLIP_SPANS) is True
def test_window_outside_every_span_does_not_overlap(self):
assert _overlaps_any_span(20.0, 21.0, self.CLIP_SPANS) is False
def test_window_straddling_a_span_edge_overlaps(self):
assert _overlaps_any_span(8.0, 9.0, self.CLIP_SPANS) is True
def test_touching_but_not_overlapping_is_not_an_overlap(self):
assert _overlaps_any_span(8.64, 9.0, self.CLIP_SPANS) is False
def test_no_spans_never_overlaps(self):
assert _overlaps_any_span(1.0, 2.0, []) is False
class TestSegmentsInSpans:
SEGMENTS = [
{"start": 2.0, "end": 10.67},
{"start": 40.0, "end": 43.0},
{"start": 127.74, "end": 134.96},
]
def test_no_spans_keeps_nothing(self):
assert _segments_in_spans(self.SEGMENTS, []) == []
def test_keeps_only_segments_inside_a_span(self):
kept = _segments_in_spans(self.SEGMENTS, SPANS)
assert kept == [self.SEGMENTS[0], self.SEGMENTS[2]]