feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
711c397dfe
commit
7b5aed79ee
@@ -264,3 +264,26 @@ class TestIntegration:
|
||||
report = modifier.validate_subtitle_layout()
|
||||
assert report["summary"]["spatial_collision"] >= 1
|
||||
assert blocking(report["severity"])
|
||||
|
||||
def test_disabled_title_is_excluded_from_validation(self, temp_fcpxml):
|
||||
"""A title with enabled="0" never renders in Final Cut
|
||||
(generate_subtitles_by_emphasis disables plain titles under an
|
||||
emphasis phrase instead of never creating them) — it must not count
|
||||
as a collision, or as outside-frame/outside-safe-area, against the
|
||||
title actually drawn in its place."""
|
||||
modifier = FCPXMLModifier(temp_fcpxml)
|
||||
titles = modifier.generate_dynamic_subtitles("Interview_A", WORDS, WORD_MODE)
|
||||
|
||||
def position(el):
|
||||
for p in el.findall("param"):
|
||||
if p.get("name") == "Position":
|
||||
return p
|
||||
return None
|
||||
|
||||
p0 = position(titles[0])
|
||||
position(titles[1]).set("value", p0.get("value"))
|
||||
titles[1].set("enabled", "0")
|
||||
|
||||
report = modifier.validate_subtitle_layout()
|
||||
assert report["summary"]["spatial_collision"] == 0
|
||||
assert not blocking(report["severity"])
|
||||
|
||||
@@ -0,0 +1,232 @@
|
||||
"""Tests for fcpxml/forced_align.py — optional phonetic forced alignment.
|
||||
|
||||
The dependency (whisperx) is not installed in CI, so the core contract under
|
||||
test is graceful degradation: when whisperx is unavailable the aligner returns
|
||||
the words unchanged. A second group injects a fake whisperx module to verify
|
||||
the refined times are written back in order and that malformed results are
|
||||
skipped rather than clobbering good timestamps.
|
||||
"""
|
||||
|
||||
import sys
|
||||
import types
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml.forced_align import ForcedAligner
|
||||
|
||||
|
||||
def _words():
|
||||
return [
|
||||
{"word": "Um,", "start": 0.0, "end": 0.5, "confidence": 0.9},
|
||||
{"word": "welcome", "start": 0.5, "end": 1.0, "confidence": 0.9},
|
||||
{"word": "show.", "start": 1.5, "end": 2.5, "confidence": 0.8},
|
||||
]
|
||||
|
||||
|
||||
def _raw_segments():
|
||||
return [
|
||||
{
|
||||
"text": "Um, welcome",
|
||||
"start": 0.0,
|
||||
"end": 1.0,
|
||||
"words": _words()[:2],
|
||||
},
|
||||
{
|
||||
"text": "show.",
|
||||
"start": 1.5,
|
||||
"end": 2.5,
|
||||
"words": _words()[2:],
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def _fake_whisperx(shift=0.4):
|
||||
"""A stand-in whisperx module that "corrects" word starts by ``shift``."""
|
||||
mod = types.SimpleNamespace()
|
||||
|
||||
def load_audio(path):
|
||||
return [0.0]
|
||||
|
||||
def load_align_model(language_code, device, model_dir=None):
|
||||
return ("MODEL", {"language": language_code})
|
||||
|
||||
def align(align_input, align_model, metadata, audio, device,
|
||||
return_char_alignments=False, chunk_size=30):
|
||||
segments = []
|
||||
for seg in align_input:
|
||||
new_words = []
|
||||
for w in seg["words"]:
|
||||
new_words.append(
|
||||
{
|
||||
"word": w["word"],
|
||||
"start": w["start"] + shift,
|
||||
"end": w["end"] + shift,
|
||||
"score": w["score"],
|
||||
}
|
||||
)
|
||||
segments.append({**seg, "words": new_words})
|
||||
return {"segments": segments}
|
||||
|
||||
mod.load_audio = load_audio
|
||||
mod.load_align_model = load_align_model
|
||||
mod.align = align
|
||||
return mod
|
||||
|
||||
|
||||
class TestForcedAlignerDegradation:
|
||||
def test_unavailable_when_whisperx_missing(self):
|
||||
assert ForcedAligner.available() is False
|
||||
|
||||
def test_returns_words_unchanged_when_whisperx_missing(self, monkeypatch):
|
||||
import builtins
|
||||
|
||||
real_import = builtins.__import__
|
||||
|
||||
def block(name, *a, **k):
|
||||
if name == "whisperx":
|
||||
raise ImportError("blocked")
|
||||
return real_import(name, *a, **k)
|
||||
|
||||
monkeypatch.setattr(builtins, "__import__", block)
|
||||
result = ForcedAligner().align(_words(), _raw_segments(), "x.wav", "en")
|
||||
assert result == _words()
|
||||
|
||||
def test_skips_when_no_words(self):
|
||||
assert ForcedAligner().align([], [], "x.wav", "en") == []
|
||||
|
||||
|
||||
class TestForcedAlignerWithWhisperX:
|
||||
@pytest.fixture
|
||||
def whisperx(self, monkeypatch):
|
||||
fake = _fake_whisperx(shift=0.4)
|
||||
monkeypatch.setitem(sys.modules, "whisperx", fake)
|
||||
return fake
|
||||
|
||||
def test_refines_timestamps_in_order(self, whisperx):
|
||||
words = _words()
|
||||
result = ForcedAligner().align(words, _raw_segments(), "x.wav", "en")
|
||||
assert [w["start"] for w in result] == [0.4, 0.9, 1.9]
|
||||
assert [w["end"] for w in result] == [0.9, 1.4, 2.9]
|
||||
# The same dict objects are returned with times overwritten in place.
|
||||
assert result[0]["start"] == 0.4
|
||||
assert words[0]["start"] == 0.4
|
||||
|
||||
def test_caches_align_model_per_language(self, whisperx, monkeypatch):
|
||||
calls = {"n": 0}
|
||||
orig = whisperx.load_align_model
|
||||
|
||||
def counting(*a, **k):
|
||||
calls["n"] += 1
|
||||
return orig(*a, **k)
|
||||
|
||||
whisperx.load_align_model = counting
|
||||
aligner = ForcedAligner()
|
||||
aligner.align(_words(), _raw_segments(), "a.wav", "en")
|
||||
aligner.align(_words(), _raw_segments(), "b.wav", "en")
|
||||
assert calls["n"] == 1
|
||||
|
||||
def test_skips_unusable_word_times(self, monkeypatch):
|
||||
fake = _fake_whisperx()
|
||||
# Force one word to come back with None start (alignment failed).
|
||||
real_align = fake.align
|
||||
|
||||
def broken(align_input, *a, **k):
|
||||
out = real_align(align_input, *a, **k)
|
||||
out["segments"][0]["words"][0]["start"] = None
|
||||
return out
|
||||
|
||||
fake.align = broken
|
||||
monkeypatch.setitem(sys.modules, "whisperx", fake)
|
||||
|
||||
words = _words()
|
||||
result = ForcedAligner().align(words, _raw_segments(), "x.wav", "en")
|
||||
# First word time untouched (None skipped), rest corrected.
|
||||
assert result[0]["start"] == 0.0
|
||||
assert result[1]["start"] == 0.9
|
||||
|
||||
def test_unexpected_exception_returns_original(self, monkeypatch):
|
||||
fake = types.SimpleNamespace()
|
||||
fake.load_audio = lambda p: [0.0]
|
||||
fake.load_align_model = lambda *a, **k: ("M", {})
|
||||
fake.align = lambda *a, **k: 1 / 0 # boom
|
||||
monkeypatch.setitem(sys.modules, "whisperx", fake)
|
||||
|
||||
words = _words()
|
||||
result = ForcedAligner().align(words, _raw_segments(), "x.wav", "en")
|
||||
assert result == words
|
||||
|
||||
|
||||
class TestTranscribeAlignmentFlag:
|
||||
"""Wire-up: transcribe() reports whether forced alignment ran."""
|
||||
|
||||
def _install_fakes(self, monkeypatch, align_shift=0.4):
|
||||
# faster_whisper
|
||||
fw = types.SimpleNamespace()
|
||||
|
||||
class _Word:
|
||||
def __init__(self, word, start, end, prob):
|
||||
self.word = word
|
||||
self.start = start
|
||||
self.end = end
|
||||
self.probability = prob
|
||||
|
||||
class _Seg:
|
||||
def __init__(self, text, start, end, words):
|
||||
self.text = text
|
||||
self.start = start
|
||||
self.end = end
|
||||
self.words = words
|
||||
|
||||
class _Info:
|
||||
language = "en"
|
||||
duration = 2.5
|
||||
|
||||
class _Model:
|
||||
def transcribe(self, path, language=None, word_timestamps=False, vad_filter=False):
|
||||
seg = _Seg(
|
||||
"Um, welcome show.",
|
||||
0.0,
|
||||
2.5,
|
||||
[
|
||||
_Word("Um,", 0.0, 0.5, 0.9),
|
||||
_Word("welcome", 0.5, 1.0, 0.9),
|
||||
_Word("show.", 1.5, 2.5, 0.8),
|
||||
],
|
||||
)
|
||||
return iter([seg]), _Info()
|
||||
|
||||
fw.WhisperModel = lambda *a, **k: _Model()
|
||||
|
||||
monkeypatch.setitem(sys.modules, "faster_whisper", fw)
|
||||
|
||||
# whisperx (only needed when align=True)
|
||||
wx = _fake_whisperx(shift=align_shift)
|
||||
monkeypatch.setitem(sys.modules, "whisperx", wx)
|
||||
|
||||
# model_manager.get_models_dir
|
||||
import fcpxml.model_manager as mm
|
||||
|
||||
monkeypatch.setattr(mm, "get_models_dir", lambda: Path("/tmp"))
|
||||
|
||||
def test_alignment_true_when_whisperx_present(self, monkeypatch, tmp_path):
|
||||
self._install_fakes(monkeypatch)
|
||||
f = tmp_path / "a.wav"
|
||||
f.write_bytes(b"RIFF0000WAVE")
|
||||
from fcpxml.transcribe import transcribe
|
||||
|
||||
result = transcribe(str(f), model_size="base", align=True)
|
||||
assert result is not None
|
||||
assert result["alignment"] is True
|
||||
assert result["words"][0]["start"] == pytest.approx(0.4)
|
||||
|
||||
def test_alignment_false_when_disabled(self, monkeypatch, tmp_path):
|
||||
self._install_fakes(monkeypatch)
|
||||
f = tmp_path / "a.wav"
|
||||
f.write_bytes(b"RIFF0000WAVE")
|
||||
from fcpxml.transcribe import transcribe
|
||||
|
||||
result = transcribe(str(f), model_size="base", align=False)
|
||||
assert result is not None
|
||||
assert result["alignment"] is False
|
||||
assert result["words"][0]["start"] == 0.0
|
||||
@@ -0,0 +1,259 @@
|
||||
"""Tests for the local LLM integration (Ollama): prompt building, JSON
|
||||
extraction, and turning a model response into validated voice actions.
|
||||
|
||||
The model itself is mocked — these tests cover the client contract
|
||||
(extraction, validation, message shape) without a running Ollama. A real
|
||||
end-to-end run lives in the manual test harness (see ENGINE notes) because it
|
||||
needs the local model server.
|
||||
"""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml import llm_local
|
||||
from fcpxml.llm_local import (
|
||||
_extract_json,
|
||||
build_edit_messages,
|
||||
generate_voice_actions,
|
||||
)
|
||||
from fcpxml.voice_actions import parse_actions
|
||||
|
||||
|
||||
def _fake_timeline() -> dict:
|
||||
"""A minimal but well-formed voice timeline (matches build_voice_timeline)."""
|
||||
return {
|
||||
"version": "1.0",
|
||||
"source": "demo.mp4",
|
||||
"rotation": 0.0,
|
||||
"language": "pt",
|
||||
"layers": {
|
||||
"transcript": True,
|
||||
"acoustics": True,
|
||||
"speakers": False,
|
||||
"emotion": False,
|
||||
"alignment": False,
|
||||
},
|
||||
"scales": {},
|
||||
"summary": {
|
||||
"duration": 12.0,
|
||||
"speaker_count": 1,
|
||||
"segment_count": 2,
|
||||
"word_count": 4,
|
||||
"avg_emphasis": 0.4,
|
||||
"peak_selection": "top 2%",
|
||||
"peak_count": 1,
|
||||
"peak_moments": [],
|
||||
},
|
||||
"speakers": [
|
||||
{
|
||||
"id": "SPEAKER_00",
|
||||
"name": "Speaker 1",
|
||||
"speaking_seconds": 12.0,
|
||||
"share": 1.0,
|
||||
"segment_count": 2,
|
||||
"avg_segment": 6.0,
|
||||
"word_count": 4,
|
||||
"samples": [],
|
||||
}
|
||||
],
|
||||
"segments": [
|
||||
{
|
||||
"start": 0.0,
|
||||
"end": 6.0,
|
||||
"speaker": "SPEAKER_00",
|
||||
"text": "Hoje vamos falar de mastopexia.",
|
||||
"gap_before": 0.0,
|
||||
"take_boundary": False,
|
||||
"avg_energy": 0.5,
|
||||
"peak_emphasis": 0.6,
|
||||
"emotion": "neutral",
|
||||
"emotion_confidence": 0.0,
|
||||
"arousal": 0.5,
|
||||
"valence": 0.6,
|
||||
"words": [
|
||||
{"text": "Hoje", "start": 0.1, "end": 0.5, "speaker": "SPEAKER_00",
|
||||
"energy": 0.3, "pitch_delta": 0.2, "rate_delta": 0.1, "pause_before": 0.1,
|
||||
"emphasis": 0.3, "emotion": "neutral", "emotion_confidence": 0.0,
|
||||
"arousal": 0.4, "valence": 0.6, "energy_raw": 0.3, "pitch_hz": 120.0},
|
||||
{"text": "vamos", "start": 0.6, "end": 0.9, "speaker": "SPEAKER_00",
|
||||
"energy": 0.4, "pitch_delta": 0.3, "rate_delta": 0.2, "pause_before": 0.0,
|
||||
"emphasis": 0.5, "emotion": "neutral", "emotion_confidence": 0.0,
|
||||
"arousal": 0.5, "valence": 0.6, "energy_raw": 0.4, "pitch_hz": 125.0},
|
||||
],
|
||||
},
|
||||
{
|
||||
"start": 6.0,
|
||||
"end": 12.0,
|
||||
"speaker": "SPEAKER_00",
|
||||
"text": "E aí cara, tá gravando?",
|
||||
"gap_before": 0.2,
|
||||
"take_boundary": False,
|
||||
"avg_energy": 0.4,
|
||||
"peak_emphasis": 0.3,
|
||||
"emotion": "neutral",
|
||||
"emotion_confidence": 0.0,
|
||||
"arousal": 0.4,
|
||||
"valence": 0.5,
|
||||
"words": [
|
||||
{"text": "E", "start": 6.1, "end": 6.3, "speaker": "SPEAKER_00",
|
||||
"energy": 0.3, "pitch_delta": 0.2, "rate_delta": 0.1, "pause_before": 0.1,
|
||||
"emphasis": 0.3, "emotion": "neutral", "emotion_confidence": 0.0,
|
||||
"arousal": 0.4, "valence": 0.5, "energy_raw": 0.3, "pitch_hz": 120.0},
|
||||
{"text": "cara", "start": 6.4, "end": 6.7, "speaker": "SPEAKER_00",
|
||||
"energy": 0.4, "pitch_delta": 0.3, "rate_delta": 0.2, "pause_before": 0.0,
|
||||
"emphasis": 0.4, "emotion": "neutral", "emotion_confidence": 0.0,
|
||||
"arousal": 0.4, "valence": 0.5, "energy_raw": 0.4, "pitch_hz": 122.0},
|
||||
],
|
||||
},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def test_extract_json_unfenced():
|
||||
text = '{"source": "x", "actions": []}'
|
||||
assert _extract_json(text) == {"source": "x", "actions": []}
|
||||
|
||||
|
||||
def test_extract_json_with_fence():
|
||||
text = '```json\n{"source": "x", "actions": [{"kind": "cut", "start": 1, "end": 2}]}\n```'
|
||||
data = _extract_json(text)
|
||||
assert data["source"] == "x"
|
||||
assert data["actions"][0]["kind"] == "cut"
|
||||
|
||||
|
||||
def test_extract_json_with_prose_around():
|
||||
text = 'Aqui está:\n{"source": "x", "actions": []}\nfim.'
|
||||
assert _extract_json(text) == {"source": "x", "actions": []}
|
||||
|
||||
|
||||
def test_extract_json_invalid_returns_none():
|
||||
assert _extract_json("no json here") is None
|
||||
assert _extract_json("") is None
|
||||
|
||||
|
||||
def test_extract_json_unwraps_wrapped_list():
|
||||
# Some models wrap the expected {"source","actions"} object in a
|
||||
# single-element list; the actions must still be found.
|
||||
text = '```json\n[{"source": "x", "actions": [{"kind": "cut", "start": 1, "end": 2}]}]\n```'
|
||||
data = _extract_json(text)
|
||||
assert isinstance(data, dict)
|
||||
assert data["source"] == "x"
|
||||
assert data["actions"][0]["kind"] == "cut"
|
||||
|
||||
|
||||
def test_build_edit_messages_shape():
|
||||
system, user = build_edit_messages(_fake_timeline())
|
||||
assert "editor" in system.lower()
|
||||
assert "demo.mp4" in user
|
||||
assert "segments" in user
|
||||
assert "SOMENTE" in user
|
||||
|
||||
|
||||
def test_generate_voice_actions_parses_model_json(monkeypatch):
|
||||
canned = json.dumps({
|
||||
"source": "demo.mp4",
|
||||
"actions": [
|
||||
{"kind": "cut", "start": 6.0, "end": 12.0,
|
||||
"reason": "papo casual com a equipe"},
|
||||
{"kind": "zoom", "start": 0.1, "end": 0.9, "params": {"scale": 1.3},
|
||||
"reason": "ênfase em 'vamos'"},
|
||||
],
|
||||
})
|
||||
monkeypatch.setattr(llm_local, "ollama_chat", lambda **kwargs: canned)
|
||||
|
||||
result = generate_voice_actions(_fake_timeline(), model="test")
|
||||
actions, errors = parse_actions([a.as_dict() for a in result["actions"]])
|
||||
assert errors == []
|
||||
assert len(actions) == 2
|
||||
assert {a.kind for a in actions} == {"cut", "zoom"}
|
||||
assert result["errors"] == []
|
||||
|
||||
|
||||
def test_generate_voice_actions_reports_bad_rows(monkeypatch):
|
||||
canned = json.dumps({
|
||||
"source": "demo.mp4",
|
||||
"actions": [
|
||||
{"kind": "cut", "start": 6.0, "end": 12.0, "reason": "ok"},
|
||||
{"kind": "bogus", "start": 1, "end": 2},
|
||||
{"kind": "zoom", "start": 0.1, "end": 0.05},
|
||||
],
|
||||
})
|
||||
monkeypatch.setattr(llm_local, "ollama_chat", lambda **kwargs: canned)
|
||||
|
||||
result = generate_voice_actions(_fake_timeline(), model="test")
|
||||
assert len(result["actions"]) == 1
|
||||
assert result["actions"][0].kind == "cut"
|
||||
assert len(result["errors"]) >= 2
|
||||
|
||||
|
||||
def test_generate_voice_actions_handles_transport_error(monkeypatch):
|
||||
def _boom(**kwargs):
|
||||
raise RuntimeError("ollama down")
|
||||
monkeypatch.setattr(llm_local, "ollama_chat", _boom)
|
||||
|
||||
result = generate_voice_actions(_fake_timeline(), model="test")
|
||||
assert result["actions"] == []
|
||||
assert any("ollama" in e.lower() for e in result["errors"])
|
||||
|
||||
|
||||
def test_build_edit_messages_is_compact():
|
||||
"""The prompt must drop the heavy per-word audio features so a real
|
||||
recording fits in the model context (the 47k-token dump made Ollama drop
|
||||
the connection)."""
|
||||
import json as _json
|
||||
|
||||
timeline = _fake_timeline()
|
||||
# pad with the kind of audio detail a real timeline carries
|
||||
timeline["segments"][0]["words"][0]["samples"] = [0.1, 0.2, 0.3]
|
||||
timeline["speakers"][0]["samples"] = [1, 2, 3]
|
||||
raw = _json.dumps(timeline, ensure_ascii=False)
|
||||
system, user = build_edit_messages(timeline)
|
||||
assert "demo.mp4" in user and "segments" in user
|
||||
# projected prompt is materially smaller than the raw timeline
|
||||
assert len(user) < len(raw)
|
||||
# heavy fields we deliberately drop never reach the model
|
||||
assert '"samples"' not in user
|
||||
assert '"energy_raw"' not in user
|
||||
assert '"pitch_hz"' not in user
|
||||
|
||||
|
||||
def test_ollama_chat_wraps_empty_response(monkeypatch):
|
||||
"""A dropped connection that yields an empty body must surface as a clear
|
||||
RuntimeError (not an unhandled JSONDecodeError crashing the pipeline)."""
|
||||
|
||||
class _Resp:
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
def json(self):
|
||||
raise ValueError("Expecting value: line 1 column 1")
|
||||
|
||||
monkeypatch.setattr(llm_local.httpx, "post", lambda *a, **k: _Resp())
|
||||
try:
|
||||
llm_local.ollama_chat(model="test", messages=[{"role": "user", "content": "x"}])
|
||||
except RuntimeError as exc:
|
||||
assert "Falha ao falar com o modelo local" in str(exc)
|
||||
else:
|
||||
raise AssertionError("expected RuntimeError")
|
||||
|
||||
|
||||
def test_list_ollama_models_parses_tags(monkeypatch):
|
||||
class _Resp:
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
def json(self):
|
||||
return {"models": [{"name": "gemma3:4b"}, {"name": "gemma3:12b"}]}
|
||||
|
||||
monkeypatch.setattr(llm_local.httpx, "get", lambda *a, **k: _Resp())
|
||||
assert llm_local.list_ollama_models() == ["gemma3:12b", "gemma3:4b"]
|
||||
|
||||
|
||||
def test_list_ollama_models_returns_empty_on_error(monkeypatch):
|
||||
def _boom(*a, **k):
|
||||
raise RuntimeError("ollama down")
|
||||
|
||||
monkeypatch.setattr(llm_local.httpx, "get", _boom)
|
||||
assert llm_local.list_ollama_models() == []
|
||||
|
||||
@@ -0,0 +1,134 @@
|
||||
"""Tests for layering subtitle generation by emphasis (etapa 5 review).
|
||||
|
||||
Covers the pure helpers in server_tools/subtitles.py that decide which words
|
||||
also get the dynamic treatment on top of the always-complete plain track, and
|
||||
which clip-relative windows a plain title must be hidden (enabled="0") under.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from server_tools.subtitles import (
|
||||
_load_emphasis_spans,
|
||||
_overlaps_any_span,
|
||||
_phrase_actions_path,
|
||||
_segments_in_spans,
|
||||
_word_in_spans,
|
||||
_words_in_spans,
|
||||
)
|
||||
|
||||
SPANS = [
|
||||
{"start": 2.0, "end": 10.7, "level": 1, "text": "abertura"},
|
||||
{"start": 127.7, "end": 135.0, "level": 2, "text": "mastopexia"},
|
||||
]
|
||||
|
||||
|
||||
def test_phrase_actions_path_matches_media_stem():
|
||||
assert _phrase_actions_path("/x/y/0E6A8290.mp4") == Path(
|
||||
"/x/y/0E6A8290_phrase_actions.json"
|
||||
)
|
||||
|
||||
|
||||
class TestLoadEmphasisSpans:
|
||||
def test_missing_file_returns_empty(self, tmp_path):
|
||||
media = tmp_path / "clip.mp4"
|
||||
media.write_bytes(b"")
|
||||
assert _load_emphasis_spans(str(media)) == []
|
||||
|
||||
def test_reads_spans_from_sibling_json(self, tmp_path):
|
||||
media = tmp_path / "clip.mp4"
|
||||
media.write_bytes(b"")
|
||||
actions_path = tmp_path / "clip_phrase_actions.json"
|
||||
actions_path.write_text(
|
||||
json.dumps({"source": "clip.mp4", "actions": [], "emphasis_spans": SPANS}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
assert _load_emphasis_spans(str(media)) == SPANS
|
||||
|
||||
def test_corrupt_json_returns_empty(self, tmp_path):
|
||||
media = tmp_path / "clip.mp4"
|
||||
media.write_bytes(b"")
|
||||
(tmp_path / "clip_phrase_actions.json").write_text("{not json", encoding="utf-8")
|
||||
assert _load_emphasis_spans(str(media)) == []
|
||||
|
||||
def test_missing_emphasis_spans_key_returns_empty(self, tmp_path):
|
||||
media = tmp_path / "clip.mp4"
|
||||
media.write_bytes(b"")
|
||||
(tmp_path / "clip_phrase_actions.json").write_text(
|
||||
json.dumps({"source": "clip.mp4", "actions": []}), encoding="utf-8"
|
||||
)
|
||||
assert _load_emphasis_spans(str(media)) == []
|
||||
|
||||
|
||||
class TestWordInSpans:
|
||||
def test_word_fully_inside_span(self):
|
||||
assert _word_in_spans(3.0, 3.4, SPANS) is True
|
||||
|
||||
def test_word_fully_outside_every_span(self):
|
||||
assert _word_in_spans(50.0, 50.4, SPANS) is False
|
||||
|
||||
def test_word_straddling_span_boundary_follows_its_midpoint(self):
|
||||
# midpoint 10.6 -> inside [2.0, 10.7)
|
||||
assert _word_in_spans(10.4, 10.8, SPANS) is True
|
||||
# midpoint 10.9 -> outside
|
||||
assert _word_in_spans(10.7, 11.1, SPANS) is False
|
||||
|
||||
def test_span_end_is_exclusive(self):
|
||||
assert _word_in_spans(10.7, 10.7, SPANS) is False
|
||||
|
||||
|
||||
class TestWordsInSpans:
|
||||
WORDS = [
|
||||
{"word": "Aquela", "start": 2.03, "end": 2.69},
|
||||
{"word": "mama", "start": 2.69, "end": 2.89},
|
||||
{"word": "fora", "start": 50.0, "end": 50.2},
|
||||
{"word": "mastopexia", "start": 127.74, "end": 128.58},
|
||||
]
|
||||
|
||||
def test_no_spans_returns_nothing(self):
|
||||
assert _words_in_spans(self.WORDS, []) == []
|
||||
|
||||
def test_keeps_only_words_inside_a_span(self):
|
||||
kept = _words_in_spans(self.WORDS, SPANS)
|
||||
assert [w["word"] for w in kept] == ["Aquela", "mama", "mastopexia"]
|
||||
|
||||
def test_does_not_remove_words_from_the_source_list(self):
|
||||
"""This is a filter for the dynamic half, not a partition — the plain
|
||||
half must still see every word, so this must never mutate `words`."""
|
||||
before = list(self.WORDS)
|
||||
_words_in_spans(self.WORDS, SPANS)
|
||||
assert self.WORDS == before
|
||||
|
||||
|
||||
class TestOverlapsAnySpan:
|
||||
CLIP_SPANS = [(0.0, 8.64), (60.0, 67.22)]
|
||||
|
||||
def test_window_inside_a_span_overlaps(self):
|
||||
assert _overlaps_any_span(1.0, 2.0, self.CLIP_SPANS) is True
|
||||
|
||||
def test_window_outside_every_span_does_not_overlap(self):
|
||||
assert _overlaps_any_span(20.0, 21.0, self.CLIP_SPANS) is False
|
||||
|
||||
def test_window_straddling_a_span_edge_overlaps(self):
|
||||
assert _overlaps_any_span(8.0, 9.0, self.CLIP_SPANS) is True
|
||||
|
||||
def test_touching_but_not_overlapping_is_not_an_overlap(self):
|
||||
assert _overlaps_any_span(8.64, 9.0, self.CLIP_SPANS) is False
|
||||
|
||||
def test_no_spans_never_overlaps(self):
|
||||
assert _overlaps_any_span(1.0, 2.0, []) is False
|
||||
|
||||
|
||||
class TestSegmentsInSpans:
|
||||
SEGMENTS = [
|
||||
{"start": 2.0, "end": 10.67},
|
||||
{"start": 40.0, "end": 43.0},
|
||||
{"start": 127.74, "end": 134.96},
|
||||
]
|
||||
|
||||
def test_no_spans_keeps_nothing(self):
|
||||
assert _segments_in_spans(self.SEGMENTS, []) == []
|
||||
|
||||
def test_keeps_only_segments_inside_a_span(self):
|
||||
kept = _segments_in_spans(self.SEGMENTS, SPANS)
|
||||
assert kept == [self.SEGMENTS[0], self.SEGMENTS[2]]
|
||||
Reference in New Issue
Block a user