Files
gart/code/tests/test_voice_timeline.py

339 lines
14 KiB
Python

"""Tests for fcpxml/voice_timeline.py — the consolidated AI-readable timeline.
The document's shape is the contract downstream consumers (rules engine, a
model reading the JSON) rely on, so these tests pin the shape as much as
the values — including that it survives every analysis layer being absent.
"""
import json
import pytest
from fcpxml.voice_timeline import (
VOICE_TIMELINE_VERSION,
build_voice_timeline,
enrich_words,
load_voice_timeline,
save_voice_timeline,
voice_timeline_path,
)
_TRANSCRIPT = {
"language": "pt",
"duration": 4.0,
"text": "isso e seguranca total",
"segments": [
{"text": "isso e", "start": 0.0, "end": 1.0},
{"text": "seguranca total", "start": 2.0, "end": 4.0},
],
"words": [
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
{"word": "total", "start": 3.0, "end": 3.5, "confidence": 0.9},
],
}
# "seguranca" is the loud, high-pitched moment
_PITCH = [(0.2, 120.0), (0.6, 118.0), (2.4, 260.0), (3.2, 130.0)]
_ENERGY = [(0.2, 0.10), (0.6, 0.12), (2.4, 0.95), (3.2, 0.20)]
class TestEnrichWords:
def test_normalizes_energy_against_loudest_word(self):
enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY)
loudest = max(enriched, key=lambda w: w["energy_norm"])
assert loudest["word"] == "seguranca"
assert loudest["energy_norm"] == pytest.approx(1.0)
def test_all_values_stay_within_unit_range(self):
enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY)
for w in enriched:
for key in ("energy_norm", "pitch_delta", "rate_delta", "emphasis"):
assert 0.0 <= w[key] <= 1.0, f"{key} out of range on {w['word']}"
def test_empty_words_returns_empty(self):
assert enrich_words([], _PITCH, _ENERGY) == []
def test_missing_tracks_give_zero_not_crash(self):
enriched = enrich_words(_TRANSCRIPT["words"], None, None)
assert all(w["energy_norm"] == 0.0 for w in enriched)
assert all(w["pitch_delta"] == 0.0 for w in enriched)
class TestBuildVoiceTimeline:
@pytest.fixture
def timeline(self, monkeypatch):
import fcpxml.voice_timeline as vt
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: _PITCH)
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: _ENERGY)
return build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT)
def test_document_has_all_top_level_layers(self, timeline):
for key in ("version", "source", "language", "scales", "summary", "speakers", "segments"):
assert key in timeline
assert timeline["version"] == VOICE_TIMELINE_VERSION
def test_scales_document_every_word_metric(self, timeline):
word = timeline["segments"][0]["words"][0]
for metric in timeline["scales"]["word"]:
assert metric in word, f"{metric} documented in scales but absent from words"
def test_scales_document_every_segment_metric(self, timeline):
segment = timeline["segments"][0]
for metric in timeline["scales"]["segment"]:
assert metric in segment, f"{metric} documented in scales but absent from segments"
def test_take_boundary_flags_a_long_gap(self, timeline):
# the fixture has a 1s gap between its two segments -> not a boundary
assert timeline["segments"][1]["gap_before"] > 0
assert timeline["segments"][1]["take_boundary"] is False
def test_summary_counts_match_the_detail(self, timeline):
summary = timeline["summary"]
assert summary["segment_count"] == len(timeline["segments"])
total_words = sum(len(s["words"]) for s in timeline["segments"])
assert summary["word_count"] == total_words
def test_words_are_grouped_under_their_segment(self, timeline):
first, second = timeline["segments"]
assert [w["text"] for w in first["words"]] == ["isso", "e"]
assert [w["text"] for w in second["words"]] == ["seguranca", "total"]
def test_segment_aggregates_reflect_their_words(self, timeline):
loud_segment = timeline["segments"][1]
quiet_segment = timeline["segments"][0]
assert loud_segment["avg_energy"] > quiet_segment["avg_energy"]
assert loud_segment["peak_emphasis"] >= max(w["emphasis"] for w in loud_segment["words"])
def test_peak_moments_are_sorted_by_emphasis(self, timeline):
peaks = timeline["summary"]["peak_moments"]
assert peaks == sorted(peaks, key=lambda m: m["emphasis"], reverse=True)
def test_defaults_to_single_speaker_without_token(self, timeline):
assert timeline["summary"]["speaker_count"] == 1
assert all(w["speaker"] == "SPEAKER_00" for s in timeline["segments"] for w in s["words"])
def test_is_json_serializable(self, timeline):
# the whole point is handing this to a model / writing it to disk
assert json.loads(json.dumps(timeline, ensure_ascii=False))["version"]
class TestDegradesWithoutAnalysisLayers:
def test_shape_survives_with_no_acoustics(self, monkeypatch):
import fcpxml.voice_timeline as vt
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None)
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None)
timeline = build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT)
assert timeline["summary"]["word_count"] == 4
assert timeline["summary"]["avg_emphasis"] >= 0.0
def test_empty_transcript_still_yields_valid_document(self, monkeypatch):
import fcpxml.voice_timeline as vt
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None)
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None)
timeline = build_voice_timeline(
"/tmp/clip.wav", {"duration": 0.0, "segments": [], "words": []}
)
assert timeline["segments"] == []
assert timeline["summary"]["word_count"] == 0
assert timeline["summary"]["avg_emphasis"] == 0.0
def test_progress_callback_is_reported(self, monkeypatch):
import fcpxml.voice_timeline as vt
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None)
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None)
seen: list[tuple[float, str]] = []
build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT, progress_cb=lambda f, s: seen.append((f, s)))
assert seen and all(0.0 <= f <= 1.0 for f, _ in seen)
class TestPersistence:
def test_round_trip(self, tmp_path):
path = tmp_path / "clip_voice_timeline.json"
timeline = {"version": "1.0", "segments": [], "summary": {}}
save_voice_timeline(timeline, path)
assert load_voice_timeline(path) == timeline
def test_accented_text_stays_readable(self, tmp_path):
path = tmp_path / "t.json"
save_voice_timeline({"segments": [{"text": "segurança"}]}, path)
assert "segurança" in path.read_text(encoding="utf-8")
def test_missing_file_returns_none(self, tmp_path):
assert load_voice_timeline(tmp_path / "absent.json") is None
def test_malformed_json_returns_none(self, tmp_path):
path = tmp_path / "bad.json"
path.write_text("{not json")
assert load_voice_timeline(path) is None
def test_wrong_shape_returns_none(self, tmp_path):
path = tmp_path / "other.json"
path.write_text('{"something": "else"}')
assert load_voice_timeline(path) is None
def test_path_next_to_media_by_default(self):
assert voice_timeline_path("/media/clip.mov").name == "clip_voice_timeline.json"
def test_path_honours_output_dir(self, tmp_path):
path = voice_timeline_path("/media/clip.mov", output_dir=str(tmp_path))
assert path.parent == tmp_path
_RAW_WORDS = [
# a loud outlier that will be cut, plus quieter material that survives
{"text": "GRITO", "start": 1.0, "end": 1.5, "speaker": "SPEAKER_00",
"energy": 0.5, "pitch_delta": 0.5, "rate_delta": 0.0, "pause_before": 0.0,
"emphasis": 0.5, "energy_raw": 1.0, "pitch_hz": 300.0},
{"text": "mastopexia", "start": 10.0, "end": 10.8, "speaker": "SPEAKER_00",
"energy": 0.2, "pitch_delta": 0.1, "rate_delta": 0.0, "pause_before": 0.0,
"emphasis": 0.1, "energy_raw": 0.4, "pitch_hz": 190.0},
{"text": "a", "start": 11.0, "end": 11.1, "speaker": "SPEAKER_00",
"energy": 0.15, "pitch_delta": 0.05, "rate_delta": 0.0, "pause_before": 0.0,
"emphasis": 0.08, "energy_raw": 0.3, "pitch_hz": 185.0},
]
_RESTRICT_TIMELINE = {
"version": "1.0", "source": "x.mp4", "speakers": [],
"segments": [
{"start": 1.0, "end": 1.5, "speaker": "SPEAKER_00", "text": "GRITO",
"gap_before": 0.0, "take_boundary": False, "avg_energy": 0.5,
"peak_emphasis": 0.5, "words": [_RAW_WORDS[0]]},
{"start": 10.0, "end": 11.1, "speaker": "SPEAKER_00",
"text": "mastopexia a", "gap_before": 8.5, "take_boundary": True,
"avg_energy": 0.17, "peak_emphasis": 0.1, "words": _RAW_WORDS[1:]},
],
}
class TestRestrictToKept:
"""Emphasis is relative. Cut the loudest moment out and everything left
is still scored against something the viewer will never see, so the
surviving material has to be re-normalized on its own."""
def test_cut_words_are_dropped(self):
from fcpxml.voice_timeline import restrict_to_kept
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
texts = [w["text"] for s in r["segments"] for w in s["words"]]
assert "GRITO" not in texts
assert "mastopexia" in texts
def test_survivors_are_rescored_against_each_other(self):
from fcpxml.voice_timeline import restrict_to_kept
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
word = next(w for s in r["segments"] for w in s["words"] if w["text"] == "mastopexia")
# was 0.2 against the shout's 1.0; alone it becomes the loudest
assert word["energy"] == pytest.approx(1.0)
def test_empty_segments_are_removed(self):
from fcpxml.voice_timeline import restrict_to_kept
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
assert len(r["segments"]) == 1
def test_no_cuts_keeps_everything(self):
from fcpxml.voice_timeline import restrict_to_kept
r = restrict_to_kept(_RESTRICT_TIMELINE, [])
assert sum(len(s["words"]) for s in r["segments"]) == 3
def test_times_stay_in_original_source_seconds(self):
from fcpxml.voice_timeline import restrict_to_kept
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
assert r["segments"][0]["start"] == 10.0
class TestSuggestZoomWindows:
def test_skips_function_words(self):
from fcpxml.voice_timeline import restrict_to_kept, suggest_zoom_windows
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
zooms = suggest_zoom_windows(r)
assert zooms and all(z["word"] != "a" for z in zooms)
def test_window_runs_from_the_word_to_the_end_of_its_line(self):
from fcpxml.voice_timeline import restrict_to_kept, suggest_zoom_windows
r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)])
z = suggest_zoom_windows(r)[0]
assert z["start"] == 10.0 and z["end"] == 11.1
def test_min_gap_keeps_zooms_apart(self):
from fcpxml.voice_timeline import suggest_zoom_windows
timeline = {"segments": [
{"start": t, "end": t + 1.0, "text": "linha",
"words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": 0.5 - i * 0.01}]}
for i, t in enumerate([0.0, 1.0, 2.0, 30.0])
]}
zooms = suggest_zoom_windows(timeline, min_gap=8.0)
assert len(zooms) == 2
def test_max_zooms_caps_the_result(self):
from fcpxml.voice_timeline import suggest_zoom_windows
timeline = {"segments": [
{"start": t, "end": t + 1.0, "text": "linha",
"words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": 0.5}]}
for t in [0.0, 20.0, 40.0, 60.0]
]}
assert len(suggest_zoom_windows(timeline, min_gap=8.0, max_zooms=2)) == 2
def test_results_are_in_chronological_order(self):
from fcpxml.voice_timeline import suggest_zoom_windows
timeline = {"segments": [
{"start": t, "end": t + 1.0, "text": "linha",
"words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": e}]}
for t, e in [(60.0, 0.9), (0.0, 0.5), (30.0, 0.7)]
]}
zooms = suggest_zoom_windows(timeline, min_gap=8.0)
assert [z["start"] for z in zooms] == sorted(z["start"] for z in zooms)
class TestSentenceEnd:
"""Transcription segments break on breath, not grammar — a sentence
routinely spans several. A zoom ending on a segment boundary releases
mid-thought, which is what makes a punch-in feel arbitrary."""
SEGS = [
{"start": 0.0, "end": 5.0, "text": "Aquela mama com um formato, que dá aquele ar",
"take_boundary": False},
{"start": 5.0, "end": 10.7, "text": "de elegância, isso é desejo de muitas mulheres, né?",
"take_boundary": False},
{"start": 11.0, "end": 14.0, "text": "Com o tempo, o corpo muda.", "take_boundary": False},
]
def test_extends_past_a_segment_that_does_not_end_a_sentence(self):
from fcpxml.voice_timeline import sentence_end
assert sentence_end(self.SEGS, 0) == 10.7
def test_stops_at_terminal_punctuation(self):
from fcpxml.voice_timeline import sentence_end
assert sentence_end(self.SEGS, 2) == 14.0
def test_never_runs_past_a_take_boundary(self):
from fcpxml.voice_timeline import sentence_end
segs = [
{"start": 0.0, "end": 5.0, "text": "frase sem fim", "take_boundary": False},
{"start": 12.0, "end": 15.0, "text": "outra tomada", "take_boundary": True},
]
assert sentence_end(segs, 0) == 5.0
def test_last_segment_without_punctuation_ends_at_itself(self):
from fcpxml.voice_timeline import sentence_end
segs = [{"start": 0.0, "end": 4.0, "text": "sem ponto final", "take_boundary": False}]
assert sentence_end(segs, 0) == 4.0