"""Tests for fcpxml/voice_timeline.py — the consolidated AI-readable timeline. The document's shape is the contract downstream consumers (rules engine, a model reading the JSON) rely on, so these tests pin the shape as much as the values — including that it survives every analysis layer being absent. """ import json import pytest from fcpxml.voice_timeline import ( VOICE_TIMELINE_VERSION, build_voice_timeline, enrich_words, load_voice_timeline, save_voice_timeline, voice_timeline_path, ) _TRANSCRIPT = { "language": "pt", "duration": 4.0, "text": "isso e seguranca total", "segments": [ {"text": "isso e", "start": 0.0, "end": 1.0}, {"text": "seguranca total", "start": 2.0, "end": 4.0}, ], "words": [ {"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9}, {"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9}, {"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9}, {"word": "total", "start": 3.0, "end": 3.5, "confidence": 0.9}, ], } # "seguranca" is the loud, high-pitched moment _PITCH = [(0.2, 120.0), (0.6, 118.0), (2.4, 260.0), (3.2, 130.0)] _ENERGY = [(0.2, 0.10), (0.6, 0.12), (2.4, 0.95), (3.2, 0.20)] class TestEnrichWords: def test_normalizes_energy_against_loudest_word(self): enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY) loudest = max(enriched, key=lambda w: w["energy_norm"]) assert loudest["word"] == "seguranca" assert loudest["energy_norm"] == pytest.approx(1.0) def test_all_values_stay_within_unit_range(self): enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY) for w in enriched: for key in ("energy_norm", "pitch_delta", "rate_delta", "emphasis"): assert 0.0 <= w[key] <= 1.0, f"{key} out of range on {w['word']}" def test_empty_words_returns_empty(self): assert enrich_words([], _PITCH, _ENERGY) == [] def test_missing_tracks_give_zero_not_crash(self): enriched = enrich_words(_TRANSCRIPT["words"], None, None) assert all(w["energy_norm"] == 0.0 for w in enriched) assert all(w["pitch_delta"] == 0.0 for w in enriched) class TestBuildVoiceTimeline: @pytest.fixture def timeline(self, monkeypatch): import fcpxml.voice_timeline as vt monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: _PITCH) monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: _ENERGY) return build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT) def test_document_has_all_top_level_layers(self, timeline): for key in ("version", "source", "language", "scales", "summary", "speakers", "segments"): assert key in timeline assert timeline["version"] == VOICE_TIMELINE_VERSION def test_scales_document_every_word_metric(self, timeline): word = timeline["segments"][0]["words"][0] for metric in timeline["scales"]["word"]: assert metric in word, f"{metric} documented in scales but absent from words" def test_scales_document_every_segment_metric(self, timeline): segment = timeline["segments"][0] for metric in timeline["scales"]["segment"]: assert metric in segment, f"{metric} documented in scales but absent from segments" def test_take_boundary_flags_a_long_gap(self, timeline): # the fixture has a 1s gap between its two segments -> not a boundary assert timeline["segments"][1]["gap_before"] > 0 assert timeline["segments"][1]["take_boundary"] is False def test_summary_counts_match_the_detail(self, timeline): summary = timeline["summary"] assert summary["segment_count"] == len(timeline["segments"]) total_words = sum(len(s["words"]) for s in timeline["segments"]) assert summary["word_count"] == total_words def test_words_are_grouped_under_their_segment(self, timeline): first, second = timeline["segments"] assert [w["text"] for w in first["words"]] == ["isso", "e"] assert [w["text"] for w in second["words"]] == ["seguranca", "total"] def test_segment_aggregates_reflect_their_words(self, timeline): loud_segment = timeline["segments"][1] quiet_segment = timeline["segments"][0] assert loud_segment["avg_energy"] > quiet_segment["avg_energy"] assert loud_segment["peak_emphasis"] >= max(w["emphasis"] for w in loud_segment["words"]) def test_peak_moments_are_sorted_by_emphasis(self, timeline): peaks = timeline["summary"]["peak_moments"] assert peaks == sorted(peaks, key=lambda m: m["emphasis"], reverse=True) def test_defaults_to_single_speaker_without_token(self, timeline): assert timeline["summary"]["speaker_count"] == 1 assert all(w["speaker"] == "SPEAKER_00" for s in timeline["segments"] for w in s["words"]) def test_is_json_serializable(self, timeline): # the whole point is handing this to a model / writing it to disk assert json.loads(json.dumps(timeline, ensure_ascii=False))["version"] class TestDegradesWithoutAnalysisLayers: def test_shape_survives_with_no_acoustics(self, monkeypatch): import fcpxml.voice_timeline as vt monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None) monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None) timeline = build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT) assert timeline["summary"]["word_count"] == 4 assert timeline["summary"]["avg_emphasis"] >= 0.0 def test_empty_transcript_still_yields_valid_document(self, monkeypatch): import fcpxml.voice_timeline as vt monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None) monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None) timeline = build_voice_timeline( "/tmp/clip.wav", {"duration": 0.0, "segments": [], "words": []} ) assert timeline["segments"] == [] assert timeline["summary"]["word_count"] == 0 assert timeline["summary"]["avg_emphasis"] == 0.0 def test_progress_callback_is_reported(self, monkeypatch): import fcpxml.voice_timeline as vt monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: None) monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: None) seen: list[tuple[float, str]] = [] build_voice_timeline("/tmp/clip.wav", _TRANSCRIPT, progress_cb=lambda f, s: seen.append((f, s))) assert seen and all(0.0 <= f <= 1.0 for f, _ in seen) class TestPersistence: def test_round_trip(self, tmp_path): path = tmp_path / "clip_voice_timeline.json" timeline = {"version": "1.0", "segments": [], "summary": {}} save_voice_timeline(timeline, path) assert load_voice_timeline(path) == timeline def test_accented_text_stays_readable(self, tmp_path): path = tmp_path / "t.json" save_voice_timeline({"segments": [{"text": "segurança"}]}, path) assert "segurança" in path.read_text(encoding="utf-8") def test_missing_file_returns_none(self, tmp_path): assert load_voice_timeline(tmp_path / "absent.json") is None def test_malformed_json_returns_none(self, tmp_path): path = tmp_path / "bad.json" path.write_text("{not json") assert load_voice_timeline(path) is None def test_wrong_shape_returns_none(self, tmp_path): path = tmp_path / "other.json" path.write_text('{"something": "else"}') assert load_voice_timeline(path) is None def test_path_next_to_media_by_default(self): assert voice_timeline_path("/media/clip.mov").name == "clip_voice_timeline.json" def test_path_honours_output_dir(self, tmp_path): path = voice_timeline_path("/media/clip.mov", output_dir=str(tmp_path)) assert path.parent == tmp_path _RAW_WORDS = [ # a loud outlier that will be cut, plus quieter material that survives {"text": "GRITO", "start": 1.0, "end": 1.5, "speaker": "SPEAKER_00", "energy": 0.5, "pitch_delta": 0.5, "rate_delta": 0.0, "pause_before": 0.0, "emphasis": 0.5, "energy_raw": 1.0, "pitch_hz": 300.0}, {"text": "mastopexia", "start": 10.0, "end": 10.8, "speaker": "SPEAKER_00", "energy": 0.2, "pitch_delta": 0.1, "rate_delta": 0.0, "pause_before": 0.0, "emphasis": 0.1, "energy_raw": 0.4, "pitch_hz": 190.0}, {"text": "a", "start": 11.0, "end": 11.1, "speaker": "SPEAKER_00", "energy": 0.15, "pitch_delta": 0.05, "rate_delta": 0.0, "pause_before": 0.0, "emphasis": 0.08, "energy_raw": 0.3, "pitch_hz": 185.0}, ] _RESTRICT_TIMELINE = { "version": "1.0", "source": "x.mp4", "speakers": [], "segments": [ {"start": 1.0, "end": 1.5, "speaker": "SPEAKER_00", "text": "GRITO", "gap_before": 0.0, "take_boundary": False, "avg_energy": 0.5, "peak_emphasis": 0.5, "words": [_RAW_WORDS[0]]}, {"start": 10.0, "end": 11.1, "speaker": "SPEAKER_00", "text": "mastopexia a", "gap_before": 8.5, "take_boundary": True, "avg_energy": 0.17, "peak_emphasis": 0.1, "words": _RAW_WORDS[1:]}, ], } class TestRestrictToKept: """Emphasis is relative. Cut the loudest moment out and everything left is still scored against something the viewer will never see, so the surviving material has to be re-normalized on its own.""" def test_cut_words_are_dropped(self): from fcpxml.voice_timeline import restrict_to_kept r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)]) texts = [w["text"] for s in r["segments"] for w in s["words"]] assert "GRITO" not in texts assert "mastopexia" in texts def test_survivors_are_rescored_against_each_other(self): from fcpxml.voice_timeline import restrict_to_kept r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)]) word = next(w for s in r["segments"] for w in s["words"] if w["text"] == "mastopexia") # was 0.2 against the shout's 1.0; alone it becomes the loudest assert word["energy"] == pytest.approx(1.0) def test_empty_segments_are_removed(self): from fcpxml.voice_timeline import restrict_to_kept r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)]) assert len(r["segments"]) == 1 def test_no_cuts_keeps_everything(self): from fcpxml.voice_timeline import restrict_to_kept r = restrict_to_kept(_RESTRICT_TIMELINE, []) assert sum(len(s["words"]) for s in r["segments"]) == 3 def test_times_stay_in_original_source_seconds(self): from fcpxml.voice_timeline import restrict_to_kept r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)]) assert r["segments"][0]["start"] == 10.0 class TestSuggestZoomWindows: def test_skips_function_words(self): from fcpxml.voice_timeline import restrict_to_kept, suggest_zoom_windows r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)]) zooms = suggest_zoom_windows(r) assert zooms and all(z["word"] != "a" for z in zooms) def test_window_runs_from_the_word_to_the_end_of_its_line(self): from fcpxml.voice_timeline import restrict_to_kept, suggest_zoom_windows r = restrict_to_kept(_RESTRICT_TIMELINE, [(0.0, 5.0)]) z = suggest_zoom_windows(r)[0] assert z["start"] == 10.0 and z["end"] == 11.1 def test_min_gap_keeps_zooms_apart(self): from fcpxml.voice_timeline import suggest_zoom_windows timeline = {"segments": [ {"start": t, "end": t + 1.0, "text": "linha", "words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": 0.5 - i * 0.01}]} for i, t in enumerate([0.0, 1.0, 2.0, 30.0]) ]} zooms = suggest_zoom_windows(timeline, min_gap=8.0) assert len(zooms) == 2 def test_max_zooms_caps_the_result(self): from fcpxml.voice_timeline import suggest_zoom_windows timeline = {"segments": [ {"start": t, "end": t + 1.0, "text": "linha", "words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": 0.5}]} for t in [0.0, 20.0, 40.0, 60.0] ]} assert len(suggest_zoom_windows(timeline, min_gap=8.0, max_zooms=2)) == 2 def test_results_are_in_chronological_order(self): from fcpxml.voice_timeline import suggest_zoom_windows timeline = {"segments": [ {"start": t, "end": t + 1.0, "text": "linha", "words": [{"text": "palavra", "start": t, "end": t + 0.5, "emphasis": e}]} for t, e in [(60.0, 0.9), (0.0, 0.5), (30.0, 0.7)] ]} zooms = suggest_zoom_windows(timeline, min_gap=8.0) assert [z["start"] for z in zooms] == sorted(z["start"] for z in zooms) class TestSentenceEnd: """Transcription segments break on breath, not grammar — a sentence routinely spans several. A zoom ending on a segment boundary releases mid-thought, which is what makes a punch-in feel arbitrary.""" SEGS = [ {"start": 0.0, "end": 5.0, "text": "Aquela mama com um formato, que dá aquele ar", "take_boundary": False}, {"start": 5.0, "end": 10.7, "text": "de elegância, isso é desejo de muitas mulheres, né?", "take_boundary": False}, {"start": 11.0, "end": 14.0, "text": "Com o tempo, o corpo muda.", "take_boundary": False}, ] def test_extends_past_a_segment_that_does_not_end_a_sentence(self): from fcpxml.voice_timeline import sentence_end assert sentence_end(self.SEGS, 0) == 10.7 def test_stops_at_terminal_punctuation(self): from fcpxml.voice_timeline import sentence_end assert sentence_end(self.SEGS, 2) == 14.0 def test_never_runs_past_a_take_boundary(self): from fcpxml.voice_timeline import sentence_end segs = [ {"start": 0.0, "end": 5.0, "text": "frase sem fim", "take_boundary": False}, {"start": 12.0, "end": 15.0, "text": "outra tomada", "take_boundary": True}, ] assert sentence_end(segs, 0) == 5.0 def test_last_segment_without_punctuation_ends_at_itself(self): from fcpxml.voice_timeline import sentence_end segs = [{"start": 0.0, "end": 4.0, "text": "sem ponto final", "take_boundary": False}] assert sentence_end(segs, 0) == 4.0