"""Tests for fcpxml/voice_features.py — acoustic features. The pure helpers (speech rate, pauses, window averaging) need no audio. The librosa-backed extractors are skipped when the optional [intelligence] extra is absent, matching the pattern in test_media_intel.py. """ import math import struct import wave import pytest from fcpxml.voice_features import ( compute_pauses, compute_speech_rate, extract_energy, extract_pitch, features_capability, word_pitch_energy, ) try: import librosa # noqa: F401 LIBROSA = True except ImportError: LIBROSA = False def _write_tone_wav(path: str, hz: float = 220.0, seconds: float = 2.0, rate: int = 22050) -> None: n = int(rate * seconds) frames = [int(20000 * math.sin(2 * math.pi * hz * i / rate)) for i in range(n)] with wave.open(path, "w") as f: f.setnchannels(1) f.setsampwidth(2) f.setframerate(rate) f.writeframes(struct.pack("<%dh" % n, *frames)) class TestComputePauses: def test_first_word_pause_is_time_from_zero(self): words = [{"start": 1.5, "end": 2.0}] assert compute_pauses(words) == [1.5] def test_gap_between_words(self): words = [{"start": 0.0, "end": 1.0}, {"start": 2.5, "end": 3.0}] assert compute_pauses(words) == [0.0, 1.5] def test_overlapping_words_clamp_to_zero(self): words = [{"start": 0.0, "end": 2.0}, {"start": 1.0, "end": 3.0}] assert compute_pauses(words) == [0.0, 0.0] def test_empty_words(self): assert compute_pauses([]) == [] class TestComputeSpeechRate: def test_rate_counts_words_in_trailing_window(self): # 3 words within a 3s window -> 1.0 word/sec at the last one words = [{"start": 0.0}, {"start": 1.0}, {"start": 2.0}] rates = compute_speech_rate(words, window_seconds=3.0) assert rates[-1] == pytest.approx(1.0) def test_old_words_fall_out_of_window(self): words = [{"start": 0.0}, {"start": 100.0}] rates = compute_speech_rate(words, window_seconds=3.0) # only the word itself is in range at t=100 assert rates[-1] == pytest.approx(1 / 3.0) def test_zero_window_is_not_a_division_error(self): assert compute_speech_rate([{"start": 0.0}], window_seconds=0.0) == [0.0] def test_empty_words(self): assert compute_speech_rate([]) == [] class TestWordPitchEnergy: def test_averages_track_values_within_word_span(self): words = [{"word": "a", "start": 0.0, "end": 1.0}] pitch = [(0.0, 100.0), (0.5, 200.0), (5.0, 999.0)] energy = [(0.0, 0.2), (1.0, 0.4)] out = word_pitch_energy(words, pitch, energy) assert out[0]["pitch_hz"] == pytest.approx(150.0) assert out[0]["energy"] == pytest.approx(0.3) def test_none_when_no_frames_in_span(self): words = [{"word": "a", "start": 10.0, "end": 11.0}] out = word_pitch_energy(words, [(0.0, 100.0)], [(0.0, 0.5)]) assert out[0]["pitch_hz"] is None assert out[0]["energy"] is None def test_none_tracks_degrade_gracefully(self): out = word_pitch_energy([{"word": "a", "start": 0.0, "end": 1.0}], None, None) assert out[0]["pitch_hz"] is None assert out[0]["energy"] is None def test_does_not_mutate_input(self): words = [{"word": "a", "start": 0.0, "end": 1.0}] word_pitch_energy(words, [(0.0, 100.0)], None) assert "pitch_hz" not in words[0] def test_word_shorter_than_hop_gets_none_not_crash(self): """A word briefer than the frame spacing may contain no frame at all.""" words = [{"word": "a", "start": 0.501, "end": 0.502}] out = word_pitch_energy(words, [(0.0, 100.0), (1.0, 200.0)], None) assert out[0]["pitch_hz"] is None class TestExtractorsDegradeGracefully: def test_missing_file_returns_none(self): assert extract_pitch("/nonexistent/audio.wav") is None assert extract_energy("/nonexistent/audio.wav") is None @pytest.mark.skipif(not LIBROSA, reason="librosa not installed") class TestExtractorsWithLibrosa: def test_capability_is_available(self): ok, _msg = features_capability() assert ok is True def test_extracts_pitch_of_known_tone(self, tmp_path): wav = tmp_path / "tone.wav" _write_tone_wav(str(wav), hz=220.0, seconds=2.0) track = extract_pitch(str(wav)) assert track is not None and len(track) > 0 hz_values = sorted(hz for _t, hz in track) median = hz_values[len(hz_values) // 2] assert median == pytest.approx(220.0, rel=0.1) def test_extracts_energy_track(self, tmp_path): wav = tmp_path / "tone.wav" _write_tone_wav(str(wav), seconds=1.0) track = extract_energy(str(wav)) assert track is not None and len(track) > 0 assert all(rms >= 0 for _t, rms in track) assert max(rms for _t, rms in track) > 0