Files
gart/code/tests/test_voice_features.py

138 lines
4.8 KiB
Python

"""Tests for fcpxml/voice_features.py — acoustic features.
The pure helpers (speech rate, pauses, window averaging) need no audio.
The librosa-backed extractors are skipped when the optional [intelligence]
extra is absent, matching the pattern in test_media_intel.py.
"""
import math
import struct
import wave
import pytest
from fcpxml.voice_features import (
compute_pauses,
compute_speech_rate,
extract_energy,
extract_pitch,
features_capability,
word_pitch_energy,
)
try:
import librosa # noqa: F401
LIBROSA = True
except ImportError:
LIBROSA = False
def _write_tone_wav(path: str, hz: float = 220.0, seconds: float = 2.0, rate: int = 22050) -> None:
n = int(rate * seconds)
frames = [int(20000 * math.sin(2 * math.pi * hz * i / rate)) for i in range(n)]
with wave.open(path, "w") as f:
f.setnchannels(1)
f.setsampwidth(2)
f.setframerate(rate)
f.writeframes(struct.pack("<%dh" % n, *frames))
class TestComputePauses:
def test_first_word_pause_is_time_from_zero(self):
words = [{"start": 1.5, "end": 2.0}]
assert compute_pauses(words) == [1.5]
def test_gap_between_words(self):
words = [{"start": 0.0, "end": 1.0}, {"start": 2.5, "end": 3.0}]
assert compute_pauses(words) == [0.0, 1.5]
def test_overlapping_words_clamp_to_zero(self):
words = [{"start": 0.0, "end": 2.0}, {"start": 1.0, "end": 3.0}]
assert compute_pauses(words) == [0.0, 0.0]
def test_empty_words(self):
assert compute_pauses([]) == []
class TestComputeSpeechRate:
def test_rate_counts_words_in_trailing_window(self):
# 3 words within a 3s window -> 1.0 word/sec at the last one
words = [{"start": 0.0}, {"start": 1.0}, {"start": 2.0}]
rates = compute_speech_rate(words, window_seconds=3.0)
assert rates[-1] == pytest.approx(1.0)
def test_old_words_fall_out_of_window(self):
words = [{"start": 0.0}, {"start": 100.0}]
rates = compute_speech_rate(words, window_seconds=3.0)
# only the word itself is in range at t=100
assert rates[-1] == pytest.approx(1 / 3.0)
def test_zero_window_is_not_a_division_error(self):
assert compute_speech_rate([{"start": 0.0}], window_seconds=0.0) == [0.0]
def test_empty_words(self):
assert compute_speech_rate([]) == []
class TestWordPitchEnergy:
def test_averages_track_values_within_word_span(self):
words = [{"word": "a", "start": 0.0, "end": 1.0}]
pitch = [(0.0, 100.0), (0.5, 200.0), (5.0, 999.0)]
energy = [(0.0, 0.2), (1.0, 0.4)]
out = word_pitch_energy(words, pitch, energy)
assert out[0]["pitch_hz"] == pytest.approx(150.0)
assert out[0]["energy"] == pytest.approx(0.3)
def test_none_when_no_frames_in_span(self):
words = [{"word": "a", "start": 10.0, "end": 11.0}]
out = word_pitch_energy(words, [(0.0, 100.0)], [(0.0, 0.5)])
assert out[0]["pitch_hz"] is None
assert out[0]["energy"] is None
def test_none_tracks_degrade_gracefully(self):
out = word_pitch_energy([{"word": "a", "start": 0.0, "end": 1.0}], None, None)
assert out[0]["pitch_hz"] is None
assert out[0]["energy"] is None
def test_does_not_mutate_input(self):
words = [{"word": "a", "start": 0.0, "end": 1.0}]
word_pitch_energy(words, [(0.0, 100.0)], None)
assert "pitch_hz" not in words[0]
def test_word_shorter_than_hop_gets_none_not_crash(self):
"""A word briefer than the frame spacing may contain no frame at all."""
words = [{"word": "a", "start": 0.501, "end": 0.502}]
out = word_pitch_energy(words, [(0.0, 100.0), (1.0, 200.0)], None)
assert out[0]["pitch_hz"] is None
class TestExtractorsDegradeGracefully:
def test_missing_file_returns_none(self):
assert extract_pitch("/nonexistent/audio.wav") is None
assert extract_energy("/nonexistent/audio.wav") is None
@pytest.mark.skipif(not LIBROSA, reason="librosa not installed")
class TestExtractorsWithLibrosa:
def test_capability_is_available(self):
ok, _msg = features_capability()
assert ok is True
def test_extracts_pitch_of_known_tone(self, tmp_path):
wav = tmp_path / "tone.wav"
_write_tone_wav(str(wav), hz=220.0, seconds=2.0)
track = extract_pitch(str(wav))
assert track is not None and len(track) > 0
hz_values = sorted(hz for _t, hz in track)
median = hz_values[len(hz_values) // 2]
assert median == pytest.approx(220.0, rel=0.1)
def test_extracts_energy_track(self, tmp_path):
wav = tmp_path / "tone.wav"
_write_tone_wav(str(wav), seconds=1.0)
track = extract_energy(str(wav))
assert track is not None and len(track) > 0
assert all(rms >= 0 for _t, rms in track)
assert max(rms for _t, rms in track) > 0