138 lines
4.8 KiB
Python
138 lines
4.8 KiB
Python
"""Tests for fcpxml/voice_features.py — acoustic features.
|
|
|
|
The pure helpers (speech rate, pauses, window averaging) need no audio.
|
|
The librosa-backed extractors are skipped when the optional [intelligence]
|
|
extra is absent, matching the pattern in test_media_intel.py.
|
|
"""
|
|
|
|
import math
|
|
import struct
|
|
import wave
|
|
|
|
import pytest
|
|
|
|
from fcpxml.voice_features import (
|
|
compute_pauses,
|
|
compute_speech_rate,
|
|
extract_energy,
|
|
extract_pitch,
|
|
features_capability,
|
|
word_pitch_energy,
|
|
)
|
|
|
|
try:
|
|
import librosa # noqa: F401
|
|
|
|
LIBROSA = True
|
|
except ImportError:
|
|
LIBROSA = False
|
|
|
|
|
|
def _write_tone_wav(path: str, hz: float = 220.0, seconds: float = 2.0, rate: int = 22050) -> None:
|
|
n = int(rate * seconds)
|
|
frames = [int(20000 * math.sin(2 * math.pi * hz * i / rate)) for i in range(n)]
|
|
with wave.open(path, "w") as f:
|
|
f.setnchannels(1)
|
|
f.setsampwidth(2)
|
|
f.setframerate(rate)
|
|
f.writeframes(struct.pack("<%dh" % n, *frames))
|
|
|
|
|
|
class TestComputePauses:
|
|
def test_first_word_pause_is_time_from_zero(self):
|
|
words = [{"start": 1.5, "end": 2.0}]
|
|
assert compute_pauses(words) == [1.5]
|
|
|
|
def test_gap_between_words(self):
|
|
words = [{"start": 0.0, "end": 1.0}, {"start": 2.5, "end": 3.0}]
|
|
assert compute_pauses(words) == [0.0, 1.5]
|
|
|
|
def test_overlapping_words_clamp_to_zero(self):
|
|
words = [{"start": 0.0, "end": 2.0}, {"start": 1.0, "end": 3.0}]
|
|
assert compute_pauses(words) == [0.0, 0.0]
|
|
|
|
def test_empty_words(self):
|
|
assert compute_pauses([]) == []
|
|
|
|
|
|
class TestComputeSpeechRate:
|
|
def test_rate_counts_words_in_trailing_window(self):
|
|
# 3 words within a 3s window -> 1.0 word/sec at the last one
|
|
words = [{"start": 0.0}, {"start": 1.0}, {"start": 2.0}]
|
|
rates = compute_speech_rate(words, window_seconds=3.0)
|
|
assert rates[-1] == pytest.approx(1.0)
|
|
|
|
def test_old_words_fall_out_of_window(self):
|
|
words = [{"start": 0.0}, {"start": 100.0}]
|
|
rates = compute_speech_rate(words, window_seconds=3.0)
|
|
# only the word itself is in range at t=100
|
|
assert rates[-1] == pytest.approx(1 / 3.0)
|
|
|
|
def test_zero_window_is_not_a_division_error(self):
|
|
assert compute_speech_rate([{"start": 0.0}], window_seconds=0.0) == [0.0]
|
|
|
|
def test_empty_words(self):
|
|
assert compute_speech_rate([]) == []
|
|
|
|
|
|
class TestWordPitchEnergy:
|
|
def test_averages_track_values_within_word_span(self):
|
|
words = [{"word": "a", "start": 0.0, "end": 1.0}]
|
|
pitch = [(0.0, 100.0), (0.5, 200.0), (5.0, 999.0)]
|
|
energy = [(0.0, 0.2), (1.0, 0.4)]
|
|
out = word_pitch_energy(words, pitch, energy)
|
|
assert out[0]["pitch_hz"] == pytest.approx(150.0)
|
|
assert out[0]["energy"] == pytest.approx(0.3)
|
|
|
|
def test_none_when_no_frames_in_span(self):
|
|
words = [{"word": "a", "start": 10.0, "end": 11.0}]
|
|
out = word_pitch_energy(words, [(0.0, 100.0)], [(0.0, 0.5)])
|
|
assert out[0]["pitch_hz"] is None
|
|
assert out[0]["energy"] is None
|
|
|
|
def test_none_tracks_degrade_gracefully(self):
|
|
out = word_pitch_energy([{"word": "a", "start": 0.0, "end": 1.0}], None, None)
|
|
assert out[0]["pitch_hz"] is None
|
|
assert out[0]["energy"] is None
|
|
|
|
def test_does_not_mutate_input(self):
|
|
words = [{"word": "a", "start": 0.0, "end": 1.0}]
|
|
word_pitch_energy(words, [(0.0, 100.0)], None)
|
|
assert "pitch_hz" not in words[0]
|
|
|
|
def test_word_shorter_than_hop_gets_none_not_crash(self):
|
|
"""A word briefer than the frame spacing may contain no frame at all."""
|
|
words = [{"word": "a", "start": 0.501, "end": 0.502}]
|
|
out = word_pitch_energy(words, [(0.0, 100.0), (1.0, 200.0)], None)
|
|
assert out[0]["pitch_hz"] is None
|
|
|
|
|
|
class TestExtractorsDegradeGracefully:
|
|
def test_missing_file_returns_none(self):
|
|
assert extract_pitch("/nonexistent/audio.wav") is None
|
|
assert extract_energy("/nonexistent/audio.wav") is None
|
|
|
|
|
|
@pytest.mark.skipif(not LIBROSA, reason="librosa not installed")
|
|
class TestExtractorsWithLibrosa:
|
|
def test_capability_is_available(self):
|
|
ok, _msg = features_capability()
|
|
assert ok is True
|
|
|
|
def test_extracts_pitch_of_known_tone(self, tmp_path):
|
|
wav = tmp_path / "tone.wav"
|
|
_write_tone_wav(str(wav), hz=220.0, seconds=2.0)
|
|
track = extract_pitch(str(wav))
|
|
assert track is not None and len(track) > 0
|
|
hz_values = sorted(hz for _t, hz in track)
|
|
median = hz_values[len(hz_values) // 2]
|
|
assert median == pytest.approx(220.0, rel=0.1)
|
|
|
|
def test_extracts_energy_track(self, tmp_path):
|
|
wav = tmp_path / "tone.wav"
|
|
_write_tone_wav(str(wav), seconds=1.0)
|
|
track = extract_energy(str(wav))
|
|
assert track is not None and len(track) > 0
|
|
assert all(rms >= 0 for _t, rms in track)
|
|
assert max(rms for _t, rms in track) > 0
|