Files
gart/code/tests/test_voice_features_tool.py
T

148 lines
5.9 KiB
Python

"""Tests for the analyze_voice_features MCP tool.
librosa and Whisper are monkeypatched so these run without the optional
extras, matching the pattern used by TestDetectBeatsHandler.
"""
import json
import struct
import wave
import pytest
def _write_silent_wav(path: str, seconds: float = 2.0) -> None:
n = int(44100 * seconds)
with wave.open(path, "w") as f:
f.setnchannels(1)
f.setsampwidth(2)
f.setframerate(44100)
f.writeframes(struct.pack("<%dh" % n, *([0] * n)))
_FAKE_TRANSCRIPT = {
"language": "pt",
"duration": 3.0,
"text": "isso e seguranca",
"segments": [{"text": "isso e seguranca", "start": 0.0, "end": 3.0}],
"words": [
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
],
}
@pytest.fixture
def wav(tmp_path):
path = tmp_path / "clip.wav"
_write_silent_wav(str(path))
return path
@pytest.fixture
def patched_analysis(monkeypatch):
"""Make the tool's transcription + librosa extractors deterministic."""
import server_tools._shared as _shared_mod
import server_tools.voice as server_mod
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
# "seguranca" (2.0-2.9s) is the loud, high-pitched, emphatic word
monkeypatch.setattr(
server_mod,
"extract_pitch",
lambda *a, **k: [(0.2, 120.0), (0.6, 118.0), (2.4, 260.0)],
)
monkeypatch.setattr(
server_mod,
"extract_energy",
lambda *a, **k: [(0.2, 0.10), (0.6, 0.12), (2.4, 0.95)],
)
class TestAnalyzeVoiceFeaturesHandler:
async def test_reports_when_librosa_unavailable(self, wav, monkeypatch):
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(
server_mod, "features_capability", lambda: (False, "componente librosa ausente.")
)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "librosa" in result[0].text.lower()
async def test_rejects_disallowed_extension(self, tmp_path):
from server import handle_analyze_voice_features
bad = tmp_path / "clip.txt"
bad.write_text("not audio")
with pytest.raises(ValueError):
await handle_analyze_voice_features({"media_path": str(bad)})
async def test_writes_features_json_with_emphasis_per_word(self, wav, patched_analysis):
from server import handle_analyze_voice_features
result = await handle_analyze_voice_features({"media_path": str(wav)})
text = result[0].text
json_path = wav.parent / "clip_voice_features.json"
assert str(json_path) in text
data = json.loads(json_path.read_text())
assert len(data["words"]) == 3
assert all("emphasis" in w for w in data["words"])
assert all(0.0 <= w["emphasis"] <= 1.0 for w in data["words"])
async def test_loudest_word_scores_highest_emphasis(self, wav, patched_analysis):
from server import handle_analyze_voice_features
await handle_analyze_voice_features({"media_path": str(wav)})
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
by_word = {w["word"]: w["emphasis"] for w in data["words"]}
assert by_word["seguranca"] > by_word["isso"]
assert by_word["seguranca"] > by_word["e"]
async def test_persisted_config_is_embedded_in_output(self, wav, patched_analysis):
from server import handle_analyze_voice_features
await handle_analyze_voice_features({"media_path": str(wav)})
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
assert "energy_threshold" in data["config"]
assert "emphasis_weights" in data["config"]
async def test_empty_transcript_reports_instead_of_crashing(self, wav, monkeypatch):
import server_tools._shared as _shared_mod
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
monkeypatch.setattr(
_shared_mod, "transcribe", lambda *a, **k: {**_FAKE_TRANSCRIPT, "words": []}
)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "no words" in result[0].text.lower()
async def test_untranscribable_media_reports_install_hint(self, wav, monkeypatch):
import server_tools._shared as _shared_mod
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "faster-whisper" in result[0].text
async def test_missing_pitch_track_degrades_without_crashing(self, wav, monkeypatch):
import server_tools._shared as _shared_mod
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
monkeypatch.setattr(server_mod, "extract_pitch", lambda *a, **k: None)
monkeypatch.setattr(server_mod, "extract_energy", lambda *a, **k: None)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "Voice Feature Analysis" in result[0].text
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
assert all(w["emphasis"] >= 0.0 for w in data["words"])