Files
gart/code/tests/test_voice_features_tool.py
T
João HenriqueandClaude Opus 5 ffaebb3f72 refactor: _shared.py vira subpacote, um módulo por papel
Eram 882 linhas de seis papéis sem relação, sob um nome que só dizia
"compartilhado" — o depósito onde tudo que servia a mais de um handler
acabava caindo.

    media       316   transcrição em cache, corte por fala, relatório
    paths       206   sandbox, limites, caminho de saída
    project     116   abrir projeto, preparar modifier/generator
    captions    112   SRT, VTT, listas com timestamp
    detection    99   flash frames, buracos, duplicados
    formatting   86   tabelas e relatórios dos handlers

O __init__ reexporta os 46 nomes, então os treze pontos que importam daqui
não mudaram.

_transcript_cut_report saiu de formatting para media: ele precisa do hint de
instalação e do _text_result, ou seja, é relatório de transcrição e não
formatação genérica — mover foi mais honesto que cruzar imports entre os
dois módulos.

Quatro testes patchavam `server_tools._shared.transcribe`; o nome agora é
ligado por _shared/media.py, então o patch passou a apontar para lá — mesmo
padrão da experiência #23.

Lint zerado, 1454 testes passando.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-19 22:35:02 -04:00

148 lines
5.9 KiB
Python

"""Tests for the analyze_voice_features MCP tool.
librosa and Whisper are monkeypatched so these run without the optional
extras, matching the pattern used by TestDetectBeatsHandler.
"""
import json
import struct
import wave
import pytest
def _write_silent_wav(path: str, seconds: float = 2.0) -> None:
n = int(44100 * seconds)
with wave.open(path, "w") as f:
f.setnchannels(1)
f.setsampwidth(2)
f.setframerate(44100)
f.writeframes(struct.pack("<%dh" % n, *([0] * n)))
_FAKE_TRANSCRIPT = {
"language": "pt",
"duration": 3.0,
"text": "isso e seguranca",
"segments": [{"text": "isso e seguranca", "start": 0.0, "end": 3.0}],
"words": [
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
],
}
@pytest.fixture
def wav(tmp_path):
path = tmp_path / "clip.wav"
_write_silent_wav(str(path))
return path
@pytest.fixture
def patched_analysis(monkeypatch):
"""Make the tool's transcription + librosa extractors deterministic."""
import server_tools._shared.media as _shared_mod
import server_tools.voice as server_mod
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
# "seguranca" (2.0-2.9s) is the loud, high-pitched, emphatic word
monkeypatch.setattr(
server_mod,
"extract_pitch",
lambda *a, **k: [(0.2, 120.0), (0.6, 118.0), (2.4, 260.0)],
)
monkeypatch.setattr(
server_mod,
"extract_energy",
lambda *a, **k: [(0.2, 0.10), (0.6, 0.12), (2.4, 0.95)],
)
class TestAnalyzeVoiceFeaturesHandler:
async def test_reports_when_librosa_unavailable(self, wav, monkeypatch):
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(
server_mod, "features_capability", lambda: (False, "componente librosa ausente.")
)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "librosa" in result[0].text.lower()
async def test_rejects_disallowed_extension(self, tmp_path):
from server import handle_analyze_voice_features
bad = tmp_path / "clip.txt"
bad.write_text("not audio")
with pytest.raises(ValueError):
await handle_analyze_voice_features({"media_path": str(bad)})
async def test_writes_features_json_with_emphasis_per_word(self, wav, patched_analysis):
from server import handle_analyze_voice_features
result = await handle_analyze_voice_features({"media_path": str(wav)})
text = result[0].text
json_path = wav.parent / "clip_voice_features.json"
assert str(json_path) in text
data = json.loads(json_path.read_text())
assert len(data["words"]) == 3
assert all("emphasis" in w for w in data["words"])
assert all(0.0 <= w["emphasis"] <= 1.0 for w in data["words"])
async def test_loudest_word_scores_highest_emphasis(self, wav, patched_analysis):
from server import handle_analyze_voice_features
await handle_analyze_voice_features({"media_path": str(wav)})
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
by_word = {w["word"]: w["emphasis"] for w in data["words"]}
assert by_word["seguranca"] > by_word["isso"]
assert by_word["seguranca"] > by_word["e"]
async def test_persisted_config_is_embedded_in_output(self, wav, patched_analysis):
from server import handle_analyze_voice_features
await handle_analyze_voice_features({"media_path": str(wav)})
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
assert "energy_threshold" in data["config"]
assert "emphasis_weights" in data["config"]
async def test_empty_transcript_reports_instead_of_crashing(self, wav, monkeypatch):
import server_tools._shared.media as _shared_mod
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
monkeypatch.setattr(
_shared_mod, "transcribe", lambda *a, **k: {**_FAKE_TRANSCRIPT, "words": []}
)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "no words" in result[0].text.lower()
async def test_untranscribable_media_reports_install_hint(self, wav, monkeypatch):
import server_tools._shared.media as _shared_mod
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "faster-whisper" in result[0].text
async def test_missing_pitch_track_degrades_without_crashing(self, wav, monkeypatch):
import server_tools._shared.media as _shared_mod
import server_tools.voice as server_mod
from server import handle_analyze_voice_features
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
monkeypatch.setattr(server_mod, "extract_pitch", lambda *a, **k: None)
monkeypatch.setattr(server_mod, "extract_energy", lambda *a, **k: None)
result = await handle_analyze_voice_features({"media_path": str(wav)})
assert "Voice Feature Analysis" in result[0].text
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
assert all(w["emphasis"] >= 0.0 for w in data["words"])