Eram 882 linhas de seis papéis sem relação, sob um nome que só dizia
"compartilhado" — o depósito onde tudo que servia a mais de um handler
acabava caindo.
media 316 transcrição em cache, corte por fala, relatório
paths 206 sandbox, limites, caminho de saída
project 116 abrir projeto, preparar modifier/generator
captions 112 SRT, VTT, listas com timestamp
detection 99 flash frames, buracos, duplicados
formatting 86 tabelas e relatórios dos handlers
O __init__ reexporta os 46 nomes, então os treze pontos que importam daqui
não mudaram.
_transcript_cut_report saiu de formatting para media: ele precisa do hint de
instalação e do _text_result, ou seja, é relatório de transcrição e não
formatação genérica — mover foi mais honesto que cruzar imports entre os
dois módulos.
Quatro testes patchavam `server_tools._shared.transcribe`; o nome agora é
ligado por _shared/media.py, então o patch passou a apontar para lá — mesmo
padrão da experiência #23.
Lint zerado, 1454 testes passando.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
148 lines
5.9 KiB
Python
148 lines
5.9 KiB
Python
"""Tests for the analyze_voice_features MCP tool.
|
|
|
|
librosa and Whisper are monkeypatched so these run without the optional
|
|
extras, matching the pattern used by TestDetectBeatsHandler.
|
|
"""
|
|
|
|
import json
|
|
import struct
|
|
import wave
|
|
|
|
import pytest
|
|
|
|
|
|
def _write_silent_wav(path: str, seconds: float = 2.0) -> None:
|
|
n = int(44100 * seconds)
|
|
with wave.open(path, "w") as f:
|
|
f.setnchannels(1)
|
|
f.setsampwidth(2)
|
|
f.setframerate(44100)
|
|
f.writeframes(struct.pack("<%dh" % n, *([0] * n)))
|
|
|
|
|
|
_FAKE_TRANSCRIPT = {
|
|
"language": "pt",
|
|
"duration": 3.0,
|
|
"text": "isso e seguranca",
|
|
"segments": [{"text": "isso e seguranca", "start": 0.0, "end": 3.0}],
|
|
"words": [
|
|
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
|
|
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
|
|
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
|
|
],
|
|
}
|
|
|
|
|
|
@pytest.fixture
|
|
def wav(tmp_path):
|
|
path = tmp_path / "clip.wav"
|
|
_write_silent_wav(str(path))
|
|
return path
|
|
|
|
|
|
@pytest.fixture
|
|
def patched_analysis(monkeypatch):
|
|
"""Make the tool's transcription + librosa extractors deterministic."""
|
|
import server_tools._shared.media as _shared_mod
|
|
import server_tools.voice as server_mod
|
|
|
|
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
|
|
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
|
|
# "seguranca" (2.0-2.9s) is the loud, high-pitched, emphatic word
|
|
monkeypatch.setattr(
|
|
server_mod,
|
|
"extract_pitch",
|
|
lambda *a, **k: [(0.2, 120.0), (0.6, 118.0), (2.4, 260.0)],
|
|
)
|
|
monkeypatch.setattr(
|
|
server_mod,
|
|
"extract_energy",
|
|
lambda *a, **k: [(0.2, 0.10), (0.6, 0.12), (2.4, 0.95)],
|
|
)
|
|
|
|
|
|
class TestAnalyzeVoiceFeaturesHandler:
|
|
async def test_reports_when_librosa_unavailable(self, wav, monkeypatch):
|
|
import server_tools.voice as server_mod
|
|
from server import handle_analyze_voice_features
|
|
|
|
monkeypatch.setattr(
|
|
server_mod, "features_capability", lambda: (False, "componente librosa ausente.")
|
|
)
|
|
result = await handle_analyze_voice_features({"media_path": str(wav)})
|
|
assert "librosa" in result[0].text.lower()
|
|
|
|
async def test_rejects_disallowed_extension(self, tmp_path):
|
|
from server import handle_analyze_voice_features
|
|
|
|
bad = tmp_path / "clip.txt"
|
|
bad.write_text("not audio")
|
|
with pytest.raises(ValueError):
|
|
await handle_analyze_voice_features({"media_path": str(bad)})
|
|
|
|
async def test_writes_features_json_with_emphasis_per_word(self, wav, patched_analysis):
|
|
from server import handle_analyze_voice_features
|
|
|
|
result = await handle_analyze_voice_features({"media_path": str(wav)})
|
|
text = result[0].text
|
|
|
|
json_path = wav.parent / "clip_voice_features.json"
|
|
assert str(json_path) in text
|
|
data = json.loads(json_path.read_text())
|
|
assert len(data["words"]) == 3
|
|
assert all("emphasis" in w for w in data["words"])
|
|
assert all(0.0 <= w["emphasis"] <= 1.0 for w in data["words"])
|
|
|
|
async def test_loudest_word_scores_highest_emphasis(self, wav, patched_analysis):
|
|
from server import handle_analyze_voice_features
|
|
|
|
await handle_analyze_voice_features({"media_path": str(wav)})
|
|
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
|
|
by_word = {w["word"]: w["emphasis"] for w in data["words"]}
|
|
assert by_word["seguranca"] > by_word["isso"]
|
|
assert by_word["seguranca"] > by_word["e"]
|
|
|
|
async def test_persisted_config_is_embedded_in_output(self, wav, patched_analysis):
|
|
from server import handle_analyze_voice_features
|
|
|
|
await handle_analyze_voice_features({"media_path": str(wav)})
|
|
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
|
|
assert "energy_threshold" in data["config"]
|
|
assert "emphasis_weights" in data["config"]
|
|
|
|
async def test_empty_transcript_reports_instead_of_crashing(self, wav, monkeypatch):
|
|
import server_tools._shared.media as _shared_mod
|
|
import server_tools.voice as server_mod
|
|
from server import handle_analyze_voice_features
|
|
|
|
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
|
|
monkeypatch.setattr(
|
|
_shared_mod, "transcribe", lambda *a, **k: {**_FAKE_TRANSCRIPT, "words": []}
|
|
)
|
|
result = await handle_analyze_voice_features({"media_path": str(wav)})
|
|
assert "no words" in result[0].text.lower()
|
|
|
|
async def test_untranscribable_media_reports_install_hint(self, wav, monkeypatch):
|
|
import server_tools._shared.media as _shared_mod
|
|
import server_tools.voice as server_mod
|
|
from server import handle_analyze_voice_features
|
|
|
|
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
|
|
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)
|
|
result = await handle_analyze_voice_features({"media_path": str(wav)})
|
|
assert "faster-whisper" in result[0].text
|
|
|
|
async def test_missing_pitch_track_degrades_without_crashing(self, wav, monkeypatch):
|
|
import server_tools._shared.media as _shared_mod
|
|
import server_tools.voice as server_mod
|
|
from server import handle_analyze_voice_features
|
|
|
|
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _FAKE_TRANSCRIPT)
|
|
monkeypatch.setattr(server_mod, "features_capability", lambda: (True, "ok"))
|
|
monkeypatch.setattr(server_mod, "extract_pitch", lambda *a, **k: None)
|
|
monkeypatch.setattr(server_mod, "extract_energy", lambda *a, **k: None)
|
|
result = await handle_analyze_voice_features({"media_path": str(wav)})
|
|
assert "Voice Feature Analysis" in result[0].text
|
|
data = json.loads((wav.parent / "clip_voice_features.json").read_text())
|
|
assert all(w["emphasis"] >= 0.0 for w in data["words"])
|