Eram 882 linhas de seis papéis sem relação, sob um nome que só dizia
"compartilhado" — o depósito onde tudo que servia a mais de um handler
acabava caindo.
media 316 transcrição em cache, corte por fala, relatório
paths 206 sandbox, limites, caminho de saída
project 116 abrir projeto, preparar modifier/generator
captions 112 SRT, VTT, listas com timestamp
detection 99 flash frames, buracos, duplicados
formatting 86 tabelas e relatórios dos handlers
O __init__ reexporta os 46 nomes, então os treze pontos que importam daqui
não mudaram.
_transcript_cut_report saiu de formatting para media: ele precisa do hint de
instalação e do _text_result, ou seja, é relatório de transcrição e não
formatação genérica — mover foi mais honesto que cruzar imports entre os
dois módulos.
Quatro testes patchavam `server_tools._shared.transcribe`; o nome agora é
ligado por _shared/media.py, então o patch passou a apontar para lá — mesmo
padrão da experiência #23.
Lint zerado, 1454 testes passando.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
112 lines
4.2 KiB
Python
112 lines
4.2 KiB
Python
"""Tests for the build_voice_timeline MCP tool."""
|
|
|
|
import json
|
|
|
|
import pytest
|
|
|
|
from tests.test_voice_features_tool import _write_silent_wav
|
|
|
|
_TRANSCRIPT = {
|
|
"language": "pt",
|
|
"duration": 4.0,
|
|
"text": "isso e seguranca total",
|
|
"segments": [
|
|
{"text": "isso e", "start": 0.0, "end": 1.0},
|
|
{"text": "seguranca total", "start": 2.0, "end": 4.0},
|
|
],
|
|
"words": [
|
|
{"word": "isso", "start": 0.0, "end": 0.4, "confidence": 0.9},
|
|
{"word": "e", "start": 0.5, "end": 0.7, "confidence": 0.9},
|
|
{"word": "seguranca", "start": 2.0, "end": 2.9, "confidence": 0.9},
|
|
{"word": "total", "start": 3.0, "end": 3.5, "confidence": 0.9},
|
|
],
|
|
}
|
|
|
|
|
|
@pytest.fixture
|
|
def wav(tmp_path):
|
|
path = tmp_path / "clip.wav"
|
|
_write_silent_wav(str(path))
|
|
return path
|
|
|
|
|
|
@pytest.fixture
|
|
def patched(monkeypatch):
|
|
"""Deterministic transcription + acoustics, no optional extras needed."""
|
|
import fcpxml.voice_timeline as vt
|
|
import server_tools._shared.media as _shared_mod
|
|
|
|
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: _TRANSCRIPT)
|
|
monkeypatch.setattr(vt, "extract_pitch", lambda *a, **k: [(2.4, 260.0), (0.2, 120.0)])
|
|
monkeypatch.setattr(vt, "extract_energy", lambda *a, **k: [(2.4, 0.95), (0.2, 0.10)])
|
|
|
|
|
|
class TestBuildVoiceTimelineHandler:
|
|
async def test_writes_timeline_json(self, wav, patched):
|
|
from server import handle_build_voice_timeline
|
|
|
|
result = await handle_build_voice_timeline({"media_path": str(wav)})
|
|
text = result[0].text
|
|
|
|
json_path = wav.parent / "clip_voice_timeline.json"
|
|
assert str(json_path) in text
|
|
data = json.loads(json_path.read_text(encoding="utf-8"))
|
|
assert data["summary"]["word_count"] == 4
|
|
assert len(data["segments"]) == 2
|
|
|
|
async def test_reports_which_layers_ran(self, wav, patched):
|
|
from server import handle_build_voice_timeline
|
|
|
|
result = await handle_build_voice_timeline({"media_path": str(wav)})
|
|
text = result[0].text
|
|
assert "Analysis Layers" in text
|
|
assert "Transcript" in text
|
|
|
|
async def test_rejects_disallowed_extension(self, tmp_path):
|
|
from server import handle_build_voice_timeline
|
|
|
|
bad = tmp_path / "clip.txt"
|
|
bad.write_text("not audio")
|
|
with pytest.raises(ValueError):
|
|
await handle_build_voice_timeline({"media_path": str(bad)})
|
|
|
|
async def test_untranscribable_media_reports_hint(self, wav, monkeypatch):
|
|
import server_tools._shared.media as _shared_mod
|
|
from server import handle_build_voice_timeline
|
|
|
|
monkeypatch.setattr(_shared_mod, "transcribe", lambda *a, **k: None)
|
|
result = await handle_build_voice_timeline({"media_path": str(wav)})
|
|
assert "faster-whisper" in result[0].text
|
|
|
|
async def test_uses_persisted_peak_settings(self, wav, patched, monkeypatch):
|
|
"""A wider percentile must surface more peak moments."""
|
|
import server_tools.voice as server_mod
|
|
from server import handle_build_voice_timeline
|
|
|
|
def config(percentile):
|
|
return {
|
|
"energy_threshold": 0.5,
|
|
"peak_percentile": percentile,
|
|
"emphasis_floor": 0.0,
|
|
"emphasis_weights": {
|
|
"energy": 0.30, "pitch_variation": 0.25, "rate_variation": 0.20,
|
|
"pause_before": 0.15, "duration": 0.10,
|
|
},
|
|
"emotion_enabled": False,
|
|
"emotion_sensitivity": 0.5,
|
|
"zoom_scale": 1.3,
|
|
"zoom_mode": "in_out",
|
|
"zoom_ease_in": 0.25,
|
|
"zoom_ease_out": 0.04,
|
|
}
|
|
|
|
monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(0.01))
|
|
await handle_build_voice_timeline({"media_path": str(wav)})
|
|
strict = json.loads((wav.parent / "clip_voice_timeline.json").read_text())
|
|
|
|
monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(1.0))
|
|
await handle_build_voice_timeline({"media_path": str(wav)})
|
|
loose = json.loads((wav.parent / "clip_voice_timeline.json").read_text())
|
|
|
|
assert loose["summary"]["peak_count"] > strict["summary"]["peak_count"]
|