87 lines
2.6 KiB
Python
87 lines
2.6 KiB
Python
"""Tests for fcpxml/diarize.py — speaker diarization (WHISPERX-inspired).
|
|
|
|
The pure assignment helpers are fully testable without pyannote; the
|
|
diarize()/diarization_capability() integration path degrades gracefully when
|
|
pyannote or a token is absent.
|
|
"""
|
|
|
|
|
|
from fcpxml.diarize import (
|
|
DEFAULT_SPEAKER,
|
|
assign_speakers,
|
|
build_speakers,
|
|
diarization_capability,
|
|
diarize,
|
|
)
|
|
|
|
|
|
def test_capability_without_token_is_false():
|
|
ok, _msg = diarization_capability("")
|
|
assert ok is False
|
|
|
|
|
|
def test_capability_without_pyannote_is_false(monkeypatch):
|
|
import builtins
|
|
|
|
real_import = builtins.__import__
|
|
|
|
def block(name, *args, **kwargs):
|
|
if name.startswith("pyannote"):
|
|
raise ImportError("blocked for test")
|
|
return real_import(name, *args, **kwargs)
|
|
|
|
monkeypatch.setattr(builtins, "__import__", block)
|
|
ok, msg = diarization_capability("hf_xyz")
|
|
assert ok is False
|
|
assert "pyannote" in msg.lower()
|
|
|
|
|
|
def test_diarize_without_token_returns_none():
|
|
assert diarize("ignored.wav", None) is None
|
|
assert diarize("ignored.wav", "") is None
|
|
|
|
|
|
def test_assign_speakers_no_tracks_uses_default():
|
|
segments = [{"start": 0.0, "end": 1.0}, {"start": 2.0, "end": 3.0}]
|
|
words = [{"start": 0.2, "end": 0.4}]
|
|
segs, ws = assign_speakers(segments, words, None)
|
|
assert all(s["speaker_id"] == DEFAULT_SPEAKER for s in segs)
|
|
assert ws[0]["speaker_id"] == DEFAULT_SPEAKER
|
|
|
|
|
|
def test_assign_speakers_overlap_mapping():
|
|
tracks = [(0.0, 2.0, "SPEAKER_A"), (2.0, 4.0, "SPEAKER_B")]
|
|
segments = [{"start": 0.5, "end": 1.5}, {"start": 2.5, "end": 3.5}]
|
|
words = [{"start": 0.9, "end": 1.0}, {"start": 3.0, "end": 3.1}]
|
|
segs, ws = assign_speakers(segments, words, tracks)
|
|
assert segs[0]["speaker_id"] == "SPEAKER_00"
|
|
assert segs[1]["speaker_id"] == "SPEAKER_01"
|
|
assert ws[0]["speaker_id"] == "SPEAKER_00"
|
|
assert ws[1]["speaker_id"] == "SPEAKER_01"
|
|
|
|
|
|
def test_assign_speakers_does_not_mutate_input():
|
|
tracks = [(0.0, 2.0, "SPEAKER_A")]
|
|
segments = [{"start": 0.5, "end": 1.5}]
|
|
words = [{"start": 0.9, "end": 1.0}]
|
|
assign_speakers(segments, words, tracks)
|
|
assert "speaker_id" not in segments[0]
|
|
assert "speaker_id" not in words[0]
|
|
|
|
|
|
def test_build_speakers_dedupes_and_orders():
|
|
segments = [
|
|
{"speaker_id": "SPEAKER_01"},
|
|
{"speaker_id": "SPEAKER_00"},
|
|
{"speaker_id": "SPEAKER_01"},
|
|
]
|
|
speakers = build_speakers(segments)
|
|
assert speakers == [
|
|
{"id": "SPEAKER_01", "name": "Speaker 1"},
|
|
{"id": "SPEAKER_00", "name": "Speaker 2"},
|
|
]
|
|
|
|
|
|
def test_build_speakers_empty():
|
|
assert build_speakers([]) == []
|