chore: atualização geral
This commit is contained in:
+31
-1
@@ -37,6 +37,36 @@ def diarization_capability(token: Optional[str]) -> Tuple[bool, str]:
|
||||
return True, "Identificação de participantes disponível."
|
||||
|
||||
|
||||
def _load_waveform(path: str) -> Optional[dict]:
|
||||
"""Decode ``path`` ourselves into the waveform dict pyannote accepts.
|
||||
|
||||
pyannote 4.x decodes audio through torchcodec, which links against a
|
||||
specific FFmpeg major version and fails outright when the installed one
|
||||
differs (``libavutil.56.dylib`` not found) — taking diarization down on
|
||||
an otherwise working machine. Handing it an already-decoded waveform
|
||||
skips that path entirely and reuses the ffmpeg extraction the acoustic
|
||||
analysis already relies on, so video containers work too.
|
||||
|
||||
Returns ``None`` when decoding is not possible, letting the caller fall
|
||||
back to passing the path and whatever pyannote can do with it.
|
||||
"""
|
||||
try:
|
||||
import soundfile
|
||||
import torch
|
||||
|
||||
from .voice_features import decodable_audio
|
||||
|
||||
with decodable_audio(path) as audio_path:
|
||||
if audio_path is None:
|
||||
return None
|
||||
data, sample_rate = soundfile.read(audio_path, dtype="float32", always_2d=True)
|
||||
# soundfile gives (samples, channels); pyannote wants (channels, samples)
|
||||
return {"waveform": torch.from_numpy(data.T), "sample_rate": int(sample_rate)}
|
||||
except Exception:
|
||||
logger.info("could not pre-decode %s for diarization", path)
|
||||
return None
|
||||
|
||||
|
||||
def diarize(
|
||||
path: str,
|
||||
token: Optional[str],
|
||||
@@ -67,7 +97,7 @@ def diarize(
|
||||
n = str(num_speakers or "").strip()
|
||||
if n.isdigit() and int(n) > 0:
|
||||
kwargs["num_speakers"] = int(n)
|
||||
result = pipe(path, **kwargs)
|
||||
result = pipe(_load_waveform(path) or path, **kwargs)
|
||||
# pyannote.audio >= 4.0 wraps the annotation; normalize to the raw one.
|
||||
if hasattr(result, "exclusive_speaker_diarization"):
|
||||
result = result.exclusive_speaker_diarization
|
||||
|
||||
Reference in New Issue
Block a user