chore: atualização geral

This commit is contained in:
João Henrique
2026-08-19 16:35:29 -04:00
parent 8fca456ceb
commit e7748c2c58
66 changed files with 13037 additions and 4237 deletions
+31 -1
View File
@@ -37,6 +37,36 @@ def diarization_capability(token: Optional[str]) -> Tuple[bool, str]:
return True, "Identificação de participantes disponível."
def _load_waveform(path: str) -> Optional[dict]:
"""Decode ``path`` ourselves into the waveform dict pyannote accepts.
pyannote 4.x decodes audio through torchcodec, which links against a
specific FFmpeg major version and fails outright when the installed one
differs (``libavutil.56.dylib`` not found) — taking diarization down on
an otherwise working machine. Handing it an already-decoded waveform
skips that path entirely and reuses the ffmpeg extraction the acoustic
analysis already relies on, so video containers work too.
Returns ``None`` when decoding is not possible, letting the caller fall
back to passing the path and whatever pyannote can do with it.
"""
try:
import soundfile
import torch
from .voice_features import decodable_audio
with decodable_audio(path) as audio_path:
if audio_path is None:
return None
data, sample_rate = soundfile.read(audio_path, dtype="float32", always_2d=True)
# soundfile gives (samples, channels); pyannote wants (channels, samples)
return {"waveform": torch.from_numpy(data.T), "sample_rate": int(sample_rate)}
except Exception:
logger.info("could not pre-decode %s for diarization", path)
return None
def diarize(
path: str,
token: Optional[str],
@@ -67,7 +97,7 @@ def diarize(
n = str(num_speakers or "").strip()
if n.isdigit() and int(n) > 0:
kwargs["num_speakers"] = int(n)
result = pipe(path, **kwargs)
result = pipe(_load_waveform(path) or path, **kwargs)
# pyannote.audio >= 4.0 wraps the annotation; normalize to the raw one.
if hasattr(result, "exclusive_speaker_diarization"):
result = result.exclusive_speaker_diarization