chore: atualização geral
This commit is contained in:
@@ -0,0 +1,220 @@
|
||||
"""Acoustic features for voice analysis — pitch, energy, rate, pauses.
|
||||
|
||||
Mirrors the ``media_intel.py`` contract: librosa is an optional dependency
|
||||
(``pip install 'fcp-mcp-server[intelligence]'``, already required by beat
|
||||
detection), imported lazily, and every extractor degrades to ``None`` when
|
||||
the library is missing or the file cannot be analyzed — never crashes.
|
||||
|
||||
``compute_speech_rate``/``compute_pauses`` are pure functions over
|
||||
word-timestamp dicts (the shape ``transcribe.py`` already produces) and need
|
||||
no audio file at all.
|
||||
"""
|
||||
|
||||
import contextlib
|
||||
import logging
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Iterator, List, Optional, Sequence, Tuple
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Human voice fundamental frequency range (covers low male to high female/child).
|
||||
PITCH_FMIN_HZ = 65.0
|
||||
PITCH_FMAX_HZ = 1000.0
|
||||
|
||||
# Formats librosa reads directly through soundfile. Anything else — notably
|
||||
# the .mov/.mp4 that source footage actually arrives in — must be decoded by
|
||||
# ffmpeg first, or analysis fails outright.
|
||||
NATIVE_AUDIO_SUFFIXES = {".wav", ".aif", ".aiff", ".flac"}
|
||||
|
||||
# Voice analysis only needs the speech band: 16 kHz mono is well above the
|
||||
# Nyquist limit for our 1 kHz pitch ceiling, and keeps the extracted file
|
||||
# small and fast to decode even for hour-long footage.
|
||||
EXTRACT_SAMPLE_RATE = 16000
|
||||
EXTRACT_TIMEOUT_SECONDS = 600
|
||||
|
||||
|
||||
@contextlib.contextmanager
|
||||
def decodable_audio(path: str) -> Iterator[Optional[str]]:
|
||||
"""Yield a path librosa can read, extracting the audio track if needed.
|
||||
|
||||
Audio files pass straight through. Video containers are decoded to a
|
||||
temporary mono WAV with ffmpeg and cleaned up on exit. Yields ``None``
|
||||
when the audio cannot be obtained (no ffmpeg, no audio track, failure),
|
||||
keeping the graceful-degradation contract of this module.
|
||||
"""
|
||||
file_path = Path(path)
|
||||
if file_path.suffix.lower() in NATIVE_AUDIO_SUFFIXES:
|
||||
yield str(file_path)
|
||||
return
|
||||
|
||||
if shutil.which("ffmpeg") is None:
|
||||
logger.info("ffmpeg not found on PATH; cannot extract audio from %s", file_path)
|
||||
yield None
|
||||
return
|
||||
|
||||
tmp_dir = tempfile.mkdtemp(prefix="fcp_voice_")
|
||||
wav_path = Path(tmp_dir) / "audio.wav"
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[
|
||||
"ffmpeg", "-hide_banner", "-nostdin", "-y",
|
||||
"-i", str(file_path),
|
||||
"-vn", # audio only: decoding video would dominate the runtime
|
||||
"-ac", "1",
|
||||
"-ar", str(EXTRACT_SAMPLE_RATE),
|
||||
str(wav_path),
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=EXTRACT_TIMEOUT_SECONDS,
|
||||
)
|
||||
if result.returncode != 0 or not wav_path.is_file():
|
||||
logger.warning("ffmpeg could not extract audio from %s", file_path)
|
||||
yield None
|
||||
else:
|
||||
yield str(wav_path)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
logger.warning("audio extraction failed for %s", file_path)
|
||||
yield None
|
||||
finally:
|
||||
shutil.rmtree(tmp_dir, ignore_errors=True)
|
||||
|
||||
|
||||
def features_capability() -> Tuple[bool, str]:
|
||||
"""Whether pitch/energy extraction is available (librosa installed)."""
|
||||
try:
|
||||
import librosa # noqa: F401
|
||||
except Exception:
|
||||
return False, "Análise acústica indisponível: componente librosa ausente."
|
||||
return True, "Análise acústica disponível."
|
||||
|
||||
|
||||
def extract_pitch(
|
||||
path: str, hop_length: int = 512, max_analysis_seconds: float = 1200.0
|
||||
) -> Optional[List[Tuple[float, float]]]:
|
||||
"""Frame-level pitch (F0) track via librosa's ``pyin``.
|
||||
|
||||
Returns ``[(time_seconds, hz), ...]`` for voiced frames only (unvoiced
|
||||
frames, where ``pyin`` reports no pitch, are dropped), or ``None`` when
|
||||
librosa is unavailable or the file cannot be analyzed.
|
||||
"""
|
||||
file_path = Path(path)
|
||||
if not file_path.is_file():
|
||||
return None
|
||||
try:
|
||||
import librosa
|
||||
except ImportError:
|
||||
logger.info("librosa not installed; pitch extraction unavailable")
|
||||
return None
|
||||
try:
|
||||
with decodable_audio(str(file_path)) as audio_path:
|
||||
if audio_path is None:
|
||||
return None
|
||||
y, sr = librosa.load(audio_path, sr=None, mono=True, duration=max_analysis_seconds)
|
||||
f0, voiced_flag, _voiced_prob = librosa.pyin(
|
||||
y, fmin=PITCH_FMIN_HZ, fmax=PITCH_FMAX_HZ, sr=sr, hop_length=hop_length
|
||||
)
|
||||
times = librosa.times_like(f0, sr=sr, hop_length=hop_length)
|
||||
except Exception:
|
||||
logger.warning("librosa pitch analysis failed for %s", file_path)
|
||||
return None
|
||||
return [
|
||||
(float(t), float(hz))
|
||||
for t, hz, voiced in zip(times, f0, voiced_flag)
|
||||
if voiced and hz == hz # ``hz == hz`` filters NaN without importing math/numpy here
|
||||
]
|
||||
|
||||
|
||||
def extract_energy(
|
||||
path: str, hop_length: int = 512, max_analysis_seconds: float = 1200.0
|
||||
) -> Optional[List[Tuple[float, float]]]:
|
||||
"""Frame-level RMS energy track via librosa.
|
||||
|
||||
Returns ``[(time_seconds, rms), ...]``, or ``None`` when librosa is
|
||||
unavailable or the file cannot be analyzed.
|
||||
"""
|
||||
file_path = Path(path)
|
||||
if not file_path.is_file():
|
||||
return None
|
||||
try:
|
||||
import librosa
|
||||
except ImportError:
|
||||
logger.info("librosa not installed; energy extraction unavailable")
|
||||
return None
|
||||
try:
|
||||
with decodable_audio(str(file_path)) as audio_path:
|
||||
if audio_path is None:
|
||||
return None
|
||||
y, sr = librosa.load(audio_path, sr=None, mono=True, duration=max_analysis_seconds)
|
||||
rms = librosa.feature.rms(y=y, hop_length=hop_length)[0]
|
||||
times = librosa.times_like(rms, sr=sr, hop_length=hop_length)
|
||||
except Exception:
|
||||
logger.warning("librosa energy analysis failed for %s", file_path)
|
||||
return None
|
||||
return [(float(t), float(r)) for t, r in zip(times, rms)]
|
||||
|
||||
|
||||
def _window_average(track: Sequence[Tuple[float, float]], start: float, end: float) -> Optional[float]:
|
||||
"""Average of ``track`` values whose timestamp falls in ``[start, end]``."""
|
||||
values = [v for t, v in track if start <= t <= end]
|
||||
if not values:
|
||||
return None
|
||||
return sum(values) / len(values)
|
||||
|
||||
|
||||
def word_pitch_energy(
|
||||
words: Sequence[dict],
|
||||
pitch_track: Optional[Sequence[Tuple[float, float]]],
|
||||
energy_track: Optional[Sequence[Tuple[float, float]]],
|
||||
) -> List[dict]:
|
||||
"""Attach average pitch/energy over each word's ``[start, end]`` span.
|
||||
|
||||
Words carry ``pitch_hz``/``energy`` (``None`` when the span has no
|
||||
voiced frames or a track is unavailable). Both tracks are the output of
|
||||
:func:`extract_pitch`/:func:`extract_energy`.
|
||||
"""
|
||||
out: List[dict] = []
|
||||
for w in words:
|
||||
ww = dict(w)
|
||||
start = float(w.get("start", 0.0))
|
||||
end = float(w.get("end", start))
|
||||
ww["pitch_hz"] = _window_average(pitch_track, start, end) if pitch_track else None
|
||||
ww["energy"] = _window_average(energy_track, start, end) if energy_track else None
|
||||
out.append(ww)
|
||||
return out
|
||||
|
||||
|
||||
def compute_speech_rate(words: Sequence[dict], window_seconds: float = 3.0) -> List[float]:
|
||||
"""Local speech rate (words/second) around each word.
|
||||
|
||||
For word *i*, counts every word whose start falls within
|
||||
``[start_i - window_seconds, start_i]`` and divides by
|
||||
``window_seconds`` — a trailing local rate, cheap to compute and stable
|
||||
against a single long/short word skewing the whole utterance's average.
|
||||
"""
|
||||
starts = [float(w.get("start", 0.0)) for w in words]
|
||||
rates: List[float] = []
|
||||
for i, s in enumerate(starts):
|
||||
lo = s - window_seconds
|
||||
count = sum(1 for t in starts[: i + 1] if t >= lo)
|
||||
rates.append(count / window_seconds if window_seconds > 0 else 0.0)
|
||||
return rates
|
||||
|
||||
|
||||
def compute_pauses(words: Sequence[dict]) -> List[float]:
|
||||
"""Silence (seconds) immediately before each word.
|
||||
|
||||
The first word's "pause before" is the time from the start of the audio
|
||||
to its own start; every other word measures the gap since the previous
|
||||
word's end (clamped to ``0`` for overlapping/adjacent words).
|
||||
"""
|
||||
pauses: List[float] = []
|
||||
prev_end = 0.0
|
||||
for w in words:
|
||||
start = float(w.get("start", 0.0))
|
||||
pauses.append(max(0.0, start - prev_end))
|
||||
prev_end = float(w.get("end", start))
|
||||
return pauses
|
||||
Reference in New Issue
Block a user