221 lines
8.2 KiB
Python
221 lines
8.2 KiB
Python
"""Acoustic features for voice analysis — pitch, energy, rate, pauses.
|
|
|
|
Mirrors the ``media_intel.py`` contract: librosa is an optional dependency
|
|
(``pip install 'fcp-mcp-server[intelligence]'``, already required by beat
|
|
detection), imported lazily, and every extractor degrades to ``None`` when
|
|
the library is missing or the file cannot be analyzed — never crashes.
|
|
|
|
``compute_speech_rate``/``compute_pauses`` are pure functions over
|
|
word-timestamp dicts (the shape ``transcribe.py`` already produces) and need
|
|
no audio file at all.
|
|
"""
|
|
|
|
import contextlib
|
|
import logging
|
|
import shutil
|
|
import subprocess
|
|
import tempfile
|
|
from pathlib import Path
|
|
from typing import Iterator, List, Optional, Sequence, Tuple
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Human voice fundamental frequency range (covers low male to high female/child).
|
|
PITCH_FMIN_HZ = 65.0
|
|
PITCH_FMAX_HZ = 1000.0
|
|
|
|
# Formats librosa reads directly through soundfile. Anything else — notably
|
|
# the .mov/.mp4 that source footage actually arrives in — must be decoded by
|
|
# ffmpeg first, or analysis fails outright.
|
|
NATIVE_AUDIO_SUFFIXES = {".wav", ".aif", ".aiff", ".flac"}
|
|
|
|
# Voice analysis only needs the speech band: 16 kHz mono is well above the
|
|
# Nyquist limit for our 1 kHz pitch ceiling, and keeps the extracted file
|
|
# small and fast to decode even for hour-long footage.
|
|
EXTRACT_SAMPLE_RATE = 16000
|
|
EXTRACT_TIMEOUT_SECONDS = 600
|
|
|
|
|
|
@contextlib.contextmanager
|
|
def decodable_audio(path: str) -> Iterator[Optional[str]]:
|
|
"""Yield a path librosa can read, extracting the audio track if needed.
|
|
|
|
Audio files pass straight through. Video containers are decoded to a
|
|
temporary mono WAV with ffmpeg and cleaned up on exit. Yields ``None``
|
|
when the audio cannot be obtained (no ffmpeg, no audio track, failure),
|
|
keeping the graceful-degradation contract of this module.
|
|
"""
|
|
file_path = Path(path)
|
|
if file_path.suffix.lower() in NATIVE_AUDIO_SUFFIXES:
|
|
yield str(file_path)
|
|
return
|
|
|
|
if shutil.which("ffmpeg") is None:
|
|
logger.info("ffmpeg not found on PATH; cannot extract audio from %s", file_path)
|
|
yield None
|
|
return
|
|
|
|
tmp_dir = tempfile.mkdtemp(prefix="fcp_voice_")
|
|
wav_path = Path(tmp_dir) / "audio.wav"
|
|
try:
|
|
result = subprocess.run(
|
|
[
|
|
"ffmpeg", "-hide_banner", "-nostdin", "-y",
|
|
"-i", str(file_path),
|
|
"-vn", # audio only: decoding video would dominate the runtime
|
|
"-ac", "1",
|
|
"-ar", str(EXTRACT_SAMPLE_RATE),
|
|
str(wav_path),
|
|
],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=EXTRACT_TIMEOUT_SECONDS,
|
|
)
|
|
if result.returncode != 0 or not wav_path.is_file():
|
|
logger.warning("ffmpeg could not extract audio from %s", file_path)
|
|
yield None
|
|
else:
|
|
yield str(wav_path)
|
|
except (OSError, subprocess.TimeoutExpired):
|
|
logger.warning("audio extraction failed for %s", file_path)
|
|
yield None
|
|
finally:
|
|
shutil.rmtree(tmp_dir, ignore_errors=True)
|
|
|
|
|
|
def features_capability() -> Tuple[bool, str]:
|
|
"""Whether pitch/energy extraction is available (librosa installed)."""
|
|
try:
|
|
import librosa # noqa: F401
|
|
except Exception:
|
|
return False, "Análise acústica indisponível: componente librosa ausente."
|
|
return True, "Análise acústica disponível."
|
|
|
|
|
|
def extract_pitch(
|
|
path: str, hop_length: int = 512, max_analysis_seconds: float = 1200.0
|
|
) -> Optional[List[Tuple[float, float]]]:
|
|
"""Frame-level pitch (F0) track via librosa's ``pyin``.
|
|
|
|
Returns ``[(time_seconds, hz), ...]`` for voiced frames only (unvoiced
|
|
frames, where ``pyin`` reports no pitch, are dropped), or ``None`` when
|
|
librosa is unavailable or the file cannot be analyzed.
|
|
"""
|
|
file_path = Path(path)
|
|
if not file_path.is_file():
|
|
return None
|
|
try:
|
|
import librosa
|
|
except ImportError:
|
|
logger.info("librosa not installed; pitch extraction unavailable")
|
|
return None
|
|
try:
|
|
with decodable_audio(str(file_path)) as audio_path:
|
|
if audio_path is None:
|
|
return None
|
|
y, sr = librosa.load(audio_path, sr=None, mono=True, duration=max_analysis_seconds)
|
|
f0, voiced_flag, _voiced_prob = librosa.pyin(
|
|
y, fmin=PITCH_FMIN_HZ, fmax=PITCH_FMAX_HZ, sr=sr, hop_length=hop_length
|
|
)
|
|
times = librosa.times_like(f0, sr=sr, hop_length=hop_length)
|
|
except Exception:
|
|
logger.warning("librosa pitch analysis failed for %s", file_path)
|
|
return None
|
|
return [
|
|
(float(t), float(hz))
|
|
for t, hz, voiced in zip(times, f0, voiced_flag)
|
|
if voiced and hz == hz # ``hz == hz`` filters NaN without importing math/numpy here
|
|
]
|
|
|
|
|
|
def extract_energy(
|
|
path: str, hop_length: int = 512, max_analysis_seconds: float = 1200.0
|
|
) -> Optional[List[Tuple[float, float]]]:
|
|
"""Frame-level RMS energy track via librosa.
|
|
|
|
Returns ``[(time_seconds, rms), ...]``, or ``None`` when librosa is
|
|
unavailable or the file cannot be analyzed.
|
|
"""
|
|
file_path = Path(path)
|
|
if not file_path.is_file():
|
|
return None
|
|
try:
|
|
import librosa
|
|
except ImportError:
|
|
logger.info("librosa not installed; energy extraction unavailable")
|
|
return None
|
|
try:
|
|
with decodable_audio(str(file_path)) as audio_path:
|
|
if audio_path is None:
|
|
return None
|
|
y, sr = librosa.load(audio_path, sr=None, mono=True, duration=max_analysis_seconds)
|
|
rms = librosa.feature.rms(y=y, hop_length=hop_length)[0]
|
|
times = librosa.times_like(rms, sr=sr, hop_length=hop_length)
|
|
except Exception:
|
|
logger.warning("librosa energy analysis failed for %s", file_path)
|
|
return None
|
|
return [(float(t), float(r)) for t, r in zip(times, rms)]
|
|
|
|
|
|
def _window_average(track: Sequence[Tuple[float, float]], start: float, end: float) -> Optional[float]:
|
|
"""Average of ``track`` values whose timestamp falls in ``[start, end]``."""
|
|
values = [v for t, v in track if start <= t <= end]
|
|
if not values:
|
|
return None
|
|
return sum(values) / len(values)
|
|
|
|
|
|
def word_pitch_energy(
|
|
words: Sequence[dict],
|
|
pitch_track: Optional[Sequence[Tuple[float, float]]],
|
|
energy_track: Optional[Sequence[Tuple[float, float]]],
|
|
) -> List[dict]:
|
|
"""Attach average pitch/energy over each word's ``[start, end]`` span.
|
|
|
|
Words carry ``pitch_hz``/``energy`` (``None`` when the span has no
|
|
voiced frames or a track is unavailable). Both tracks are the output of
|
|
:func:`extract_pitch`/:func:`extract_energy`.
|
|
"""
|
|
out: List[dict] = []
|
|
for w in words:
|
|
ww = dict(w)
|
|
start = float(w.get("start", 0.0))
|
|
end = float(w.get("end", start))
|
|
ww["pitch_hz"] = _window_average(pitch_track, start, end) if pitch_track else None
|
|
ww["energy"] = _window_average(energy_track, start, end) if energy_track else None
|
|
out.append(ww)
|
|
return out
|
|
|
|
|
|
def compute_speech_rate(words: Sequence[dict], window_seconds: float = 3.0) -> List[float]:
|
|
"""Local speech rate (words/second) around each word.
|
|
|
|
For word *i*, counts every word whose start falls within
|
|
``[start_i - window_seconds, start_i]`` and divides by
|
|
``window_seconds`` — a trailing local rate, cheap to compute and stable
|
|
against a single long/short word skewing the whole utterance's average.
|
|
"""
|
|
starts = [float(w.get("start", 0.0)) for w in words]
|
|
rates: List[float] = []
|
|
for i, s in enumerate(starts):
|
|
lo = s - window_seconds
|
|
count = sum(1 for t in starts[: i + 1] if t >= lo)
|
|
rates.append(count / window_seconds if window_seconds > 0 else 0.0)
|
|
return rates
|
|
|
|
|
|
def compute_pauses(words: Sequence[dict]) -> List[float]:
|
|
"""Silence (seconds) immediately before each word.
|
|
|
|
The first word's "pause before" is the time from the start of the audio
|
|
to its own start; every other word measures the gap since the previous
|
|
word's end (clamped to ``0`` for overlapping/adjacent words).
|
|
"""
|
|
pauses: List[float] = []
|
|
prev_end = 0.0
|
|
for w in words:
|
|
start = float(w.get("start", 0.0))
|
|
pauses.append(max(0.0, start - prev_end))
|
|
prev_end = float(w.get("end", start))
|
|
return pauses
|