"""Acoustic features for voice analysis — pitch, energy, rate, pauses. Mirrors the ``media_intel.py`` contract: librosa is an optional dependency (``pip install 'fcp-mcp-server[intelligence]'``, already required by beat detection), imported lazily, and every extractor degrades to ``None`` when the library is missing or the file cannot be analyzed — never crashes. ``compute_speech_rate``/``compute_pauses`` are pure functions over word-timestamp dicts (the shape ``transcribe.py`` already produces) and need no audio file at all. """ import contextlib import logging import shutil import subprocess import tempfile from pathlib import Path from typing import Iterator, List, Optional, Sequence, Tuple logger = logging.getLogger(__name__) # Human voice fundamental frequency range (covers low male to high female/child). PITCH_FMIN_HZ = 65.0 PITCH_FMAX_HZ = 1000.0 # Formats librosa reads directly through soundfile. Anything else — notably # the .mov/.mp4 that source footage actually arrives in — must be decoded by # ffmpeg first, or analysis fails outright. NATIVE_AUDIO_SUFFIXES = {".wav", ".aif", ".aiff", ".flac"} # Voice analysis only needs the speech band: 16 kHz mono is well above the # Nyquist limit for our 1 kHz pitch ceiling, and keeps the extracted file # small and fast to decode even for hour-long footage. EXTRACT_SAMPLE_RATE = 16000 EXTRACT_TIMEOUT_SECONDS = 600 @contextlib.contextmanager def decodable_audio(path: str) -> Iterator[Optional[str]]: """Yield a path librosa can read, extracting the audio track if needed. Audio files pass straight through. Video containers are decoded to a temporary mono WAV with ffmpeg and cleaned up on exit. Yields ``None`` when the audio cannot be obtained (no ffmpeg, no audio track, failure), keeping the graceful-degradation contract of this module. """ file_path = Path(path) if file_path.suffix.lower() in NATIVE_AUDIO_SUFFIXES: yield str(file_path) return if shutil.which("ffmpeg") is None: logger.info("ffmpeg not found on PATH; cannot extract audio from %s", file_path) yield None return tmp_dir = tempfile.mkdtemp(prefix="fcp_voice_") wav_path = Path(tmp_dir) / "audio.wav" try: result = subprocess.run( [ "ffmpeg", "-hide_banner", "-nostdin", "-y", "-i", str(file_path), "-vn", # audio only: decoding video would dominate the runtime "-ac", "1", "-ar", str(EXTRACT_SAMPLE_RATE), str(wav_path), ], capture_output=True, text=True, timeout=EXTRACT_TIMEOUT_SECONDS, ) if result.returncode != 0 or not wav_path.is_file(): logger.warning("ffmpeg could not extract audio from %s", file_path) yield None else: yield str(wav_path) except (OSError, subprocess.TimeoutExpired): logger.warning("audio extraction failed for %s", file_path) yield None finally: shutil.rmtree(tmp_dir, ignore_errors=True) def features_capability() -> Tuple[bool, str]: """Whether pitch/energy extraction is available (librosa installed).""" try: import librosa # noqa: F401 except Exception: return False, "Análise acústica indisponível: componente librosa ausente." return True, "Análise acústica disponível." def extract_pitch( path: str, hop_length: int = 512, max_analysis_seconds: float = 1200.0 ) -> Optional[List[Tuple[float, float]]]: """Frame-level pitch (F0) track via librosa's ``pyin``. Returns ``[(time_seconds, hz), ...]`` for voiced frames only (unvoiced frames, where ``pyin`` reports no pitch, are dropped), or ``None`` when librosa is unavailable or the file cannot be analyzed. """ file_path = Path(path) if not file_path.is_file(): return None try: import librosa except ImportError: logger.info("librosa not installed; pitch extraction unavailable") return None try: with decodable_audio(str(file_path)) as audio_path: if audio_path is None: return None y, sr = librosa.load(audio_path, sr=None, mono=True, duration=max_analysis_seconds) f0, voiced_flag, _voiced_prob = librosa.pyin( y, fmin=PITCH_FMIN_HZ, fmax=PITCH_FMAX_HZ, sr=sr, hop_length=hop_length ) times = librosa.times_like(f0, sr=sr, hop_length=hop_length) except Exception: logger.warning("librosa pitch analysis failed for %s", file_path) return None return [ (float(t), float(hz)) for t, hz, voiced in zip(times, f0, voiced_flag) if voiced and hz == hz # ``hz == hz`` filters NaN without importing math/numpy here ] def extract_energy( path: str, hop_length: int = 512, max_analysis_seconds: float = 1200.0 ) -> Optional[List[Tuple[float, float]]]: """Frame-level RMS energy track via librosa. Returns ``[(time_seconds, rms), ...]``, or ``None`` when librosa is unavailable or the file cannot be analyzed. """ file_path = Path(path) if not file_path.is_file(): return None try: import librosa except ImportError: logger.info("librosa not installed; energy extraction unavailable") return None try: with decodable_audio(str(file_path)) as audio_path: if audio_path is None: return None y, sr = librosa.load(audio_path, sr=None, mono=True, duration=max_analysis_seconds) rms = librosa.feature.rms(y=y, hop_length=hop_length)[0] times = librosa.times_like(rms, sr=sr, hop_length=hop_length) except Exception: logger.warning("librosa energy analysis failed for %s", file_path) return None return [(float(t), float(r)) for t, r in zip(times, rms)] def _window_average(track: Sequence[Tuple[float, float]], start: float, end: float) -> Optional[float]: """Average of ``track`` values whose timestamp falls in ``[start, end]``.""" values = [v for t, v in track if start <= t <= end] if not values: return None return sum(values) / len(values) def word_pitch_energy( words: Sequence[dict], pitch_track: Optional[Sequence[Tuple[float, float]]], energy_track: Optional[Sequence[Tuple[float, float]]], ) -> List[dict]: """Attach average pitch/energy over each word's ``[start, end]`` span. Words carry ``pitch_hz``/``energy`` (``None`` when the span has no voiced frames or a track is unavailable). Both tracks are the output of :func:`extract_pitch`/:func:`extract_energy`. """ out: List[dict] = [] for w in words: ww = dict(w) start = float(w.get("start", 0.0)) end = float(w.get("end", start)) ww["pitch_hz"] = _window_average(pitch_track, start, end) if pitch_track else None ww["energy"] = _window_average(energy_track, start, end) if energy_track else None out.append(ww) return out def compute_speech_rate(words: Sequence[dict], window_seconds: float = 3.0) -> List[float]: """Local speech rate (words/second) around each word. For word *i*, counts every word whose start falls within ``[start_i - window_seconds, start_i]`` and divides by ``window_seconds`` — a trailing local rate, cheap to compute and stable against a single long/short word skewing the whole utterance's average. """ starts = [float(w.get("start", 0.0)) for w in words] rates: List[float] = [] for i, s in enumerate(starts): lo = s - window_seconds count = sum(1 for t in starts[: i + 1] if t >= lo) rates.append(count / window_seconds if window_seconds > 0 else 0.0) return rates def compute_pauses(words: Sequence[dict]) -> List[float]: """Silence (seconds) immediately before each word. The first word's "pause before" is the time from the start of the audio to its own start; every other word measures the gap since the previous word's end (clamped to ``0`` for overlapping/adjacent words). """ pauses: List[float] = [] prev_end = 0.0 for w in words: start = float(w.get("start", 0.0)) pauses.append(max(0.0, start - prev_end)) prev_end = float(w.get("end", start)) return pauses