178 lines
6.7 KiB
Python
Executable File
178 lines
6.7 KiB
Python
Executable File
"""Media intelligence — real analysis of source media referenced by timelines.
|
|
|
|
v0.10 slice 1: audio silence detection via ffmpeg's silencedetect filter.
|
|
No new Python dependencies: ffmpeg is invoked as a bounded subprocess
|
|
(list-form arguments, validated numeric parameters, hard timeout), and
|
|
detection degrades gracefully — ``detect_silence`` returns ``None`` when
|
|
ffmpeg is unavailable or the file cannot be analyzed, so callers can fall
|
|
back or report instead of crashing.
|
|
"""
|
|
|
|
import logging
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
from pathlib import Path
|
|
from typing import List, Optional, Tuple
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Hard ceiling on a single ffmpeg analysis pass. Decoding audio-only is far
|
|
# faster than realtime, so this covers multi-hour media while still bounding
|
|
# an adversarial/corrupt file that makes the decoder hang.
|
|
PROBE_TIMEOUT_SECONDS = 120
|
|
|
|
# silencedetect prints times as plain seconds on stderr; starts can be
|
|
# slightly negative (encoder priming samples), so allow a leading minus.
|
|
_SILENCE_START_RE = re.compile(r"silence_start:\s*(-?\d+(?:\.\d+)?)")
|
|
_SILENCE_END_RE = re.compile(r"silence_end:\s*(-?\d+(?:\.\d+)?)")
|
|
_DURATION_RE = re.compile(r"Duration:\s*(\d+):(\d\d):(\d\d(?:\.\d+)?)")
|
|
|
|
|
|
def parse_silencedetect_output(
|
|
stderr: str, total_duration: Optional[float] = None
|
|
) -> List[Tuple[float, float]]:
|
|
"""Parse ffmpeg silencedetect stderr into (start, end) ranges in seconds.
|
|
|
|
A trailing ``silence_start`` with no matching ``silence_end`` (media that
|
|
ends silent) is closed at ``total_duration`` when known, otherwise dropped.
|
|
"""
|
|
ranges: List[Tuple[float, float]] = []
|
|
pending: Optional[float] = None
|
|
for line in stderr.splitlines():
|
|
start_match = _SILENCE_START_RE.search(line)
|
|
if start_match:
|
|
pending = max(0.0, float(start_match.group(1)))
|
|
continue
|
|
end_match = _SILENCE_END_RE.search(line)
|
|
if end_match and pending is not None:
|
|
ranges.append((pending, float(end_match.group(1))))
|
|
pending = None
|
|
if pending is not None and total_duration is not None and total_duration > pending:
|
|
ranges.append((pending, total_duration))
|
|
return ranges
|
|
|
|
|
|
def map_silence_to_timeline(
|
|
silences: List[Tuple[float, float]],
|
|
source_start: float,
|
|
clip_duration: float,
|
|
timeline_offset: float,
|
|
) -> List[Tuple[float, float]]:
|
|
"""Map source-time silence ranges onto the timeline.
|
|
|
|
A clip uses ``[source_start, source_start + clip_duration)`` of its source
|
|
media and sits at ``timeline_offset``. Ranges outside the used window are
|
|
excluded; ranges overlapping its edges are clamped.
|
|
"""
|
|
source_end = source_start + clip_duration
|
|
mapped: List[Tuple[float, float]] = []
|
|
for start, end in silences:
|
|
clamped_start = max(start, source_start)
|
|
clamped_end = min(end, source_end)
|
|
if clamped_end <= clamped_start:
|
|
continue
|
|
mapped.append((
|
|
timeline_offset + (clamped_start - source_start),
|
|
timeline_offset + (clamped_end - source_start),
|
|
))
|
|
return mapped
|
|
|
|
|
|
def detect_beats(
|
|
path: str, max_analysis_seconds: float = 1200.0
|
|
) -> Optional[dict]:
|
|
"""Detect musical beats in an audio file via librosa's beat tracker.
|
|
|
|
librosa is an optional dependency (``pip install 'fcp-mcp-server[intelligence]'``);
|
|
without it, or when the file is missing/unreadable, this returns ``None``
|
|
so callers can degrade to a helpful message instead of crashing.
|
|
|
|
Returns:
|
|
``{'bpm': float, 'beats': [seconds, ...]}`` or ``None``.
|
|
"""
|
|
file_path = Path(path)
|
|
if not file_path.is_file():
|
|
return None
|
|
try:
|
|
import librosa
|
|
except ImportError:
|
|
logger.info("librosa not installed; beat detection unavailable")
|
|
return None
|
|
try:
|
|
# duration cap bounds memory on adversarially long media
|
|
y, sr = librosa.load(str(file_path), sr=None, mono=True,
|
|
duration=max_analysis_seconds)
|
|
tempo, frames = librosa.beat.beat_track(y=y, sr=sr)
|
|
beats = librosa.frames_to_time(frames, sr=sr)
|
|
except Exception:
|
|
logger.warning("librosa beat analysis failed for %s", file_path)
|
|
return None
|
|
bpm = float(tempo[0] if hasattr(tempo, "__len__") else tempo)
|
|
return {"bpm": bpm, "beats": [float(b) for b in beats]}
|
|
|
|
|
|
def media_src_to_path(src: str) -> str:
|
|
"""Convert an FCPXML media src (``file://`` URL or plain path) to a filesystem path."""
|
|
if src.startswith("file://"):
|
|
from urllib.parse import unquote, urlparse
|
|
|
|
return unquote(urlparse(src).path)
|
|
return src
|
|
|
|
|
|
def _parse_total_duration(stderr: str) -> Optional[float]:
|
|
match = _DURATION_RE.search(stderr)
|
|
if not match:
|
|
return None
|
|
hours, minutes, seconds = match.groups()
|
|
return int(hours) * 3600 + int(minutes) * 60 + float(seconds)
|
|
|
|
|
|
def detect_silence(
|
|
path: str, noise_db: float = -30.0, min_duration: float = 0.5
|
|
) -> Optional[List[Tuple[float, float]]]:
|
|
"""Detect silence in an audio/video file's first audio stream.
|
|
|
|
Returns (start, end) ranges in source seconds, or ``None`` when the file
|
|
is missing, ffmpeg is unavailable, or analysis fails. Raises ``ValueError``
|
|
on out-of-bounds parameters (they end up in a subprocess argument, so they
|
|
are validated, not trusted).
|
|
"""
|
|
if not (-120.0 <= noise_db <= 0.0):
|
|
raise ValueError(f"noise_db must be between -120 and 0 dB, got {noise_db}")
|
|
if not (0 < min_duration <= 3600):
|
|
raise ValueError(f"min_duration must be between 0 and 3600 seconds, got {min_duration}")
|
|
|
|
file_path = Path(path)
|
|
if not file_path.is_file():
|
|
return None
|
|
if shutil.which("ffmpeg") is None:
|
|
logger.info("ffmpeg not found on PATH; silence detection unavailable")
|
|
return None
|
|
|
|
try:
|
|
result = subprocess.run(
|
|
[
|
|
"ffmpeg", "-hide_banner", "-nostdin",
|
|
"-i", str(file_path),
|
|
# -vn: silence detection only needs the audio stream. Without it
|
|
# ffmpeg decodes the full video track into the null muxer, which
|
|
# blows past PROBE_TIMEOUT_SECONDS on long/high-bitrate files.
|
|
"-vn",
|
|
"-af", f"silencedetect=noise={float(noise_db)}dB:d={float(min_duration)}",
|
|
"-f", "null", "-",
|
|
],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=PROBE_TIMEOUT_SECONDS,
|
|
)
|
|
except (OSError, subprocess.TimeoutExpired):
|
|
logger.warning("ffmpeg silence analysis failed for %s", file_path)
|
|
return None
|
|
if result.returncode != 0:
|
|
return None
|
|
return parse_silencedetect_output(
|
|
result.stderr, total_duration=_parse_total_duration(result.stderr)
|
|
)
|