"""Emphasis index — how much a spoken word "pops" acoustically. Pure functions over already-extracted per-word features (energy, pitch delta, rate delta, pause before, duration); no I/O, no external dependency. Combines them into a single ``[0, 1]`` score, configurable via :class:`EmphasisWeights` so the weighting can be tuned (and persisted, see ``model_manager.load_voice_analysis_config``) without touching code. """ from dataclasses import dataclass from typing import List, Sequence _FIELDS = ("energy", "pitch_variation", "rate_variation", "pause_before", "duration") @dataclass class EmphasisWeights: energy: float = 0.30 pitch_variation: float = 0.25 rate_variation: float = 0.20 pause_before: float = 0.15 duration: float = 0.10 def as_dict(self) -> dict: return {field: getattr(self, field) for field in _FIELDS} @classmethod def from_dict(cls, data: dict) -> "EmphasisWeights": defaults = cls() return cls(**{field: float(data.get(field, getattr(defaults, field))) for field in _FIELDS}) def _clamp01(x: float) -> float: return max(0.0, min(1.0, x)) def pause_weight( pause_before: float, max_pause: float = 1.5, ignore_above: float = 3.0 ) -> float: """How much a preceding silence counts as emphasis, in ``[0, 1]``. A short beat before a word is real emphasis: the speaker is setting it up. A *long* gap is not — it is an edit point, a B-roll insert, or the other person in the room talking. Measured on real footage, gaps of 6-9s were scoring as the most emphatic moments in the recording purely because the scale saturated, ranking a scene change above a word the speaker actually hit hard. So the contribution rises up to ``max_pause`` and then drops to zero past ``ignore_above``, instead of saturating. Set ``ignore_above`` to ``0`` to disable the cutoff and keep the old saturating behaviour. """ if pause_before <= 0 or max_pause <= 0: return 0.0 if ignore_above > 0 and pause_before > ignore_above: return 0.0 return _clamp01(pause_before / max_pause) def compute_emphasis( energy: float, pitch_delta: float, rate_delta: float, pause_before: float, word_duration: float, weights: EmphasisWeights = EmphasisWeights(), *, max_pause: float = 1.5, max_duration: float = 1.0, pause_ignore_above: float = 3.0, ) -> float: """Emphasis score in ``[0, 1]`` for one word. ``energy``/``pitch_delta``/``rate_delta`` are expected already normalized to roughly ``[0, 1]`` (deltas may be negative — only their magnitude counts as emphasis). ``pause_before``/``word_duration`` are raw seconds; duration saturates at ``max_duration``, while the pause contribution is shaped by :func:`pause_weight`. """ energy_n = _clamp01(energy) pitch_n = _clamp01(abs(pitch_delta)) rate_n = _clamp01(abs(rate_delta)) pause_n = pause_weight(pause_before, max_pause, pause_ignore_above) duration_n = _clamp01(word_duration / max_duration) if max_duration > 0 else 0.0 total_weight = sum(getattr(weights, field) for field in _FIELDS) if total_weight <= 0: return 0.0 score = ( weights.energy * energy_n + weights.pitch_variation * pitch_n + weights.rate_variation * rate_n + weights.pause_before * pause_n + weights.duration * duration_n ) return _clamp01(score / total_weight) def annotate_emphasis( words: Sequence[dict], weights: EmphasisWeights = EmphasisWeights(), *, max_pause: float = 1.5, max_duration: float = 1.0, pause_ignore_above: float = 3.0, ) -> List[dict]: """Return copies of ``words`` with an ``"emphasis"`` key added. Each word dict is expected to carry ``energy``, ``pitch_delta``, ``rate_delta``, ``pause_before`` (all pre-computed, e.g. by ``voice_features.py``), plus ``start``/``end`` — or an explicit ``duration`` — to derive word length. """ out: List[dict] = [] for w in words: ww = dict(w) duration = ww.get("duration") if duration is None: duration = max(0.0, float(ww.get("end", 0.0)) - float(ww.get("start", 0.0))) ww["emphasis"] = compute_emphasis( energy=float(ww.get("energy") or 0.0), pitch_delta=float(ww.get("pitch_delta") or 0.0), rate_delta=float(ww.get("rate_delta") or 0.0), pause_before=float(ww.get("pause_before") or 0.0), word_duration=float(duration), weights=weights, max_pause=max_pause, max_duration=max_duration, pause_ignore_above=pause_ignore_above, ) out.append(ww) return out