134 lines
4.7 KiB
Python
134 lines
4.7 KiB
Python
"""Emphasis index — how much a spoken word "pops" acoustically.
|
|
|
|
Pure functions over already-extracted per-word features (energy, pitch
|
|
delta, rate delta, pause before, duration); no I/O, no external dependency.
|
|
Combines them into a single ``[0, 1]`` score, configurable via
|
|
:class:`EmphasisWeights` so the weighting can be tuned (and persisted,
|
|
see ``model_manager.load_voice_analysis_config``) without touching code.
|
|
"""
|
|
|
|
from dataclasses import dataclass
|
|
from typing import List, Sequence
|
|
|
|
_FIELDS = ("energy", "pitch_variation", "rate_variation", "pause_before", "duration")
|
|
|
|
|
|
@dataclass
|
|
class EmphasisWeights:
|
|
energy: float = 0.30
|
|
pitch_variation: float = 0.25
|
|
rate_variation: float = 0.20
|
|
pause_before: float = 0.15
|
|
duration: float = 0.10
|
|
|
|
def as_dict(self) -> dict:
|
|
return {field: getattr(self, field) for field in _FIELDS}
|
|
|
|
@classmethod
|
|
def from_dict(cls, data: dict) -> "EmphasisWeights":
|
|
defaults = cls()
|
|
return cls(**{field: float(data.get(field, getattr(defaults, field))) for field in _FIELDS})
|
|
|
|
|
|
def _clamp01(x: float) -> float:
|
|
return max(0.0, min(1.0, x))
|
|
|
|
|
|
def pause_weight(
|
|
pause_before: float, max_pause: float = 1.5, ignore_above: float = 3.0
|
|
) -> float:
|
|
"""How much a preceding silence counts as emphasis, in ``[0, 1]``.
|
|
|
|
A short beat before a word is real emphasis: the speaker is setting it
|
|
up. A *long* gap is not — it is an edit point, a B-roll insert, or the
|
|
other person in the room talking. Measured on real footage, gaps of
|
|
6-9s were scoring as the most emphatic moments in the recording purely
|
|
because the scale saturated, ranking a scene change above a word the
|
|
speaker actually hit hard.
|
|
|
|
So the contribution rises up to ``max_pause`` and then drops to zero
|
|
past ``ignore_above``, instead of saturating. Set ``ignore_above`` to
|
|
``0`` to disable the cutoff and keep the old saturating behaviour.
|
|
"""
|
|
if pause_before <= 0 or max_pause <= 0:
|
|
return 0.0
|
|
if ignore_above > 0 and pause_before > ignore_above:
|
|
return 0.0
|
|
return _clamp01(pause_before / max_pause)
|
|
|
|
|
|
def compute_emphasis(
|
|
energy: float,
|
|
pitch_delta: float,
|
|
rate_delta: float,
|
|
pause_before: float,
|
|
word_duration: float,
|
|
weights: EmphasisWeights = EmphasisWeights(),
|
|
*,
|
|
max_pause: float = 1.5,
|
|
max_duration: float = 1.0,
|
|
pause_ignore_above: float = 3.0,
|
|
) -> float:
|
|
"""Emphasis score in ``[0, 1]`` for one word.
|
|
|
|
``energy``/``pitch_delta``/``rate_delta`` are expected already
|
|
normalized to roughly ``[0, 1]`` (deltas may be negative — only their
|
|
magnitude counts as emphasis). ``pause_before``/``word_duration`` are
|
|
raw seconds; duration saturates at ``max_duration``, while the pause
|
|
contribution is shaped by :func:`pause_weight`.
|
|
"""
|
|
energy_n = _clamp01(energy)
|
|
pitch_n = _clamp01(abs(pitch_delta))
|
|
rate_n = _clamp01(abs(rate_delta))
|
|
pause_n = pause_weight(pause_before, max_pause, pause_ignore_above)
|
|
duration_n = _clamp01(word_duration / max_duration) if max_duration > 0 else 0.0
|
|
|
|
total_weight = sum(getattr(weights, field) for field in _FIELDS)
|
|
if total_weight <= 0:
|
|
return 0.0
|
|
|
|
score = (
|
|
weights.energy * energy_n
|
|
+ weights.pitch_variation * pitch_n
|
|
+ weights.rate_variation * rate_n
|
|
+ weights.pause_before * pause_n
|
|
+ weights.duration * duration_n
|
|
)
|
|
return _clamp01(score / total_weight)
|
|
|
|
|
|
def annotate_emphasis(
|
|
words: Sequence[dict],
|
|
weights: EmphasisWeights = EmphasisWeights(),
|
|
*,
|
|
max_pause: float = 1.5,
|
|
max_duration: float = 1.0,
|
|
pause_ignore_above: float = 3.0,
|
|
) -> List[dict]:
|
|
"""Return copies of ``words`` with an ``"emphasis"`` key added.
|
|
|
|
Each word dict is expected to carry ``energy``, ``pitch_delta``,
|
|
``rate_delta``, ``pause_before`` (all pre-computed, e.g. by
|
|
``voice_features.py``), plus ``start``/``end`` — or an explicit
|
|
``duration`` — to derive word length.
|
|
"""
|
|
out: List[dict] = []
|
|
for w in words:
|
|
ww = dict(w)
|
|
duration = ww.get("duration")
|
|
if duration is None:
|
|
duration = max(0.0, float(ww.get("end", 0.0)) - float(ww.get("start", 0.0)))
|
|
ww["emphasis"] = compute_emphasis(
|
|
energy=float(ww.get("energy") or 0.0),
|
|
pitch_delta=float(ww.get("pitch_delta") or 0.0),
|
|
rate_delta=float(ww.get("rate_delta") or 0.0),
|
|
pause_before=float(ww.get("pause_before") or 0.0),
|
|
word_duration=float(duration),
|
|
weights=weights,
|
|
max_pause=max_pause,
|
|
max_duration=max_duration,
|
|
pause_ignore_above=pause_ignore_above,
|
|
)
|
|
out.append(ww)
|
|
return out
|