Files
gart/code/fcpxml/emphasis.py
T

134 lines
4.7 KiB
Python

"""Emphasis index — how much a spoken word "pops" acoustically.
Pure functions over already-extracted per-word features (energy, pitch
delta, rate delta, pause before, duration); no I/O, no external dependency.
Combines them into a single ``[0, 1]`` score, configurable via
:class:`EmphasisWeights` so the weighting can be tuned (and persisted,
see ``model_manager.load_voice_analysis_config``) without touching code.
"""
from dataclasses import dataclass
from typing import List, Sequence
_FIELDS = ("energy", "pitch_variation", "rate_variation", "pause_before", "duration")
@dataclass
class EmphasisWeights:
energy: float = 0.30
pitch_variation: float = 0.25
rate_variation: float = 0.20
pause_before: float = 0.15
duration: float = 0.10
def as_dict(self) -> dict:
return {field: getattr(self, field) for field in _FIELDS}
@classmethod
def from_dict(cls, data: dict) -> "EmphasisWeights":
defaults = cls()
return cls(**{field: float(data.get(field, getattr(defaults, field))) for field in _FIELDS})
def _clamp01(x: float) -> float:
return max(0.0, min(1.0, x))
def pause_weight(
pause_before: float, max_pause: float = 1.5, ignore_above: float = 3.0
) -> float:
"""How much a preceding silence counts as emphasis, in ``[0, 1]``.
A short beat before a word is real emphasis: the speaker is setting it
up. A *long* gap is not — it is an edit point, a B-roll insert, or the
other person in the room talking. Measured on real footage, gaps of
6-9s were scoring as the most emphatic moments in the recording purely
because the scale saturated, ranking a scene change above a word the
speaker actually hit hard.
So the contribution rises up to ``max_pause`` and then drops to zero
past ``ignore_above``, instead of saturating. Set ``ignore_above`` to
``0`` to disable the cutoff and keep the old saturating behaviour.
"""
if pause_before <= 0 or max_pause <= 0:
return 0.0
if ignore_above > 0 and pause_before > ignore_above:
return 0.0
return _clamp01(pause_before / max_pause)
def compute_emphasis(
energy: float,
pitch_delta: float,
rate_delta: float,
pause_before: float,
word_duration: float,
weights: EmphasisWeights = EmphasisWeights(),
*,
max_pause: float = 1.5,
max_duration: float = 1.0,
pause_ignore_above: float = 3.0,
) -> float:
"""Emphasis score in ``[0, 1]`` for one word.
``energy``/``pitch_delta``/``rate_delta`` are expected already
normalized to roughly ``[0, 1]`` (deltas may be negative — only their
magnitude counts as emphasis). ``pause_before``/``word_duration`` are
raw seconds; duration saturates at ``max_duration``, while the pause
contribution is shaped by :func:`pause_weight`.
"""
energy_n = _clamp01(energy)
pitch_n = _clamp01(abs(pitch_delta))
rate_n = _clamp01(abs(rate_delta))
pause_n = pause_weight(pause_before, max_pause, pause_ignore_above)
duration_n = _clamp01(word_duration / max_duration) if max_duration > 0 else 0.0
total_weight = sum(getattr(weights, field) for field in _FIELDS)
if total_weight <= 0:
return 0.0
score = (
weights.energy * energy_n
+ weights.pitch_variation * pitch_n
+ weights.rate_variation * rate_n
+ weights.pause_before * pause_n
+ weights.duration * duration_n
)
return _clamp01(score / total_weight)
def annotate_emphasis(
words: Sequence[dict],
weights: EmphasisWeights = EmphasisWeights(),
*,
max_pause: float = 1.5,
max_duration: float = 1.0,
pause_ignore_above: float = 3.0,
) -> List[dict]:
"""Return copies of ``words`` with an ``"emphasis"`` key added.
Each word dict is expected to carry ``energy``, ``pitch_delta``,
``rate_delta``, ``pause_before`` (all pre-computed, e.g. by
``voice_features.py``), plus ``start``/``end`` — or an explicit
``duration`` — to derive word length.
"""
out: List[dict] = []
for w in words:
ww = dict(w)
duration = ww.get("duration")
if duration is None:
duration = max(0.0, float(ww.get("end", 0.0)) - float(ww.get("start", 0.0)))
ww["emphasis"] = compute_emphasis(
energy=float(ww.get("energy") or 0.0),
pitch_delta=float(ww.get("pitch_delta") or 0.0),
rate_delta=float(ww.get("rate_delta") or 0.0),
pause_before=float(ww.get("pause_before") or 0.0),
word_duration=float(duration),
weights=weights,
max_pause=max_pause,
max_duration=max_duration,
pause_ignore_above=pause_ignore_above,
)
out.append(ww)
return out