chore: atualização geral
This commit is contained in:
@@ -0,0 +1,133 @@
|
||||
"""Emphasis index — how much a spoken word "pops" acoustically.
|
||||
|
||||
Pure functions over already-extracted per-word features (energy, pitch
|
||||
delta, rate delta, pause before, duration); no I/O, no external dependency.
|
||||
Combines them into a single ``[0, 1]`` score, configurable via
|
||||
:class:`EmphasisWeights` so the weighting can be tuned (and persisted,
|
||||
see ``model_manager.load_voice_analysis_config``) without touching code.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import List, Sequence
|
||||
|
||||
_FIELDS = ("energy", "pitch_variation", "rate_variation", "pause_before", "duration")
|
||||
|
||||
|
||||
@dataclass
|
||||
class EmphasisWeights:
|
||||
energy: float = 0.30
|
||||
pitch_variation: float = 0.25
|
||||
rate_variation: float = 0.20
|
||||
pause_before: float = 0.15
|
||||
duration: float = 0.10
|
||||
|
||||
def as_dict(self) -> dict:
|
||||
return {field: getattr(self, field) for field in _FIELDS}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, data: dict) -> "EmphasisWeights":
|
||||
defaults = cls()
|
||||
return cls(**{field: float(data.get(field, getattr(defaults, field))) for field in _FIELDS})
|
||||
|
||||
|
||||
def _clamp01(x: float) -> float:
|
||||
return max(0.0, min(1.0, x))
|
||||
|
||||
|
||||
def pause_weight(
|
||||
pause_before: float, max_pause: float = 1.5, ignore_above: float = 3.0
|
||||
) -> float:
|
||||
"""How much a preceding silence counts as emphasis, in ``[0, 1]``.
|
||||
|
||||
A short beat before a word is real emphasis: the speaker is setting it
|
||||
up. A *long* gap is not — it is an edit point, a B-roll insert, or the
|
||||
other person in the room talking. Measured on real footage, gaps of
|
||||
6-9s were scoring as the most emphatic moments in the recording purely
|
||||
because the scale saturated, ranking a scene change above a word the
|
||||
speaker actually hit hard.
|
||||
|
||||
So the contribution rises up to ``max_pause`` and then drops to zero
|
||||
past ``ignore_above``, instead of saturating. Set ``ignore_above`` to
|
||||
``0`` to disable the cutoff and keep the old saturating behaviour.
|
||||
"""
|
||||
if pause_before <= 0 or max_pause <= 0:
|
||||
return 0.0
|
||||
if ignore_above > 0 and pause_before > ignore_above:
|
||||
return 0.0
|
||||
return _clamp01(pause_before / max_pause)
|
||||
|
||||
|
||||
def compute_emphasis(
|
||||
energy: float,
|
||||
pitch_delta: float,
|
||||
rate_delta: float,
|
||||
pause_before: float,
|
||||
word_duration: float,
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
*,
|
||||
max_pause: float = 1.5,
|
||||
max_duration: float = 1.0,
|
||||
pause_ignore_above: float = 3.0,
|
||||
) -> float:
|
||||
"""Emphasis score in ``[0, 1]`` for one word.
|
||||
|
||||
``energy``/``pitch_delta``/``rate_delta`` are expected already
|
||||
normalized to roughly ``[0, 1]`` (deltas may be negative — only their
|
||||
magnitude counts as emphasis). ``pause_before``/``word_duration`` are
|
||||
raw seconds; duration saturates at ``max_duration``, while the pause
|
||||
contribution is shaped by :func:`pause_weight`.
|
||||
"""
|
||||
energy_n = _clamp01(energy)
|
||||
pitch_n = _clamp01(abs(pitch_delta))
|
||||
rate_n = _clamp01(abs(rate_delta))
|
||||
pause_n = pause_weight(pause_before, max_pause, pause_ignore_above)
|
||||
duration_n = _clamp01(word_duration / max_duration) if max_duration > 0 else 0.0
|
||||
|
||||
total_weight = sum(getattr(weights, field) for field in _FIELDS)
|
||||
if total_weight <= 0:
|
||||
return 0.0
|
||||
|
||||
score = (
|
||||
weights.energy * energy_n
|
||||
+ weights.pitch_variation * pitch_n
|
||||
+ weights.rate_variation * rate_n
|
||||
+ weights.pause_before * pause_n
|
||||
+ weights.duration * duration_n
|
||||
)
|
||||
return _clamp01(score / total_weight)
|
||||
|
||||
|
||||
def annotate_emphasis(
|
||||
words: Sequence[dict],
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
*,
|
||||
max_pause: float = 1.5,
|
||||
max_duration: float = 1.0,
|
||||
pause_ignore_above: float = 3.0,
|
||||
) -> List[dict]:
|
||||
"""Return copies of ``words`` with an ``"emphasis"`` key added.
|
||||
|
||||
Each word dict is expected to carry ``energy``, ``pitch_delta``,
|
||||
``rate_delta``, ``pause_before`` (all pre-computed, e.g. by
|
||||
``voice_features.py``), plus ``start``/``end`` — or an explicit
|
||||
``duration`` — to derive word length.
|
||||
"""
|
||||
out: List[dict] = []
|
||||
for w in words:
|
||||
ww = dict(w)
|
||||
duration = ww.get("duration")
|
||||
if duration is None:
|
||||
duration = max(0.0, float(ww.get("end", 0.0)) - float(ww.get("start", 0.0)))
|
||||
ww["emphasis"] = compute_emphasis(
|
||||
energy=float(ww.get("energy") or 0.0),
|
||||
pitch_delta=float(ww.get("pitch_delta") or 0.0),
|
||||
rate_delta=float(ww.get("rate_delta") or 0.0),
|
||||
pause_before=float(ww.get("pause_before") or 0.0),
|
||||
word_duration=float(duration),
|
||||
weights=weights,
|
||||
max_pause=max_pause,
|
||||
max_duration=max_duration,
|
||||
pause_ignore_above=pause_ignore_above,
|
||||
)
|
||||
out.append(ww)
|
||||
return out
|
||||
Reference in New Issue
Block a user