506 lines
21 KiB
Python
506 lines
21 KiB
Python
"""Voice timeline — the consolidated, AI-readable view of how a video is spoken.
|
||
|
||
This is the *source of truth* between analysis and editing: it merges what
|
||
was said (transcript), who said it (diarization), and how it was said
|
||
(pitch/energy/rate/pauses → emphasis) into one JSON document, decoupling the
|
||
audio analysis from FCPXML generation entirely.
|
||
|
||
The shape is designed to be handed to a language model so it can reason about
|
||
the narrative — which beats carry weight, where a speaker changes, where the
|
||
delivery peaks — and decide how to direct the edit. Two design choices serve
|
||
that goal:
|
||
|
||
* **Layered, not flat.** A ``summary`` gives the whole picture in a few
|
||
numbers, ``segments`` group words into utterances with their own
|
||
aggregates, and ``words`` hold the fine detail. A model can reason from
|
||
the top layer and only descend where it matters, instead of parsing
|
||
thousands of word rows to find the shape of the piece.
|
||
* **Normalized, self-describing values.** Every acoustic value is 0–1 and
|
||
relative to *this* recording (a quiet podcast and a shouted ad both use
|
||
the full range), and ``scales`` documents that contract inline, so the
|
||
numbers are interpretable without external context.
|
||
"""
|
||
|
||
import json
|
||
import logging
|
||
from pathlib import Path
|
||
from typing import Callable, List, Optional, Sequence, Tuple
|
||
|
||
from .diarize import DEFAULT_SPEAKER, assign_speakers, build_speakers, diarize
|
||
from .emphasis import EmphasisWeights, annotate_emphasis
|
||
from .voice_features import (
|
||
compute_pauses,
|
||
compute_speech_rate,
|
||
extract_energy,
|
||
extract_pitch,
|
||
word_pitch_energy,
|
||
)
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
VOICE_TIMELINE_VERSION = "1.0"
|
||
|
||
# Silence long enough to mean the take stopped rather than the speaker paused.
|
||
# On real footage, boundaries between retakes showed gaps of 3.6-19.8s while
|
||
# dramatic beats inside a delivered line stayed under ~2s.
|
||
TAKE_BOUNDARY_GAP = 3.0
|
||
|
||
# How to read the values in this document, split by the level they live on.
|
||
# Embedded in the output so a model consuming the JSON needs no external
|
||
# documentation — and kept honest: a metric listed under "word" must exist on
|
||
# every word row, and one under "segment" on every segment row.
|
||
VALUE_SCALES = {
|
||
"word": {
|
||
"energy": "0-1, loudness relative to the loudest moment of this recording",
|
||
"pitch_delta": "0-1, how far this word's pitch sits from the speaker's average",
|
||
"rate_delta": "0-1, how much the local speaking rate departs from the average",
|
||
"pause_before": "seconds of silence immediately before the word",
|
||
"emphasis": "0-1 combined index; high values are punch-in/highlight candidates",
|
||
},
|
||
"segment": {
|
||
"gap_before": "seconds of silence before this line",
|
||
"take_boundary": "true when the gap is long enough that the take likely restarted here",
|
||
"avg_energy": "0-1 mean loudness across the line",
|
||
"peak_emphasis": "0-1 highest emphasis of any word in the line",
|
||
},
|
||
}
|
||
|
||
|
||
def _normalize(value: Optional[float], maximum: float) -> float:
|
||
"""Scale ``value`` into 0-1 against ``maximum`` (0.0 when unavailable)."""
|
||
if value is None or maximum <= 0:
|
||
return 0.0
|
||
return max(0.0, min(1.0, value / maximum))
|
||
|
||
|
||
def _round_word(word: dict) -> dict:
|
||
"""One word row, rounded to a size a model can read without noise.
|
||
|
||
The raw ``energy_raw``/``pitch_hz`` ride along beside the normalized
|
||
values so the document can be re-analyzed over a subset later. That
|
||
matters after cutting: every normalized value is relative to the
|
||
loudest moment of the *whole* recording, and if that moment gets cut
|
||
the survivors are scored against something that no longer exists.
|
||
"""
|
||
return {
|
||
"text": word.get("word", ""),
|
||
"start": round(float(word.get("start", 0.0)), 3),
|
||
"end": round(float(word.get("end", 0.0)), 3),
|
||
"speaker": word.get("speaker_id", DEFAULT_SPEAKER),
|
||
"energy": round(word.get("energy_norm", 0.0), 3),
|
||
"pitch_delta": round(word.get("pitch_delta", 0.0), 3),
|
||
"rate_delta": round(word.get("rate_delta", 0.0), 3),
|
||
"pause_before": round(word.get("pause_before", 0.0), 3),
|
||
"emphasis": round(word.get("emphasis", 0.0), 3),
|
||
"energy_raw": word.get("energy"),
|
||
"pitch_hz": word.get("pitch_hz"),
|
||
}
|
||
|
||
|
||
def enrich_words(
|
||
words: Sequence[dict],
|
||
pitch_track: Optional[Sequence] = None,
|
||
energy_track: Optional[Sequence] = None,
|
||
weights: EmphasisWeights = EmphasisWeights(),
|
||
already_measured: bool = False,
|
||
) -> List[dict]:
|
||
"""Attach normalized acoustic features + the emphasis index to each word.
|
||
|
||
Normalization is per-recording: energy against the loudest word, pitch
|
||
against the spread around this recording's average, rate against the
|
||
largest local departure. That makes the numbers comparable within a
|
||
piece regardless of how it was recorded.
|
||
|
||
Set ``already_measured`` when the words already carry ``energy`` and
|
||
``pitch_hz`` from a previous pass — re-analyzing a subset, say. The
|
||
frame tracks are then unnecessary, and sampling them again would
|
||
overwrite good values with ``None``.
|
||
"""
|
||
if not words:
|
||
return []
|
||
|
||
if not already_measured:
|
||
words = word_pitch_energy(words, pitch_track, energy_track)
|
||
rates = compute_speech_rate(words)
|
||
pauses = compute_pauses(words)
|
||
|
||
energies = [w["energy"] for w in words if w.get("energy") is not None]
|
||
max_energy = max(energies) if energies else 0.0
|
||
pitches = [w["pitch_hz"] for w in words if w.get("pitch_hz") is not None]
|
||
avg_pitch = sum(pitches) / len(pitches) if pitches else 0.0
|
||
pitch_span = (max(pitches) - min(pitches)) if len(pitches) > 1 else 0.0
|
||
avg_rate = sum(rates) / len(rates) if rates else 0.0
|
||
max_rate = max(rates) if rates else 0.0
|
||
|
||
enriched: List[dict] = []
|
||
for i, w in enumerate(words):
|
||
ww = dict(w)
|
||
ww["energy_norm"] = _normalize(w.get("energy"), max_energy)
|
||
pitch = w.get("pitch_hz")
|
||
ww["pitch_delta"] = (
|
||
_normalize(abs(pitch - avg_pitch), pitch_span) if pitch is not None else 0.0
|
||
)
|
||
ww["rate_delta"] = _normalize(abs(rates[i] - avg_rate), max_rate)
|
||
ww["pause_before"] = pauses[i]
|
||
enriched.append(ww)
|
||
|
||
annotated = annotate_emphasis(
|
||
[{**w, "energy": w["energy_norm"]} for w in enriched], weights=weights
|
||
)
|
||
for word, scored in zip(enriched, annotated):
|
||
word["emphasis"] = scored["emphasis"]
|
||
return enriched
|
||
|
||
|
||
def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]:
|
||
"""Group enriched words under their segment, with per-segment aggregates.
|
||
|
||
The aggregates are what let a model judge a whole utterance ("this line
|
||
is delivered hot, that one trails off") without reading every word.
|
||
"""
|
||
rows: List[dict] = []
|
||
previous_end = 0.0
|
||
for seg in segments:
|
||
start = float(seg.get("start", 0.0))
|
||
end = float(seg.get("end", 0.0))
|
||
in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end]
|
||
energies = [w["energy_norm"] for w in in_seg]
|
||
emphases = [w["emphasis"] for w in in_seg]
|
||
gap = max(0.0, start - previous_end)
|
||
rows.append(
|
||
{
|
||
"start": round(start, 3),
|
||
"end": round(end, 3),
|
||
"speaker": seg.get("speaker_id", DEFAULT_SPEAKER),
|
||
"text": (seg.get("text") or "").strip(),
|
||
# Silence before this line. Long gaps are where the camera
|
||
# stopped or the take restarted, so this is the structural
|
||
# hint for splitting a recording into takes — the same signal
|
||
# that is *noise* for emphasis (see emphasis.pause_weight).
|
||
"gap_before": round(gap, 3),
|
||
"take_boundary": gap >= TAKE_BOUNDARY_GAP,
|
||
"avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0,
|
||
"peak_emphasis": round(max(emphases), 3) if emphases else 0.0,
|
||
"words": [_round_word(w) for w in in_seg],
|
||
}
|
||
)
|
||
previous_end = end
|
||
return rows
|
||
|
||
|
||
# Words too common to ever be the point of a punch-in. A zoom lands on what a
|
||
# sentence is *about*, and an article spoken loudly is still an article.
|
||
_FUNCTION_WORDS = {
|
||
"a", "o", "e", "de", "da", "do", "que", "é", "em", "um", "uma", "as", "os",
|
||
"no", "na", "com", "pra", "para", "por", "se", "mais", "isso", "aí", "tudo",
|
||
"ao", "à", "dos", "das", "nos", "nas", "ou", "mas", "já", "ele", "ela",
|
||
"eu", "você", "seu", "sua", "meu", "minha", "esse", "essa", "aquele",
|
||
}
|
||
|
||
|
||
def _survives(start: float, end: float, cuts: Sequence[Tuple[float, float]]) -> bool:
|
||
"""Whether a span lies entirely outside every removed range."""
|
||
return all(end <= cut_start or start >= cut_end for cut_start, cut_end in cuts)
|
||
|
||
|
||
def restrict_to_kept(
|
||
timeline: dict,
|
||
cut_ranges: Sequence[Tuple[float, float]],
|
||
weights: EmphasisWeights = EmphasisWeights(),
|
||
peak_percentile: float = 0.02,
|
||
emphasis_floor: float = 0.25,
|
||
) -> dict:
|
||
"""Re-analyze a timeline over only the material that survives ``cut_ranges``.
|
||
|
||
Emphasis is *relative*: energy is scored against the loudest word,
|
||
pitch against the spread of the recording. Cut the loudest moment out —
|
||
a laugh, an aside to the crew — and every remaining score is measured
|
||
against something the viewer will never see. Re-running the
|
||
normalization over just the survivors is what makes "the most emphatic
|
||
line of the final video" a meaningful question.
|
||
|
||
Returns a timeline of the same shape, with times still in original
|
||
source seconds so the result can be fed straight back as actions.
|
||
"""
|
||
kept_words = [
|
||
w
|
||
for segment in timeline.get("segments", [])
|
||
for w in segment.get("words", [])
|
||
if _survives(w["start"], w["end"], cut_ranges)
|
||
]
|
||
# enrich_words expects the raw analysis keys, not the normalized ones.
|
||
raw = [
|
||
{
|
||
"word": w["text"],
|
||
"start": w["start"],
|
||
"end": w["end"],
|
||
"speaker_id": w.get("speaker", DEFAULT_SPEAKER),
|
||
"energy": w.get("energy_raw"),
|
||
"pitch_hz": w.get("pitch_hz"),
|
||
}
|
||
for w in kept_words
|
||
]
|
||
enriched = enrich_words(raw, weights=weights, already_measured=True)
|
||
|
||
kept_segments = [
|
||
{**s, "words": [w for w in s.get("words", []) if _survives(w["start"], w["end"], cut_ranges)]}
|
||
for s in timeline.get("segments", [])
|
||
]
|
||
kept_segments = [s for s in kept_segments if s["words"]]
|
||
rows = _segment_rows(
|
||
[{"text": s["text"], "start": s["start"], "end": s["end"],
|
||
"speaker_id": s.get("speaker", DEFAULT_SPEAKER)} for s in kept_segments],
|
||
enriched,
|
||
)
|
||
duration = sum(s["end"] - s["start"] for s in rows)
|
||
return {
|
||
**timeline,
|
||
"summary": _summary(enriched, rows, timeline.get("speakers", []),
|
||
duration, peak_percentile, emphasis_floor),
|
||
"segments": rows,
|
||
}
|
||
|
||
|
||
def sentence_end(segments: Sequence[dict], index: int) -> float:
|
||
"""Where the sentence starting at ``segments[index]`` actually finishes.
|
||
|
||
Transcription segments break on breath and timing, not on grammar — a
|
||
sentence routinely spans two or three of them ("…que dá aquele ar" /
|
||
"de elegância, isso é desejo de muitas mulheres, né?"). A zoom that
|
||
ends on a segment boundary would therefore release mid-thought, so the
|
||
window is extended until a segment closes with terminal punctuation.
|
||
"""
|
||
last = float(segments[index]["end"])
|
||
for offset, segment in enumerate(segments[index:]):
|
||
# A long gap means the take stopped; never run a zoom across that.
|
||
# Checked before adopting the end, or the boundary segment's own
|
||
# end would already have been taken.
|
||
if offset > 0 and segment.get("take_boundary"):
|
||
break
|
||
last = float(segment["end"])
|
||
if (segment.get("text") or "").strip().endswith((".", "!", "?", "…")):
|
||
break
|
||
return last
|
||
|
||
|
||
def suggest_zoom_windows(
|
||
timeline: dict,
|
||
min_gap: float = 8.0,
|
||
max_zooms: Optional[int] = None,
|
||
) -> List[dict]:
|
||
"""Propose punch-in windows over a timeline's strongest lines.
|
||
|
||
One zoom per line at most, taken from the line's most emphatic
|
||
*content* word — a loudly spoken "a" is still an article, so function
|
||
words are skipped. The window runs from that word to the end of its
|
||
line, which is the shape the edit wants: the move lands with the word
|
||
and holds through the rest of the phrase.
|
||
|
||
``min_gap`` keeps successive zooms apart; effects stacked close
|
||
together read as nervous editing rather than emphasis.
|
||
"""
|
||
segments = timeline.get("segments", [])
|
||
candidates: List[dict] = []
|
||
for i, segment in enumerate(segments):
|
||
content = [
|
||
w for w in segment.get("words", [])
|
||
if w["text"].strip(",.!?;:").lower() not in _FUNCTION_WORDS
|
||
]
|
||
if not content:
|
||
continue
|
||
best = max(content, key=lambda w: w["emphasis"])
|
||
candidates.append({
|
||
"start": best["start"],
|
||
# Hold through to the end of the sentence, not of the segment —
|
||
# releasing mid-thought is what makes a punch-in feel arbitrary.
|
||
"end": sentence_end(segments, i),
|
||
"word": best["text"],
|
||
"emphasis": best["emphasis"],
|
||
"line": segment["text"],
|
||
})
|
||
|
||
chosen: List[dict] = []
|
||
for candidate in sorted(candidates, key=lambda c: c["emphasis"], reverse=True):
|
||
if max_zooms is not None and len(chosen) >= max_zooms:
|
||
break
|
||
if any(abs(candidate["start"] - c["start"]) < min_gap for c in chosen):
|
||
continue
|
||
chosen.append(candidate)
|
||
return sorted(chosen, key=lambda c: c["start"])
|
||
|
||
|
||
def speaker_profiles(segments: Sequence[dict], duration: float) -> List[dict]:
|
||
"""Per-speaker statistics and sample lines, so a person can tell who is who.
|
||
|
||
A bare ``SPEAKER_00`` label is useless for deciding whose audio to cut.
|
||
What identifies a role is *how* someone participates: an interviewer or
|
||
a crew member asks short questions and holds little of the runtime,
|
||
while the subject speaks in long stretches. ``avg_segment`` and
|
||
``share`` capture exactly that contrast, and the sample lines confirm
|
||
it in the person's own words.
|
||
"""
|
||
by_speaker: dict = {}
|
||
for seg in segments:
|
||
sid = seg.get("speaker", seg.get("speaker_id", DEFAULT_SPEAKER))
|
||
length = max(0.0, float(seg.get("end", 0.0)) - float(seg.get("start", 0.0)))
|
||
entry = by_speaker.setdefault(sid, {"seconds": 0.0, "segments": [], "words": 0})
|
||
entry["seconds"] += length
|
||
entry["words"] += len(seg.get("words", []))
|
||
entry["segments"].append(seg)
|
||
|
||
profiles: List[dict] = []
|
||
for i, (sid, entry) in enumerate(
|
||
sorted(by_speaker.items(), key=lambda kv: kv[1]["seconds"], reverse=True)
|
||
):
|
||
count = len(entry["segments"])
|
||
# Longest lines identify a role far better than the first ones: a
|
||
# question and an answer look alike at the start of a recording.
|
||
longest = sorted(
|
||
entry["segments"],
|
||
key=lambda s: float(s.get("end", 0)) - float(s.get("start", 0)),
|
||
reverse=True,
|
||
)[:3]
|
||
profiles.append({
|
||
"id": sid,
|
||
"name": f"Speaker {i + 1}",
|
||
"speaking_seconds": round(entry["seconds"], 2),
|
||
"share": round(entry["seconds"] / duration, 3) if duration > 0 else 0.0,
|
||
"segment_count": count,
|
||
"avg_segment": round(entry["seconds"] / count, 2) if count else 0.0,
|
||
"word_count": entry["words"],
|
||
"samples": [(s.get("text") or "").strip()[:160] for s in longest],
|
||
})
|
||
return profiles
|
||
|
||
|
||
def select_peaks(
|
||
words: Sequence[dict], percentile: float, floor: float
|
||
) -> List[dict]:
|
||
"""The most emphatic words: the top ``percentile`` fraction, above ``floor``.
|
||
|
||
Selection is relative on purpose. The emphasis index is a weighted
|
||
average whose real range depends entirely on the material — a measured
|
||
interview peaks around 0.5 while an energetic ad reaches much higher —
|
||
so any fixed cutoff either floods one and selects nothing in the other.
|
||
Asking for "the top 2%" instead yields a usable handful either way.
|
||
|
||
``floor`` is only a sanity guard for genuinely flat audio, where even
|
||
the top of the distribution carries no emphasis worth cutting on.
|
||
"""
|
||
ranked = sorted(words, key=lambda w: w["emphasis"], reverse=True)
|
||
keep = max(1, round(len(ranked) * percentile)) if ranked else 0
|
||
return [w for w in ranked[:keep] if w["emphasis"] >= floor]
|
||
|
||
|
||
def _summary(words: Sequence[dict], segments: Sequence[dict], speakers: Sequence[dict],
|
||
duration: float, peak_percentile: float, emphasis_floor: float) -> dict:
|
||
"""The top layer: the shape of the piece in a handful of numbers."""
|
||
emphases = [w["emphasis"] for w in words]
|
||
peaks = select_peaks(words, peak_percentile, emphasis_floor)
|
||
return {
|
||
"duration": round(duration, 3),
|
||
"speaker_count": len(speakers),
|
||
"segment_count": len(segments),
|
||
"word_count": len(words),
|
||
"avg_emphasis": round(sum(emphases) / len(emphases), 3) if emphases else 0.0,
|
||
"peak_selection": f"top {peak_percentile:.0%} of words, minimum emphasis {emphasis_floor:.2f}",
|
||
"peak_count": len(peaks),
|
||
"peak_moments": [
|
||
{
|
||
"time": round(float(w.get("start", 0.0)), 3),
|
||
"text": w.get("word", ""),
|
||
"speaker": w.get("speaker_id", DEFAULT_SPEAKER),
|
||
"emphasis": round(w["emphasis"], 3),
|
||
}
|
||
for w in sorted(peaks, key=lambda w: w["emphasis"], reverse=True)[:20]
|
||
],
|
||
}
|
||
|
||
|
||
def build_voice_timeline(
|
||
media_path: str,
|
||
transcript: dict,
|
||
hf_token: Optional[str] = None,
|
||
num_speakers: str = "",
|
||
weights: EmphasisWeights = EmphasisWeights(),
|
||
peak_percentile: float = 0.02,
|
||
emphasis_floor: float = 0.25,
|
||
progress_cb: Optional[Callable[[float, str], None]] = None,
|
||
) -> dict:
|
||
"""Build the consolidated voice timeline for one media file.
|
||
|
||
Every analysis layer is optional and degrades independently: without
|
||
librosa the acoustic values are ``0.0``; without a diarization token
|
||
every word belongs to ``SPEAKER_00``. The document's shape never
|
||
changes, so downstream consumers (the rules engine, or a model reading
|
||
the JSON) can rely on it.
|
||
"""
|
||
def report(fraction: float, stage: str) -> None:
|
||
if progress_cb:
|
||
progress_cb(fraction, stage)
|
||
|
||
report(0.1, "Analisando tom e energia...")
|
||
pitch_track = extract_pitch(media_path)
|
||
energy_track = extract_energy(media_path)
|
||
|
||
report(0.5, "Calculando ênfase...")
|
||
words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights)
|
||
|
||
report(0.7, "Identificando participantes...")
|
||
tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None
|
||
segments, words = assign_speakers(transcript.get("segments", []), words, tracks)
|
||
speakers = build_speakers(segments)
|
||
|
||
report(0.9, "Montando linha do tempo...")
|
||
duration = float(transcript.get("duration", 0.0))
|
||
segment_rows = _segment_rows(segments, words)
|
||
return {
|
||
"version": VOICE_TIMELINE_VERSION,
|
||
"source": Path(media_path).name,
|
||
"language": transcript.get("language", ""),
|
||
# What actually ran, not what was installed — a consumer must be able
|
||
# to tell "this speech is flat" from "the acoustics never loaded",
|
||
# since both leave the same zeros in the data.
|
||
"layers": {
|
||
"transcript": bool(transcript.get("words")),
|
||
"acoustics": pitch_track is not None or energy_track is not None,
|
||
"speakers": tracks is not None,
|
||
},
|
||
"scales": VALUE_SCALES,
|
||
"summary": _summary(
|
||
words, segments, speakers, duration, peak_percentile, emphasis_floor
|
||
),
|
||
"speakers": speaker_profiles(segment_rows, duration),
|
||
"segments": segment_rows,
|
||
}
|
||
|
||
|
||
def voice_timeline_path(media_path: str, output_dir: Optional[str] = None) -> Path:
|
||
"""Where the ``_voice_timeline.json`` for ``media_path`` lives.
|
||
|
||
Mirrors ``_transcript.json``: next to the media, or in the chosen
|
||
project folder when one is set.
|
||
"""
|
||
p = Path(media_path)
|
||
if output_dir:
|
||
directory = Path(output_dir).expanduser()
|
||
directory.mkdir(parents=True, exist_ok=True)
|
||
return directory / f"{p.stem}_voice_timeline.json"
|
||
return p.with_name(p.stem + "_voice_timeline.json")
|
||
|
||
|
||
def save_voice_timeline(timeline: dict, path: Path) -> None:
|
||
"""Write the timeline as UTF-8 JSON (accented transcripts stay readable)."""
|
||
with open(path, "w", encoding="utf-8") as f:
|
||
json.dump(timeline, f, ensure_ascii=False, indent=2)
|
||
|
||
|
||
def load_voice_timeline(path: Path) -> Optional[dict]:
|
||
"""Read a cached voice timeline, or ``None`` when absent/unreadable."""
|
||
try:
|
||
with open(path, encoding="utf-8") as f:
|
||
data = json.load(f)
|
||
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
||
return None
|
||
return data if isinstance(data, dict) and "segments" in data else None
|