"""Voice timeline — the consolidated, AI-readable view of how a video is spoken. This is the *source of truth* between analysis and editing: it merges what was said (transcript), who said it (diarization), and how it was said (pitch/energy/rate/pauses → emphasis) into one JSON document, decoupling the audio analysis from FCPXML generation entirely. The shape is designed to be handed to a language model so it can reason about the narrative — which beats carry weight, where a speaker changes, where the delivery peaks — and decide how to direct the edit. Two design choices serve that goal: * **Layered, not flat.** A ``summary`` gives the whole picture in a few numbers, ``segments`` group words into utterances with their own aggregates, and ``words`` hold the fine detail. A model can reason from the top layer and only descend where it matters, instead of parsing thousands of word rows to find the shape of the piece. * **Normalized, self-describing values.** Every acoustic value is 0–1 and relative to *this* recording (a quiet podcast and a shouted ad both use the full range), and ``scales`` documents that contract inline, so the numbers are interpretable without external context. """ import json import logging from pathlib import Path from typing import Callable, List, Optional, Sequence, Tuple from .diarize import DEFAULT_SPEAKER, assign_speakers, build_speakers, diarize from .emphasis import EmphasisWeights, annotate_emphasis from .voice_features import ( compute_pauses, compute_speech_rate, extract_energy, extract_pitch, word_pitch_energy, ) logger = logging.getLogger(__name__) VOICE_TIMELINE_VERSION = "1.0" # Silence long enough to mean the take stopped rather than the speaker paused. # On real footage, boundaries between retakes showed gaps of 3.6-19.8s while # dramatic beats inside a delivered line stayed under ~2s. TAKE_BOUNDARY_GAP = 3.0 # How to read the values in this document, split by the level they live on. # Embedded in the output so a model consuming the JSON needs no external # documentation — and kept honest: a metric listed under "word" must exist on # every word row, and one under "segment" on every segment row. VALUE_SCALES = { "word": { "energy": "0-1, loudness relative to the loudest moment of this recording", "pitch_delta": "0-1, how far this word's pitch sits from the speaker's average", "rate_delta": "0-1, how much the local speaking rate departs from the average", "pause_before": "seconds of silence immediately before the word", "emphasis": "0-1 combined index; high values are punch-in/highlight candidates", "emotion": "heuristic label from delivery: neutral, excited, tense, calm, reflective", "emotion_confidence": "0-1 confidence in the heuristic emotion label", "arousal": "0-1 vocal activation from energy/rate/pitch movement", "valence": "0-1 rough positive tone; lower values suggest tension/weight", }, "segment": { "gap_before": "seconds of silence before this line", "take_boundary": "true when the gap is long enough that the take likely restarted here", "avg_energy": "0-1 mean loudness across the line", "peak_emphasis": "0-1 highest emphasis of any word in the line", "emotion": "dominant delivery emotion across the line", "emotion_confidence": "0-1 confidence in the dominant segment emotion", "arousal": "0-1 mean vocal activation across the line", "valence": "0-1 mean rough positive tone across the line", }, } def _normalize(value: Optional[float], maximum: float) -> float: """Scale ``value`` into 0-1 against ``maximum`` (0.0 when unavailable).""" if value is None or maximum <= 0: return 0.0 return max(0.0, min(1.0, value / maximum)) def _round_word(word: dict) -> dict: """One word row, rounded to a size a model can read without noise. The raw ``energy_raw``/``pitch_hz`` ride along beside the normalized values so the document can be re-analyzed over a subset later. That matters after cutting: every normalized value is relative to the loudest moment of the *whole* recording, and if that moment gets cut the survivors are scored against something that no longer exists. """ return { "text": word.get("word", ""), "start": round(float(word.get("start", 0.0)), 3), "end": round(float(word.get("end", 0.0)), 3), "speaker": word.get("speaker_id", DEFAULT_SPEAKER), "energy": round(word.get("energy_norm", 0.0), 3), "pitch_delta": round(word.get("pitch_delta", 0.0), 3), "rate_delta": round(word.get("rate_delta", 0.0), 3), "pause_before": round(word.get("pause_before", 0.0), 3), "emphasis": round(word.get("emphasis", 0.0), 3), "emotion": word.get("emotion", "neutral"), "emotion_confidence": round(word.get("emotion_confidence", 0.0), 3), "arousal": round(word.get("arousal", 0.0), 3), "valence": round(word.get("valence", 0.5), 3), "energy_raw": word.get("energy"), "pitch_hz": word.get("pitch_hz"), } def _emotion_for_word(word: dict, enabled: bool, sensitivity: float) -> dict: """Classify delivery emotion from normalized acoustic features. This is deliberately a local heuristic rather than a claimed clinical emotion model. It gives the editor a useful signal about delivery shape while degrading predictably when acoustic extraction is unavailable. """ if not enabled: return { "emotion": "neutral", "emotion_confidence": 0.0, "arousal": 0.0, "valence": 0.5, } energy = float(word.get("energy_norm", 0.0)) pitch = float(word.get("pitch_delta", 0.0)) rate = float(word.get("rate_delta", 0.0)) pause = min(float(word.get("pause_before", 0.0)) / 2.0, 1.0) emphasis = float(word.get("emphasis", 0.0)) arousal = max(0.0, min(1.0, energy * 0.45 + pitch * 0.25 + rate * 0.20 + emphasis * 0.10)) valence = max(0.0, min(1.0, 0.55 + energy * 0.15 - pause * 0.20 - rate * 0.10)) if arousal >= 0.68 and valence >= 0.50: label = "excited" confidence = arousal elif arousal >= 0.58 and valence < 0.50: label = "tense" confidence = max(arousal, 1.0 - valence) elif arousal <= 0.28 and pause >= 0.25: label = "reflective" confidence = max(1.0 - arousal, pause) elif arousal <= 0.35: label = "calm" confidence = 1.0 - arousal else: label = "neutral" confidence = 1.0 - abs(arousal - 0.5) * 2.0 confidence = max(0.0, min(1.0, confidence)) if confidence < sensitivity: label = "neutral" return { "emotion": label, "emotion_confidence": confidence, "arousal": arousal, "valence": valence, } def annotate_emotions(words: Sequence[dict], enabled: bool, sensitivity: float) -> List[dict]: """Attach heuristic emotion labels to enriched word rows.""" return [ {**w, **_emotion_for_word(w, enabled, sensitivity)} for w in words ] def enrich_words( words: Sequence[dict], pitch_track: Optional[Sequence] = None, energy_track: Optional[Sequence] = None, weights: EmphasisWeights = EmphasisWeights(), already_measured: bool = False, ) -> List[dict]: """Attach normalized acoustic features + the emphasis index to each word. Normalization is per-recording: energy against the loudest word, pitch against the spread around this recording's average, rate against the largest local departure. That makes the numbers comparable within a piece regardless of how it was recorded. Set ``already_measured`` when the words already carry ``energy`` and ``pitch_hz`` from a previous pass — re-analyzing a subset, say. The frame tracks are then unnecessary, and sampling them again would overwrite good values with ``None``. """ if not words: return [] if not already_measured: words = word_pitch_energy(words, pitch_track, energy_track) rates = compute_speech_rate(words) pauses = compute_pauses(words) energies = [w["energy"] for w in words if w.get("energy") is not None] max_energy = max(energies) if energies else 0.0 pitches = [w["pitch_hz"] for w in words if w.get("pitch_hz") is not None] avg_pitch = sum(pitches) / len(pitches) if pitches else 0.0 pitch_span = (max(pitches) - min(pitches)) if len(pitches) > 1 else 0.0 avg_rate = sum(rates) / len(rates) if rates else 0.0 max_rate = max(rates) if rates else 0.0 enriched: List[dict] = [] for i, w in enumerate(words): ww = dict(w) ww["energy_norm"] = _normalize(w.get("energy"), max_energy) pitch = w.get("pitch_hz") ww["pitch_delta"] = ( _normalize(abs(pitch - avg_pitch), pitch_span) if pitch is not None else 0.0 ) ww["rate_delta"] = _normalize(abs(rates[i] - avg_rate), max_rate) ww["pause_before"] = pauses[i] enriched.append(ww) annotated = annotate_emphasis( [{**w, "energy": w["energy_norm"]} for w in enriched], weights=weights ) for word, scored in zip(enriched, annotated): word["emphasis"] = scored["emphasis"] return enriched def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]: """Group enriched words under their segment, with per-segment aggregates. The aggregates are what let a model judge a whole utterance ("this line is delivered hot, that one trails off") without reading every word. """ rows: List[dict] = [] previous_end = 0.0 for seg in segments: start = float(seg.get("start", 0.0)) end = float(seg.get("end", 0.0)) in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end] energies = [w["energy_norm"] for w in in_seg] emphases = [w["emphasis"] for w in in_seg] arousals = [w.get("arousal", 0.0) for w in in_seg] valences = [w.get("valence", 0.5) for w in in_seg] emotions = [w.get("emotion", "neutral") for w in in_seg] dominant = max(set(emotions), key=emotions.count) if emotions else "neutral" emotion_confidences = [ w.get("emotion_confidence", 0.0) for w in in_seg if w.get("emotion") == dominant ] gap = max(0.0, start - previous_end) rows.append( { "start": round(start, 3), "end": round(end, 3), "speaker": seg.get("speaker_id", DEFAULT_SPEAKER), "text": (seg.get("text") or "").strip(), # Silence before this line. Long gaps are where the camera # stopped or the take restarted, so this is the structural # hint for splitting a recording into takes — the same signal # that is *noise* for emphasis (see emphasis.pause_weight). "gap_before": round(gap, 3), "take_boundary": gap >= TAKE_BOUNDARY_GAP, "avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0, "peak_emphasis": round(max(emphases), 3) if emphases else 0.0, "emotion": dominant, "emotion_confidence": ( round(sum(emotion_confidences) / len(emotion_confidences), 3) if emotion_confidences else 0.0 ), "arousal": round(sum(arousals) / len(arousals), 3) if arousals else 0.0, "valence": round(sum(valences) / len(valences), 3) if valences else 0.5, "words": [_round_word(w) for w in in_seg], } ) previous_end = end return rows # Words too common to ever be the point of a punch-in. A zoom lands on what a # sentence is *about*, and an article spoken loudly is still an article. _FUNCTION_WORDS = { "a", "o", "e", "de", "da", "do", "que", "é", "em", "um", "uma", "as", "os", "no", "na", "com", "pra", "para", "por", "se", "mais", "isso", "aí", "tudo", "ao", "à", "dos", "das", "nos", "nas", "ou", "mas", "já", "ele", "ela", "eu", "você", "seu", "sua", "meu", "minha", "esse", "essa", "aquele", } def _survives(start: float, end: float, cuts: Sequence[Tuple[float, float]]) -> bool: """Whether a span lies entirely outside every removed range.""" return all(end <= cut_start or start >= cut_end for cut_start, cut_end in cuts) def restrict_to_kept( timeline: dict, cut_ranges: Sequence[Tuple[float, float]], weights: EmphasisWeights = EmphasisWeights(), peak_percentile: float = 0.02, emphasis_floor: float = 0.25, ) -> dict: """Re-analyze a timeline over only the material that survives ``cut_ranges``. Emphasis is *relative*: energy is scored against the loudest word, pitch against the spread of the recording. Cut the loudest moment out — a laugh, an aside to the crew — and every remaining score is measured against something the viewer will never see. Re-running the normalization over just the survivors is what makes "the most emphatic line of the final video" a meaningful question. Returns a timeline of the same shape, with times still in original source seconds so the result can be fed straight back as actions. """ kept_words = [ w for segment in timeline.get("segments", []) for w in segment.get("words", []) if _survives(w["start"], w["end"], cut_ranges) ] # enrich_words expects the raw analysis keys, not the normalized ones. raw = [ { "word": w["text"], "start": w["start"], "end": w["end"], "speaker_id": w.get("speaker", DEFAULT_SPEAKER), "energy": w.get("energy_raw"), "pitch_hz": w.get("pitch_hz"), } for w in kept_words ] enriched = enrich_words(raw, weights=weights, already_measured=True) kept_segments = [ {**s, "words": [w for w in s.get("words", []) if _survives(w["start"], w["end"], cut_ranges)]} for s in timeline.get("segments", []) ] kept_segments = [s for s in kept_segments if s["words"]] rows = _segment_rows( [{"text": s["text"], "start": s["start"], "end": s["end"], "speaker_id": s.get("speaker", DEFAULT_SPEAKER)} for s in kept_segments], enriched, ) duration = sum(s["end"] - s["start"] for s in rows) return { **timeline, "summary": _summary(enriched, rows, timeline.get("speakers", []), duration, peak_percentile, emphasis_floor), "segments": rows, } def sentence_end(segments: Sequence[dict], index: int) -> float: """Where the sentence starting at ``segments[index]`` actually finishes. Transcription segments break on breath and timing, not on grammar — a sentence routinely spans two or three of them ("…que dá aquele ar" / "de elegância, isso é desejo de muitas mulheres, né?"). A zoom that ends on a segment boundary would therefore release mid-thought, so the window is extended until a segment closes with terminal punctuation. """ last = float(segments[index]["end"]) for offset, segment in enumerate(segments[index:]): # A long gap means the take stopped; never run a zoom across that. # Checked before adopting the end, or the boundary segment's own # end would already have been taken. if offset > 0 and segment.get("take_boundary"): break last = float(segment["end"]) if (segment.get("text") or "").strip().endswith((".", "!", "?", "…")): break return last def suggest_zoom_windows( timeline: dict, min_gap: float = 8.0, max_zooms: Optional[int] = None, ) -> List[dict]: """Propose punch-in windows over a timeline's strongest lines. One zoom per line at most, taken from the line's most emphatic *content* word — a loudly spoken "a" is still an article, so function words are skipped. The window runs from that word to the end of its line, which is the shape the edit wants: the move lands with the word and holds through the rest of the phrase. ``min_gap`` keeps successive zooms apart; effects stacked close together read as nervous editing rather than emphasis. """ segments = timeline.get("segments", []) candidates: List[dict] = [] for i, segment in enumerate(segments): content = [ w for w in segment.get("words", []) if w["text"].strip(",.!?;:").lower() not in _FUNCTION_WORDS ] if not content: continue best = max(content, key=lambda w: w["emphasis"]) candidates.append({ "start": best["start"], # Hold through to the end of the sentence, not of the segment — # releasing mid-thought is what makes a punch-in feel arbitrary. "end": sentence_end(segments, i), "word": best["text"], "emphasis": best["emphasis"], "line": segment["text"], }) chosen: List[dict] = [] for candidate in sorted(candidates, key=lambda c: c["emphasis"], reverse=True): if max_zooms is not None and len(chosen) >= max_zooms: break if any(abs(candidate["start"] - c["start"]) < min_gap for c in chosen): continue chosen.append(candidate) return sorted(chosen, key=lambda c: c["start"]) def speaker_profiles(segments: Sequence[dict], duration: float) -> List[dict]: """Per-speaker statistics and sample lines, so a person can tell who is who. A bare ``SPEAKER_00`` label is useless for deciding whose audio to cut. What identifies a role is *how* someone participates: an interviewer or a crew member asks short questions and holds little of the runtime, while the subject speaks in long stretches. ``avg_segment`` and ``share`` capture exactly that contrast, and the sample lines confirm it in the person's own words. """ by_speaker: dict = {} for seg in segments: sid = seg.get("speaker", seg.get("speaker_id", DEFAULT_SPEAKER)) length = max(0.0, float(seg.get("end", 0.0)) - float(seg.get("start", 0.0))) entry = by_speaker.setdefault(sid, {"seconds": 0.0, "segments": [], "words": 0}) entry["seconds"] += length entry["words"] += len(seg.get("words", [])) entry["segments"].append(seg) profiles: List[dict] = [] for i, (sid, entry) in enumerate( sorted(by_speaker.items(), key=lambda kv: kv[1]["seconds"], reverse=True) ): count = len(entry["segments"]) # Longest lines identify a role far better than the first ones: a # question and an answer look alike at the start of a recording. longest = sorted( entry["segments"], key=lambda s: float(s.get("end", 0)) - float(s.get("start", 0)), reverse=True, )[:3] profiles.append({ "id": sid, "name": f"Speaker {i + 1}", "speaking_seconds": round(entry["seconds"], 2), "share": round(entry["seconds"] / duration, 3) if duration > 0 else 0.0, "segment_count": count, "avg_segment": round(entry["seconds"] / count, 2) if count else 0.0, "word_count": entry["words"], "samples": [(s.get("text") or "").strip()[:160] for s in longest], }) return profiles def select_peaks( words: Sequence[dict], percentile: float, floor: float ) -> List[dict]: """The most emphatic words: the top ``percentile`` fraction, above ``floor``. Selection is relative on purpose. The emphasis index is a weighted average whose real range depends entirely on the material — a measured interview peaks around 0.5 while an energetic ad reaches much higher — so any fixed cutoff either floods one and selects nothing in the other. Asking for "the top 2%" instead yields a usable handful either way. ``floor`` is only a sanity guard for genuinely flat audio, where even the top of the distribution carries no emphasis worth cutting on. """ ranked = sorted(words, key=lambda w: w["emphasis"], reverse=True) keep = max(1, round(len(ranked) * percentile)) if ranked else 0 return [w for w in ranked[:keep] if w["emphasis"] >= floor] def _summary(words: Sequence[dict], segments: Sequence[dict], speakers: Sequence[dict], duration: float, peak_percentile: float, emphasis_floor: float) -> dict: """The top layer: the shape of the piece in a handful of numbers.""" emphases = [w["emphasis"] for w in words] peaks = select_peaks(words, peak_percentile, emphasis_floor) return { "duration": round(duration, 3), "speaker_count": len(speakers), "segment_count": len(segments), "word_count": len(words), "avg_emphasis": round(sum(emphases) / len(emphases), 3) if emphases else 0.0, "peak_selection": f"top {peak_percentile:.0%} of words, minimum emphasis {emphasis_floor:.2f}", "peak_count": len(peaks), "peak_moments": [ { "time": round(float(w.get("start", 0.0)), 3), "text": w.get("word", ""), "speaker": w.get("speaker_id", DEFAULT_SPEAKER), "emphasis": round(w["emphasis"], 3), } for w in sorted(peaks, key=lambda w: w["emphasis"], reverse=True)[:20] ], } def build_voice_timeline( media_path: str, transcript: dict, hf_token: Optional[str] = None, num_speakers: str = "", weights: EmphasisWeights = EmphasisWeights(), peak_percentile: float = 0.02, emphasis_floor: float = 0.25, emotion_enabled: bool = False, emotion_sensitivity: float = 0.5, rotation: float = 0.0, progress_cb: Optional[Callable[[float, str], None]] = None, ) -> dict: """Build the consolidated voice timeline for one media file. Every analysis layer is optional and degrades independently: without librosa the acoustic values are ``0.0``; without a diarization token every word belongs to ``SPEAKER_00``. The document's shape never changes, so downstream consumers (the rules engine, or a model reading the JSON) can rely on it. """ def report(fraction: float, stage: str) -> None: if progress_cb: progress_cb(fraction, stage) report(0.1, "Analisando tom e energia...") pitch_track = extract_pitch(media_path) energy_track = extract_energy(media_path) report(0.5, "Calculando ênfase...") words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights) words = annotate_emotions(words, emotion_enabled, emotion_sensitivity) report(0.7, "Identificando participantes...") tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None segments, words = assign_speakers(transcript.get("segments", []), words, tracks) speakers = build_speakers(segments) report(0.9, "Montando linha do tempo...") duration = float(transcript.get("duration", 0.0)) segment_rows = _segment_rows(segments, words) return { "version": VOICE_TIMELINE_VERSION, "source": Path(media_path).name, # Edit-time correction from the clip's Transform filter in the FCPXML # (e.g. straightening a tilted phone shot) — 0.0 when the clip has none # or the caller didn't resolve one. "rotation": rotation, "language": transcript.get("language", ""), # What actually ran, not what was installed — a consumer must be able # to tell "this speech is flat" from "the acoustics never loaded", # since both leave the same zeros in the data. "layers": { "transcript": bool(transcript.get("words")), "acoustics": pitch_track is not None or energy_track is not None, "speakers": tracks is not None, "emotion": bool(emotion_enabled), "alignment": bool(transcript.get("alignment")), }, "scales": VALUE_SCALES, "summary": _summary( words, segments, speakers, duration, peak_percentile, emphasis_floor ), "speakers": speaker_profiles(segment_rows, duration), "segments": segment_rows, } def voice_timeline_path(media_path: str, output_dir: Optional[str] = None) -> Path: """Where the ``_voice_timeline.json`` for ``media_path`` lives. Mirrors ``_transcript.json``: next to the media, or in the chosen project folder when one is set. """ p = Path(media_path) if output_dir: directory = Path(output_dir).expanduser() directory.mkdir(parents=True, exist_ok=True) return directory / f"{p.stem}_voice_timeline.json" return p.with_name(p.stem + "_voice_timeline.json") def save_voice_timeline(timeline: dict, path: Path) -> None: """Write the timeline as UTF-8 JSON (accented transcripts stay readable).""" with open(path, "w", encoding="utf-8") as f: json.dump(timeline, f, ensure_ascii=False, indent=2) def load_voice_timeline(path: Path) -> Optional[dict]: """Read a cached voice timeline, or ``None`` when absent/unreadable.""" try: with open(path, encoding="utf-8") as f: data = json.load(f) except (OSError, json.JSONDecodeError, UnicodeDecodeError): return None return data if isinstance(data, dict) and "segments" in data else None