Files
gart/code/fcpxml/voice_timeline.py
T
João HenriqueandClaude Sonnet 5 7b5aed79ee feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-21 18:26:04 -04:00

601 lines
25 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Voice timeline — the consolidated, AI-readable view of how a video is spoken.
This is the *source of truth* between analysis and editing: it merges what
was said (transcript), who said it (diarization), and how it was said
(pitch/energy/rate/pauses → emphasis) into one JSON document, decoupling the
audio analysis from FCPXML generation entirely.
The shape is designed to be handed to a language model so it can reason about
the narrative — which beats carry weight, where a speaker changes, where the
delivery peaks — and decide how to direct the edit. Two design choices serve
that goal:
* **Layered, not flat.** A ``summary`` gives the whole picture in a few
numbers, ``segments`` group words into utterances with their own
aggregates, and ``words`` hold the fine detail. A model can reason from
the top layer and only descend where it matters, instead of parsing
thousands of word rows to find the shape of the piece.
* **Normalized, self-describing values.** Every acoustic value is 0–1 and
relative to *this* recording (a quiet podcast and a shouted ad both use
the full range), and ``scales`` documents that contract inline, so the
numbers are interpretable without external context.
"""
import json
import logging
from pathlib import Path
from typing import Callable, List, Optional, Sequence, Tuple
from .diarize import DEFAULT_SPEAKER, assign_speakers, build_speakers, diarize
from .emphasis import EmphasisWeights, annotate_emphasis
from .voice_features import (
compute_pauses,
compute_speech_rate,
extract_energy,
extract_pitch,
word_pitch_energy,
)
logger = logging.getLogger(__name__)
VOICE_TIMELINE_VERSION = "1.0"
# Silence long enough to mean the take stopped rather than the speaker paused.
# On real footage, boundaries between retakes showed gaps of 3.6-19.8s while
# dramatic beats inside a delivered line stayed under ~2s.
TAKE_BOUNDARY_GAP = 3.0
# How to read the values in this document, split by the level they live on.
# Embedded in the output so a model consuming the JSON needs no external
# documentation — and kept honest: a metric listed under "word" must exist on
# every word row, and one under "segment" on every segment row.
VALUE_SCALES = {
"word": {
"energy": "0-1, loudness relative to the loudest moment of this recording",
"pitch_delta": "0-1, how far this word's pitch sits from the speaker's average",
"rate_delta": "0-1, how much the local speaking rate departs from the average",
"pause_before": "seconds of silence immediately before the word",
"emphasis": "0-1 combined index; high values are punch-in/highlight candidates",
"emotion": "heuristic label from delivery: neutral, excited, tense, calm, reflective",
"emotion_confidence": "0-1 confidence in the heuristic emotion label",
"arousal": "0-1 vocal activation from energy/rate/pitch movement",
"valence": "0-1 rough positive tone; lower values suggest tension/weight",
},
"segment": {
"gap_before": "seconds of silence before this line",
"take_boundary": "true when the gap is long enough that the take likely restarted here",
"avg_energy": "0-1 mean loudness across the line",
"peak_emphasis": "0-1 highest emphasis of any word in the line",
"emotion": "dominant delivery emotion across the line",
"emotion_confidence": "0-1 confidence in the dominant segment emotion",
"arousal": "0-1 mean vocal activation across the line",
"valence": "0-1 mean rough positive tone across the line",
},
}
def _normalize(value: Optional[float], maximum: float) -> float:
"""Scale ``value`` into 0-1 against ``maximum`` (0.0 when unavailable)."""
if value is None or maximum <= 0:
return 0.0
return max(0.0, min(1.0, value / maximum))
def _round_word(word: dict) -> dict:
"""One word row, rounded to a size a model can read without noise.
The raw ``energy_raw``/``pitch_hz`` ride along beside the normalized
values so the document can be re-analyzed over a subset later. That
matters after cutting: every normalized value is relative to the
loudest moment of the *whole* recording, and if that moment gets cut
the survivors are scored against something that no longer exists.
"""
return {
"text": word.get("word", ""),
"start": round(float(word.get("start", 0.0)), 3),
"end": round(float(word.get("end", 0.0)), 3),
"speaker": word.get("speaker_id", DEFAULT_SPEAKER),
"energy": round(word.get("energy_norm", 0.0), 3),
"pitch_delta": round(word.get("pitch_delta", 0.0), 3),
"rate_delta": round(word.get("rate_delta", 0.0), 3),
"pause_before": round(word.get("pause_before", 0.0), 3),
"emphasis": round(word.get("emphasis", 0.0), 3),
"emotion": word.get("emotion", "neutral"),
"emotion_confidence": round(word.get("emotion_confidence", 0.0), 3),
"arousal": round(word.get("arousal", 0.0), 3),
"valence": round(word.get("valence", 0.5), 3),
"energy_raw": word.get("energy"),
"pitch_hz": word.get("pitch_hz"),
}
def _emotion_for_word(word: dict, enabled: bool, sensitivity: float) -> dict:
"""Classify delivery emotion from normalized acoustic features.
This is deliberately a local heuristic rather than a claimed clinical
emotion model. It gives the editor a useful signal about delivery shape
while degrading predictably when acoustic extraction is unavailable.
"""
if not enabled:
return {
"emotion": "neutral",
"emotion_confidence": 0.0,
"arousal": 0.0,
"valence": 0.5,
}
energy = float(word.get("energy_norm", 0.0))
pitch = float(word.get("pitch_delta", 0.0))
rate = float(word.get("rate_delta", 0.0))
pause = min(float(word.get("pause_before", 0.0)) / 2.0, 1.0)
emphasis = float(word.get("emphasis", 0.0))
arousal = max(0.0, min(1.0, energy * 0.45 + pitch * 0.25 + rate * 0.20 + emphasis * 0.10))
valence = max(0.0, min(1.0, 0.55 + energy * 0.15 - pause * 0.20 - rate * 0.10))
if arousal >= 0.68 and valence >= 0.50:
label = "excited"
confidence = arousal
elif arousal >= 0.58 and valence < 0.50:
label = "tense"
confidence = max(arousal, 1.0 - valence)
elif arousal <= 0.28 and pause >= 0.25:
label = "reflective"
confidence = max(1.0 - arousal, pause)
elif arousal <= 0.35:
label = "calm"
confidence = 1.0 - arousal
else:
label = "neutral"
confidence = 1.0 - abs(arousal - 0.5) * 2.0
confidence = max(0.0, min(1.0, confidence))
if confidence < sensitivity:
label = "neutral"
return {
"emotion": label,
"emotion_confidence": confidence,
"arousal": arousal,
"valence": valence,
}
def annotate_emotions(words: Sequence[dict], enabled: bool, sensitivity: float) -> List[dict]:
"""Attach heuristic emotion labels to enriched word rows."""
return [
{**w, **_emotion_for_word(w, enabled, sensitivity)}
for w in words
]
def enrich_words(
words: Sequence[dict],
pitch_track: Optional[Sequence] = None,
energy_track: Optional[Sequence] = None,
weights: EmphasisWeights = EmphasisWeights(),
already_measured: bool = False,
) -> List[dict]:
"""Attach normalized acoustic features + the emphasis index to each word.
Normalization is per-recording: energy against the loudest word, pitch
against the spread around this recording's average, rate against the
largest local departure. That makes the numbers comparable within a
piece regardless of how it was recorded.
Set ``already_measured`` when the words already carry ``energy`` and
``pitch_hz`` from a previous pass — re-analyzing a subset, say. The
frame tracks are then unnecessary, and sampling them again would
overwrite good values with ``None``.
"""
if not words:
return []
if not already_measured:
words = word_pitch_energy(words, pitch_track, energy_track)
rates = compute_speech_rate(words)
pauses = compute_pauses(words)
energies = [w["energy"] for w in words if w.get("energy") is not None]
max_energy = max(energies) if energies else 0.0
pitches = [w["pitch_hz"] for w in words if w.get("pitch_hz") is not None]
avg_pitch = sum(pitches) / len(pitches) if pitches else 0.0
pitch_span = (max(pitches) - min(pitches)) if len(pitches) > 1 else 0.0
avg_rate = sum(rates) / len(rates) if rates else 0.0
max_rate = max(rates) if rates else 0.0
enriched: List[dict] = []
for i, w in enumerate(words):
ww = dict(w)
ww["energy_norm"] = _normalize(w.get("energy"), max_energy)
pitch = w.get("pitch_hz")
ww["pitch_delta"] = (
_normalize(abs(pitch - avg_pitch), pitch_span) if pitch is not None else 0.0
)
ww["rate_delta"] = _normalize(abs(rates[i] - avg_rate), max_rate)
ww["pause_before"] = pauses[i]
enriched.append(ww)
annotated = annotate_emphasis(
[{**w, "energy": w["energy_norm"]} for w in enriched], weights=weights
)
for word, scored in zip(enriched, annotated):
word["emphasis"] = scored["emphasis"]
return enriched
def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]:
"""Group enriched words under their segment, with per-segment aggregates.
The aggregates are what let a model judge a whole utterance ("this line
is delivered hot, that one trails off") without reading every word.
"""
rows: List[dict] = []
previous_end = 0.0
for seg in segments:
start = float(seg.get("start", 0.0))
end = float(seg.get("end", 0.0))
in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end]
energies = [w["energy_norm"] for w in in_seg]
emphases = [w["emphasis"] for w in in_seg]
arousals = [w.get("arousal", 0.0) for w in in_seg]
valences = [w.get("valence", 0.5) for w in in_seg]
emotions = [w.get("emotion", "neutral") for w in in_seg]
dominant = max(set(emotions), key=emotions.count) if emotions else "neutral"
emotion_confidences = [
w.get("emotion_confidence", 0.0) for w in in_seg if w.get("emotion") == dominant
]
gap = max(0.0, start - previous_end)
rows.append(
{
"start": round(start, 3),
"end": round(end, 3),
"speaker": seg.get("speaker_id", DEFAULT_SPEAKER),
"text": (seg.get("text") or "").strip(),
# Silence before this line. Long gaps are where the camera
# stopped or the take restarted, so this is the structural
# hint for splitting a recording into takes — the same signal
# that is *noise* for emphasis (see emphasis.pause_weight).
"gap_before": round(gap, 3),
"take_boundary": gap >= TAKE_BOUNDARY_GAP,
"avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0,
"peak_emphasis": round(max(emphases), 3) if emphases else 0.0,
"emotion": dominant,
"emotion_confidence": (
round(sum(emotion_confidences) / len(emotion_confidences), 3)
if emotion_confidences else 0.0
),
"arousal": round(sum(arousals) / len(arousals), 3) if arousals else 0.0,
"valence": round(sum(valences) / len(valences), 3) if valences else 0.5,
"words": [_round_word(w) for w in in_seg],
}
)
previous_end = end
return rows
# Words too common to ever be the point of a punch-in. A zoom lands on what a
# sentence is *about*, and an article spoken loudly is still an article.
_FUNCTION_WORDS = {
"a", "o", "e", "de", "da", "do", "que", "é", "em", "um", "uma", "as", "os",
"no", "na", "com", "pra", "para", "por", "se", "mais", "isso", "aí", "tudo",
"ao", "à", "dos", "das", "nos", "nas", "ou", "mas", "já", "ele", "ela",
"eu", "você", "seu", "sua", "meu", "minha", "esse", "essa", "aquele",
}
def _survives(start: float, end: float, cuts: Sequence[Tuple[float, float]]) -> bool:
"""Whether a span lies entirely outside every removed range."""
return all(end <= cut_start or start >= cut_end for cut_start, cut_end in cuts)
def restrict_to_kept(
timeline: dict,
cut_ranges: Sequence[Tuple[float, float]],
weights: EmphasisWeights = EmphasisWeights(),
peak_percentile: float = 0.02,
emphasis_floor: float = 0.25,
) -> dict:
"""Re-analyze a timeline over only the material that survives ``cut_ranges``.
Emphasis is *relative*: energy is scored against the loudest word,
pitch against the spread of the recording. Cut the loudest moment out —
a laugh, an aside to the crew — and every remaining score is measured
against something the viewer will never see. Re-running the
normalization over just the survivors is what makes "the most emphatic
line of the final video" a meaningful question.
Returns a timeline of the same shape, with times still in original
source seconds so the result can be fed straight back as actions.
"""
kept_words = [
w
for segment in timeline.get("segments", [])
for w in segment.get("words", [])
if _survives(w["start"], w["end"], cut_ranges)
]
# enrich_words expects the raw analysis keys, not the normalized ones.
raw = [
{
"word": w["text"],
"start": w["start"],
"end": w["end"],
"speaker_id": w.get("speaker", DEFAULT_SPEAKER),
"energy": w.get("energy_raw"),
"pitch_hz": w.get("pitch_hz"),
}
for w in kept_words
]
enriched = enrich_words(raw, weights=weights, already_measured=True)
kept_segments = [
{**s, "words": [w for w in s.get("words", []) if _survives(w["start"], w["end"], cut_ranges)]}
for s in timeline.get("segments", [])
]
kept_segments = [s for s in kept_segments if s["words"]]
rows = _segment_rows(
[{"text": s["text"], "start": s["start"], "end": s["end"],
"speaker_id": s.get("speaker", DEFAULT_SPEAKER)} for s in kept_segments],
enriched,
)
duration = sum(s["end"] - s["start"] for s in rows)
return {
**timeline,
"summary": _summary(enriched, rows, timeline.get("speakers", []),
duration, peak_percentile, emphasis_floor),
"segments": rows,
}
def sentence_end(segments: Sequence[dict], index: int) -> float:
"""Where the sentence starting at ``segments[index]`` actually finishes.
Transcription segments break on breath and timing, not on grammar — a
sentence routinely spans two or three of them ("…que dá aquele ar" /
"de elegância, isso é desejo de muitas mulheres, né?"). A zoom that
ends on a segment boundary would therefore release mid-thought, so the
window is extended until a segment closes with terminal punctuation.
"""
last = float(segments[index]["end"])
for offset, segment in enumerate(segments[index:]):
# A long gap means the take stopped; never run a zoom across that.
# Checked before adopting the end, or the boundary segment's own
# end would already have been taken.
if offset > 0 and segment.get("take_boundary"):
break
last = float(segment["end"])
if (segment.get("text") or "").strip().endswith((".", "!", "?", "…")):
break
return last
def suggest_zoom_windows(
timeline: dict,
min_gap: float = 8.0,
max_zooms: Optional[int] = None,
) -> List[dict]:
"""Propose punch-in windows over a timeline's strongest lines.
One zoom per line at most, taken from the line's most emphatic
*content* word — a loudly spoken "a" is still an article, so function
words are skipped. The window runs from that word to the end of its
line, which is the shape the edit wants: the move lands with the word
and holds through the rest of the phrase.
``min_gap`` keeps successive zooms apart; effects stacked close
together read as nervous editing rather than emphasis.
"""
segments = timeline.get("segments", [])
candidates: List[dict] = []
for i, segment in enumerate(segments):
content = [
w for w in segment.get("words", [])
if w["text"].strip(",.!?;:").lower() not in _FUNCTION_WORDS
]
if not content:
continue
best = max(content, key=lambda w: w["emphasis"])
candidates.append({
"start": best["start"],
# Hold through to the end of the sentence, not of the segment —
# releasing mid-thought is what makes a punch-in feel arbitrary.
"end": sentence_end(segments, i),
"word": best["text"],
"emphasis": best["emphasis"],
"line": segment["text"],
})
chosen: List[dict] = []
for candidate in sorted(candidates, key=lambda c: c["emphasis"], reverse=True):
if max_zooms is not None and len(chosen) >= max_zooms:
break
if any(abs(candidate["start"] - c["start"]) < min_gap for c in chosen):
continue
chosen.append(candidate)
return sorted(chosen, key=lambda c: c["start"])
def speaker_profiles(segments: Sequence[dict], duration: float) -> List[dict]:
"""Per-speaker statistics and sample lines, so a person can tell who is who.
A bare ``SPEAKER_00`` label is useless for deciding whose audio to cut.
What identifies a role is *how* someone participates: an interviewer or
a crew member asks short questions and holds little of the runtime,
while the subject speaks in long stretches. ``avg_segment`` and
``share`` capture exactly that contrast, and the sample lines confirm
it in the person's own words.
"""
by_speaker: dict = {}
for seg in segments:
sid = seg.get("speaker", seg.get("speaker_id", DEFAULT_SPEAKER))
length = max(0.0, float(seg.get("end", 0.0)) - float(seg.get("start", 0.0)))
entry = by_speaker.setdefault(sid, {"seconds": 0.0, "segments": [], "words": 0})
entry["seconds"] += length
entry["words"] += len(seg.get("words", []))
entry["segments"].append(seg)
profiles: List[dict] = []
for i, (sid, entry) in enumerate(
sorted(by_speaker.items(), key=lambda kv: kv[1]["seconds"], reverse=True)
):
count = len(entry["segments"])
# Longest lines identify a role far better than the first ones: a
# question and an answer look alike at the start of a recording.
longest = sorted(
entry["segments"],
key=lambda s: float(s.get("end", 0)) - float(s.get("start", 0)),
reverse=True,
)[:3]
profiles.append({
"id": sid,
"name": f"Speaker {i + 1}",
"speaking_seconds": round(entry["seconds"], 2),
"share": round(entry["seconds"] / duration, 3) if duration > 0 else 0.0,
"segment_count": count,
"avg_segment": round(entry["seconds"] / count, 2) if count else 0.0,
"word_count": entry["words"],
"samples": [(s.get("text") or "").strip()[:160] for s in longest],
})
return profiles
def select_peaks(
words: Sequence[dict], percentile: float, floor: float
) -> List[dict]:
"""The most emphatic words: the top ``percentile`` fraction, above ``floor``.
Selection is relative on purpose. The emphasis index is a weighted
average whose real range depends entirely on the material — a measured
interview peaks around 0.5 while an energetic ad reaches much higher —
so any fixed cutoff either floods one and selects nothing in the other.
Asking for "the top 2%" instead yields a usable handful either way.
``floor`` is only a sanity guard for genuinely flat audio, where even
the top of the distribution carries no emphasis worth cutting on.
"""
ranked = sorted(words, key=lambda w: w["emphasis"], reverse=True)
keep = max(1, round(len(ranked) * percentile)) if ranked else 0
return [w for w in ranked[:keep] if w["emphasis"] >= floor]
def _summary(words: Sequence[dict], segments: Sequence[dict], speakers: Sequence[dict],
duration: float, peak_percentile: float, emphasis_floor: float) -> dict:
"""The top layer: the shape of the piece in a handful of numbers."""
emphases = [w["emphasis"] for w in words]
peaks = select_peaks(words, peak_percentile, emphasis_floor)
return {
"duration": round(duration, 3),
"speaker_count": len(speakers),
"segment_count": len(segments),
"word_count": len(words),
"avg_emphasis": round(sum(emphases) / len(emphases), 3) if emphases else 0.0,
"peak_selection": f"top {peak_percentile:.0%} of words, minimum emphasis {emphasis_floor:.2f}",
"peak_count": len(peaks),
"peak_moments": [
{
"time": round(float(w.get("start", 0.0)), 3),
"text": w.get("word", ""),
"speaker": w.get("speaker_id", DEFAULT_SPEAKER),
"emphasis": round(w["emphasis"], 3),
}
for w in sorted(peaks, key=lambda w: w["emphasis"], reverse=True)[:20]
],
}
def build_voice_timeline(
media_path: str,
transcript: dict,
hf_token: Optional[str] = None,
num_speakers: str = "",
weights: EmphasisWeights = EmphasisWeights(),
peak_percentile: float = 0.02,
emphasis_floor: float = 0.25,
emotion_enabled: bool = False,
emotion_sensitivity: float = 0.5,
rotation: float = 0.0,
progress_cb: Optional[Callable[[float, str], None]] = None,
) -> dict:
"""Build the consolidated voice timeline for one media file.
Every analysis layer is optional and degrades independently: without
librosa the acoustic values are ``0.0``; without a diarization token
every word belongs to ``SPEAKER_00``. The document's shape never
changes, so downstream consumers (the rules engine, or a model reading
the JSON) can rely on it.
"""
def report(fraction: float, stage: str) -> None:
if progress_cb:
progress_cb(fraction, stage)
report(0.1, "Analisando tom e energia...")
pitch_track = extract_pitch(media_path)
energy_track = extract_energy(media_path)
report(0.5, "Calculando ênfase...")
words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights)
words = annotate_emotions(words, emotion_enabled, emotion_sensitivity)
report(0.7, "Identificando participantes...")
tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None
segments, words = assign_speakers(transcript.get("segments", []), words, tracks)
speakers = build_speakers(segments)
report(0.9, "Montando linha do tempo...")
duration = float(transcript.get("duration", 0.0))
segment_rows = _segment_rows(segments, words)
return {
"version": VOICE_TIMELINE_VERSION,
"source": Path(media_path).name,
# Edit-time correction from the clip's Transform filter in the FCPXML
# (e.g. straightening a tilted phone shot) — 0.0 when the clip has none
# or the caller didn't resolve one.
"rotation": rotation,
"language": transcript.get("language", ""),
# What actually ran, not what was installed — a consumer must be able
# to tell "this speech is flat" from "the acoustics never loaded",
# since both leave the same zeros in the data.
"layers": {
"transcript": bool(transcript.get("words")),
"acoustics": pitch_track is not None or energy_track is not None,
"speakers": tracks is not None,
"emotion": bool(emotion_enabled),
"alignment": bool(transcript.get("alignment")),
},
"scales": VALUE_SCALES,
"summary": _summary(
words, segments, speakers, duration, peak_percentile, emphasis_floor
),
"speakers": speaker_profiles(segment_rows, duration),
"segments": segment_rows,
}
def voice_timeline_path(media_path: str, output_dir: Optional[str] = None) -> Path:
"""Where the ``_voice_timeline.json`` for ``media_path`` lives.
Mirrors ``_transcript.json``: next to the media, or in the chosen
project folder when one is set.
"""
p = Path(media_path)
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
return directory / f"{p.stem}_voice_timeline.json"
return p.with_name(p.stem + "_voice_timeline.json")
def save_voice_timeline(timeline: dict, path: Path) -> None:
"""Write the timeline as UTF-8 JSON (accented transcripts stay readable)."""
with open(path, "w", encoding="utf-8") as f:
json.dump(timeline, f, ensure_ascii=False, indent=2)
def load_voice_timeline(path: Path) -> Optional[dict]:
"""Read a cached voice timeline, or ``None`` when absent/unreadable."""
try:
with open(path, encoding="utf-8") as f:
data = json.load(f)
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
return None
return data if isinstance(data, dict) and "segments" in data else None