chore: atualização geral
This commit is contained in:
@@ -0,0 +1,505 @@
|
||||
"""Voice timeline — the consolidated, AI-readable view of how a video is spoken.
|
||||
|
||||
This is the *source of truth* between analysis and editing: it merges what
|
||||
was said (transcript), who said it (diarization), and how it was said
|
||||
(pitch/energy/rate/pauses → emphasis) into one JSON document, decoupling the
|
||||
audio analysis from FCPXML generation entirely.
|
||||
|
||||
The shape is designed to be handed to a language model so it can reason about
|
||||
the narrative — which beats carry weight, where a speaker changes, where the
|
||||
delivery peaks — and decide how to direct the edit. Two design choices serve
|
||||
that goal:
|
||||
|
||||
* **Layered, not flat.** A ``summary`` gives the whole picture in a few
|
||||
numbers, ``segments`` group words into utterances with their own
|
||||
aggregates, and ``words`` hold the fine detail. A model can reason from
|
||||
the top layer and only descend where it matters, instead of parsing
|
||||
thousands of word rows to find the shape of the piece.
|
||||
* **Normalized, self-describing values.** Every acoustic value is 0–1 and
|
||||
relative to *this* recording (a quiet podcast and a shouted ad both use
|
||||
the full range), and ``scales`` documents that contract inline, so the
|
||||
numbers are interpretable without external context.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Callable, List, Optional, Sequence, Tuple
|
||||
|
||||
from .diarize import DEFAULT_SPEAKER, assign_speakers, build_speakers, diarize
|
||||
from .emphasis import EmphasisWeights, annotate_emphasis
|
||||
from .voice_features import (
|
||||
compute_pauses,
|
||||
compute_speech_rate,
|
||||
extract_energy,
|
||||
extract_pitch,
|
||||
word_pitch_energy,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
VOICE_TIMELINE_VERSION = "1.0"
|
||||
|
||||
# Silence long enough to mean the take stopped rather than the speaker paused.
|
||||
# On real footage, boundaries between retakes showed gaps of 3.6-19.8s while
|
||||
# dramatic beats inside a delivered line stayed under ~2s.
|
||||
TAKE_BOUNDARY_GAP = 3.0
|
||||
|
||||
# How to read the values in this document, split by the level they live on.
|
||||
# Embedded in the output so a model consuming the JSON needs no external
|
||||
# documentation — and kept honest: a metric listed under "word" must exist on
|
||||
# every word row, and one under "segment" on every segment row.
|
||||
VALUE_SCALES = {
|
||||
"word": {
|
||||
"energy": "0-1, loudness relative to the loudest moment of this recording",
|
||||
"pitch_delta": "0-1, how far this word's pitch sits from the speaker's average",
|
||||
"rate_delta": "0-1, how much the local speaking rate departs from the average",
|
||||
"pause_before": "seconds of silence immediately before the word",
|
||||
"emphasis": "0-1 combined index; high values are punch-in/highlight candidates",
|
||||
},
|
||||
"segment": {
|
||||
"gap_before": "seconds of silence before this line",
|
||||
"take_boundary": "true when the gap is long enough that the take likely restarted here",
|
||||
"avg_energy": "0-1 mean loudness across the line",
|
||||
"peak_emphasis": "0-1 highest emphasis of any word in the line",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _normalize(value: Optional[float], maximum: float) -> float:
|
||||
"""Scale ``value`` into 0-1 against ``maximum`` (0.0 when unavailable)."""
|
||||
if value is None or maximum <= 0:
|
||||
return 0.0
|
||||
return max(0.0, min(1.0, value / maximum))
|
||||
|
||||
|
||||
def _round_word(word: dict) -> dict:
|
||||
"""One word row, rounded to a size a model can read without noise.
|
||||
|
||||
The raw ``energy_raw``/``pitch_hz`` ride along beside the normalized
|
||||
values so the document can be re-analyzed over a subset later. That
|
||||
matters after cutting: every normalized value is relative to the
|
||||
loudest moment of the *whole* recording, and if that moment gets cut
|
||||
the survivors are scored against something that no longer exists.
|
||||
"""
|
||||
return {
|
||||
"text": word.get("word", ""),
|
||||
"start": round(float(word.get("start", 0.0)), 3),
|
||||
"end": round(float(word.get("end", 0.0)), 3),
|
||||
"speaker": word.get("speaker_id", DEFAULT_SPEAKER),
|
||||
"energy": round(word.get("energy_norm", 0.0), 3),
|
||||
"pitch_delta": round(word.get("pitch_delta", 0.0), 3),
|
||||
"rate_delta": round(word.get("rate_delta", 0.0), 3),
|
||||
"pause_before": round(word.get("pause_before", 0.0), 3),
|
||||
"emphasis": round(word.get("emphasis", 0.0), 3),
|
||||
"energy_raw": word.get("energy"),
|
||||
"pitch_hz": word.get("pitch_hz"),
|
||||
}
|
||||
|
||||
|
||||
def enrich_words(
|
||||
words: Sequence[dict],
|
||||
pitch_track: Optional[Sequence] = None,
|
||||
energy_track: Optional[Sequence] = None,
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
already_measured: bool = False,
|
||||
) -> List[dict]:
|
||||
"""Attach normalized acoustic features + the emphasis index to each word.
|
||||
|
||||
Normalization is per-recording: energy against the loudest word, pitch
|
||||
against the spread around this recording's average, rate against the
|
||||
largest local departure. That makes the numbers comparable within a
|
||||
piece regardless of how it was recorded.
|
||||
|
||||
Set ``already_measured`` when the words already carry ``energy`` and
|
||||
``pitch_hz`` from a previous pass — re-analyzing a subset, say. The
|
||||
frame tracks are then unnecessary, and sampling them again would
|
||||
overwrite good values with ``None``.
|
||||
"""
|
||||
if not words:
|
||||
return []
|
||||
|
||||
if not already_measured:
|
||||
words = word_pitch_energy(words, pitch_track, energy_track)
|
||||
rates = compute_speech_rate(words)
|
||||
pauses = compute_pauses(words)
|
||||
|
||||
energies = [w["energy"] for w in words if w.get("energy") is not None]
|
||||
max_energy = max(energies) if energies else 0.0
|
||||
pitches = [w["pitch_hz"] for w in words if w.get("pitch_hz") is not None]
|
||||
avg_pitch = sum(pitches) / len(pitches) if pitches else 0.0
|
||||
pitch_span = (max(pitches) - min(pitches)) if len(pitches) > 1 else 0.0
|
||||
avg_rate = sum(rates) / len(rates) if rates else 0.0
|
||||
max_rate = max(rates) if rates else 0.0
|
||||
|
||||
enriched: List[dict] = []
|
||||
for i, w in enumerate(words):
|
||||
ww = dict(w)
|
||||
ww["energy_norm"] = _normalize(w.get("energy"), max_energy)
|
||||
pitch = w.get("pitch_hz")
|
||||
ww["pitch_delta"] = (
|
||||
_normalize(abs(pitch - avg_pitch), pitch_span) if pitch is not None else 0.0
|
||||
)
|
||||
ww["rate_delta"] = _normalize(abs(rates[i] - avg_rate), max_rate)
|
||||
ww["pause_before"] = pauses[i]
|
||||
enriched.append(ww)
|
||||
|
||||
annotated = annotate_emphasis(
|
||||
[{**w, "energy": w["energy_norm"]} for w in enriched], weights=weights
|
||||
)
|
||||
for word, scored in zip(enriched, annotated):
|
||||
word["emphasis"] = scored["emphasis"]
|
||||
return enriched
|
||||
|
||||
|
||||
def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]:
|
||||
"""Group enriched words under their segment, with per-segment aggregates.
|
||||
|
||||
The aggregates are what let a model judge a whole utterance ("this line
|
||||
is delivered hot, that one trails off") without reading every word.
|
||||
"""
|
||||
rows: List[dict] = []
|
||||
previous_end = 0.0
|
||||
for seg in segments:
|
||||
start = float(seg.get("start", 0.0))
|
||||
end = float(seg.get("end", 0.0))
|
||||
in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end]
|
||||
energies = [w["energy_norm"] for w in in_seg]
|
||||
emphases = [w["emphasis"] for w in in_seg]
|
||||
gap = max(0.0, start - previous_end)
|
||||
rows.append(
|
||||
{
|
||||
"start": round(start, 3),
|
||||
"end": round(end, 3),
|
||||
"speaker": seg.get("speaker_id", DEFAULT_SPEAKER),
|
||||
"text": (seg.get("text") or "").strip(),
|
||||
# Silence before this line. Long gaps are where the camera
|
||||
# stopped or the take restarted, so this is the structural
|
||||
# hint for splitting a recording into takes — the same signal
|
||||
# that is *noise* for emphasis (see emphasis.pause_weight).
|
||||
"gap_before": round(gap, 3),
|
||||
"take_boundary": gap >= TAKE_BOUNDARY_GAP,
|
||||
"avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0,
|
||||
"peak_emphasis": round(max(emphases), 3) if emphases else 0.0,
|
||||
"words": [_round_word(w) for w in in_seg],
|
||||
}
|
||||
)
|
||||
previous_end = end
|
||||
return rows
|
||||
|
||||
|
||||
# Words too common to ever be the point of a punch-in. A zoom lands on what a
|
||||
# sentence is *about*, and an article spoken loudly is still an article.
|
||||
_FUNCTION_WORDS = {
|
||||
"a", "o", "e", "de", "da", "do", "que", "é", "em", "um", "uma", "as", "os",
|
||||
"no", "na", "com", "pra", "para", "por", "se", "mais", "isso", "aí", "tudo",
|
||||
"ao", "à", "dos", "das", "nos", "nas", "ou", "mas", "já", "ele", "ela",
|
||||
"eu", "você", "seu", "sua", "meu", "minha", "esse", "essa", "aquele",
|
||||
}
|
||||
|
||||
|
||||
def _survives(start: float, end: float, cuts: Sequence[Tuple[float, float]]) -> bool:
|
||||
"""Whether a span lies entirely outside every removed range."""
|
||||
return all(end <= cut_start or start >= cut_end for cut_start, cut_end in cuts)
|
||||
|
||||
|
||||
def restrict_to_kept(
|
||||
timeline: dict,
|
||||
cut_ranges: Sequence[Tuple[float, float]],
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
peak_percentile: float = 0.02,
|
||||
emphasis_floor: float = 0.25,
|
||||
) -> dict:
|
||||
"""Re-analyze a timeline over only the material that survives ``cut_ranges``.
|
||||
|
||||
Emphasis is *relative*: energy is scored against the loudest word,
|
||||
pitch against the spread of the recording. Cut the loudest moment out —
|
||||
a laugh, an aside to the crew — and every remaining score is measured
|
||||
against something the viewer will never see. Re-running the
|
||||
normalization over just the survivors is what makes "the most emphatic
|
||||
line of the final video" a meaningful question.
|
||||
|
||||
Returns a timeline of the same shape, with times still in original
|
||||
source seconds so the result can be fed straight back as actions.
|
||||
"""
|
||||
kept_words = [
|
||||
w
|
||||
for segment in timeline.get("segments", [])
|
||||
for w in segment.get("words", [])
|
||||
if _survives(w["start"], w["end"], cut_ranges)
|
||||
]
|
||||
# enrich_words expects the raw analysis keys, not the normalized ones.
|
||||
raw = [
|
||||
{
|
||||
"word": w["text"],
|
||||
"start": w["start"],
|
||||
"end": w["end"],
|
||||
"speaker_id": w.get("speaker", DEFAULT_SPEAKER),
|
||||
"energy": w.get("energy_raw"),
|
||||
"pitch_hz": w.get("pitch_hz"),
|
||||
}
|
||||
for w in kept_words
|
||||
]
|
||||
enriched = enrich_words(raw, weights=weights, already_measured=True)
|
||||
|
||||
kept_segments = [
|
||||
{**s, "words": [w for w in s.get("words", []) if _survives(w["start"], w["end"], cut_ranges)]}
|
||||
for s in timeline.get("segments", [])
|
||||
]
|
||||
kept_segments = [s for s in kept_segments if s["words"]]
|
||||
rows = _segment_rows(
|
||||
[{"text": s["text"], "start": s["start"], "end": s["end"],
|
||||
"speaker_id": s.get("speaker", DEFAULT_SPEAKER)} for s in kept_segments],
|
||||
enriched,
|
||||
)
|
||||
duration = sum(s["end"] - s["start"] for s in rows)
|
||||
return {
|
||||
**timeline,
|
||||
"summary": _summary(enriched, rows, timeline.get("speakers", []),
|
||||
duration, peak_percentile, emphasis_floor),
|
||||
"segments": rows,
|
||||
}
|
||||
|
||||
|
||||
def sentence_end(segments: Sequence[dict], index: int) -> float:
|
||||
"""Where the sentence starting at ``segments[index]`` actually finishes.
|
||||
|
||||
Transcription segments break on breath and timing, not on grammar — a
|
||||
sentence routinely spans two or three of them ("…que dá aquele ar" /
|
||||
"de elegância, isso é desejo de muitas mulheres, né?"). A zoom that
|
||||
ends on a segment boundary would therefore release mid-thought, so the
|
||||
window is extended until a segment closes with terminal punctuation.
|
||||
"""
|
||||
last = float(segments[index]["end"])
|
||||
for offset, segment in enumerate(segments[index:]):
|
||||
# A long gap means the take stopped; never run a zoom across that.
|
||||
# Checked before adopting the end, or the boundary segment's own
|
||||
# end would already have been taken.
|
||||
if offset > 0 and segment.get("take_boundary"):
|
||||
break
|
||||
last = float(segment["end"])
|
||||
if (segment.get("text") or "").strip().endswith((".", "!", "?", "…")):
|
||||
break
|
||||
return last
|
||||
|
||||
|
||||
def suggest_zoom_windows(
|
||||
timeline: dict,
|
||||
min_gap: float = 8.0,
|
||||
max_zooms: Optional[int] = None,
|
||||
) -> List[dict]:
|
||||
"""Propose punch-in windows over a timeline's strongest lines.
|
||||
|
||||
One zoom per line at most, taken from the line's most emphatic
|
||||
*content* word — a loudly spoken "a" is still an article, so function
|
||||
words are skipped. The window runs from that word to the end of its
|
||||
line, which is the shape the edit wants: the move lands with the word
|
||||
and holds through the rest of the phrase.
|
||||
|
||||
``min_gap`` keeps successive zooms apart; effects stacked close
|
||||
together read as nervous editing rather than emphasis.
|
||||
"""
|
||||
segments = timeline.get("segments", [])
|
||||
candidates: List[dict] = []
|
||||
for i, segment in enumerate(segments):
|
||||
content = [
|
||||
w for w in segment.get("words", [])
|
||||
if w["text"].strip(",.!?;:").lower() not in _FUNCTION_WORDS
|
||||
]
|
||||
if not content:
|
||||
continue
|
||||
best = max(content, key=lambda w: w["emphasis"])
|
||||
candidates.append({
|
||||
"start": best["start"],
|
||||
# Hold through to the end of the sentence, not of the segment —
|
||||
# releasing mid-thought is what makes a punch-in feel arbitrary.
|
||||
"end": sentence_end(segments, i),
|
||||
"word": best["text"],
|
||||
"emphasis": best["emphasis"],
|
||||
"line": segment["text"],
|
||||
})
|
||||
|
||||
chosen: List[dict] = []
|
||||
for candidate in sorted(candidates, key=lambda c: c["emphasis"], reverse=True):
|
||||
if max_zooms is not None and len(chosen) >= max_zooms:
|
||||
break
|
||||
if any(abs(candidate["start"] - c["start"]) < min_gap for c in chosen):
|
||||
continue
|
||||
chosen.append(candidate)
|
||||
return sorted(chosen, key=lambda c: c["start"])
|
||||
|
||||
|
||||
def speaker_profiles(segments: Sequence[dict], duration: float) -> List[dict]:
|
||||
"""Per-speaker statistics and sample lines, so a person can tell who is who.
|
||||
|
||||
A bare ``SPEAKER_00`` label is useless for deciding whose audio to cut.
|
||||
What identifies a role is *how* someone participates: an interviewer or
|
||||
a crew member asks short questions and holds little of the runtime,
|
||||
while the subject speaks in long stretches. ``avg_segment`` and
|
||||
``share`` capture exactly that contrast, and the sample lines confirm
|
||||
it in the person's own words.
|
||||
"""
|
||||
by_speaker: dict = {}
|
||||
for seg in segments:
|
||||
sid = seg.get("speaker", seg.get("speaker_id", DEFAULT_SPEAKER))
|
||||
length = max(0.0, float(seg.get("end", 0.0)) - float(seg.get("start", 0.0)))
|
||||
entry = by_speaker.setdefault(sid, {"seconds": 0.0, "segments": [], "words": 0})
|
||||
entry["seconds"] += length
|
||||
entry["words"] += len(seg.get("words", []))
|
||||
entry["segments"].append(seg)
|
||||
|
||||
profiles: List[dict] = []
|
||||
for i, (sid, entry) in enumerate(
|
||||
sorted(by_speaker.items(), key=lambda kv: kv[1]["seconds"], reverse=True)
|
||||
):
|
||||
count = len(entry["segments"])
|
||||
# Longest lines identify a role far better than the first ones: a
|
||||
# question and an answer look alike at the start of a recording.
|
||||
longest = sorted(
|
||||
entry["segments"],
|
||||
key=lambda s: float(s.get("end", 0)) - float(s.get("start", 0)),
|
||||
reverse=True,
|
||||
)[:3]
|
||||
profiles.append({
|
||||
"id": sid,
|
||||
"name": f"Speaker {i + 1}",
|
||||
"speaking_seconds": round(entry["seconds"], 2),
|
||||
"share": round(entry["seconds"] / duration, 3) if duration > 0 else 0.0,
|
||||
"segment_count": count,
|
||||
"avg_segment": round(entry["seconds"] / count, 2) if count else 0.0,
|
||||
"word_count": entry["words"],
|
||||
"samples": [(s.get("text") or "").strip()[:160] for s in longest],
|
||||
})
|
||||
return profiles
|
||||
|
||||
|
||||
def select_peaks(
|
||||
words: Sequence[dict], percentile: float, floor: float
|
||||
) -> List[dict]:
|
||||
"""The most emphatic words: the top ``percentile`` fraction, above ``floor``.
|
||||
|
||||
Selection is relative on purpose. The emphasis index is a weighted
|
||||
average whose real range depends entirely on the material — a measured
|
||||
interview peaks around 0.5 while an energetic ad reaches much higher —
|
||||
so any fixed cutoff either floods one and selects nothing in the other.
|
||||
Asking for "the top 2%" instead yields a usable handful either way.
|
||||
|
||||
``floor`` is only a sanity guard for genuinely flat audio, where even
|
||||
the top of the distribution carries no emphasis worth cutting on.
|
||||
"""
|
||||
ranked = sorted(words, key=lambda w: w["emphasis"], reverse=True)
|
||||
keep = max(1, round(len(ranked) * percentile)) if ranked else 0
|
||||
return [w for w in ranked[:keep] if w["emphasis"] >= floor]
|
||||
|
||||
|
||||
def _summary(words: Sequence[dict], segments: Sequence[dict], speakers: Sequence[dict],
|
||||
duration: float, peak_percentile: float, emphasis_floor: float) -> dict:
|
||||
"""The top layer: the shape of the piece in a handful of numbers."""
|
||||
emphases = [w["emphasis"] for w in words]
|
||||
peaks = select_peaks(words, peak_percentile, emphasis_floor)
|
||||
return {
|
||||
"duration": round(duration, 3),
|
||||
"speaker_count": len(speakers),
|
||||
"segment_count": len(segments),
|
||||
"word_count": len(words),
|
||||
"avg_emphasis": round(sum(emphases) / len(emphases), 3) if emphases else 0.0,
|
||||
"peak_selection": f"top {peak_percentile:.0%} of words, minimum emphasis {emphasis_floor:.2f}",
|
||||
"peak_count": len(peaks),
|
||||
"peak_moments": [
|
||||
{
|
||||
"time": round(float(w.get("start", 0.0)), 3),
|
||||
"text": w.get("word", ""),
|
||||
"speaker": w.get("speaker_id", DEFAULT_SPEAKER),
|
||||
"emphasis": round(w["emphasis"], 3),
|
||||
}
|
||||
for w in sorted(peaks, key=lambda w: w["emphasis"], reverse=True)[:20]
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def build_voice_timeline(
|
||||
media_path: str,
|
||||
transcript: dict,
|
||||
hf_token: Optional[str] = None,
|
||||
num_speakers: str = "",
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
peak_percentile: float = 0.02,
|
||||
emphasis_floor: float = 0.25,
|
||||
progress_cb: Optional[Callable[[float, str], None]] = None,
|
||||
) -> dict:
|
||||
"""Build the consolidated voice timeline for one media file.
|
||||
|
||||
Every analysis layer is optional and degrades independently: without
|
||||
librosa the acoustic values are ``0.0``; without a diarization token
|
||||
every word belongs to ``SPEAKER_00``. The document's shape never
|
||||
changes, so downstream consumers (the rules engine, or a model reading
|
||||
the JSON) can rely on it.
|
||||
"""
|
||||
def report(fraction: float, stage: str) -> None:
|
||||
if progress_cb:
|
||||
progress_cb(fraction, stage)
|
||||
|
||||
report(0.1, "Analisando tom e energia...")
|
||||
pitch_track = extract_pitch(media_path)
|
||||
energy_track = extract_energy(media_path)
|
||||
|
||||
report(0.5, "Calculando ênfase...")
|
||||
words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights)
|
||||
|
||||
report(0.7, "Identificando participantes...")
|
||||
tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None
|
||||
segments, words = assign_speakers(transcript.get("segments", []), words, tracks)
|
||||
speakers = build_speakers(segments)
|
||||
|
||||
report(0.9, "Montando linha do tempo...")
|
||||
duration = float(transcript.get("duration", 0.0))
|
||||
segment_rows = _segment_rows(segments, words)
|
||||
return {
|
||||
"version": VOICE_TIMELINE_VERSION,
|
||||
"source": Path(media_path).name,
|
||||
"language": transcript.get("language", ""),
|
||||
# What actually ran, not what was installed — a consumer must be able
|
||||
# to tell "this speech is flat" from "the acoustics never loaded",
|
||||
# since both leave the same zeros in the data.
|
||||
"layers": {
|
||||
"transcript": bool(transcript.get("words")),
|
||||
"acoustics": pitch_track is not None or energy_track is not None,
|
||||
"speakers": tracks is not None,
|
||||
},
|
||||
"scales": VALUE_SCALES,
|
||||
"summary": _summary(
|
||||
words, segments, speakers, duration, peak_percentile, emphasis_floor
|
||||
),
|
||||
"speakers": speaker_profiles(segment_rows, duration),
|
||||
"segments": segment_rows,
|
||||
}
|
||||
|
||||
|
||||
def voice_timeline_path(media_path: str, output_dir: Optional[str] = None) -> Path:
|
||||
"""Where the ``_voice_timeline.json`` for ``media_path`` lives.
|
||||
|
||||
Mirrors ``_transcript.json``: next to the media, or in the chosen
|
||||
project folder when one is set.
|
||||
"""
|
||||
p = Path(media_path)
|
||||
if output_dir:
|
||||
directory = Path(output_dir).expanduser()
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
return directory / f"{p.stem}_voice_timeline.json"
|
||||
return p.with_name(p.stem + "_voice_timeline.json")
|
||||
|
||||
|
||||
def save_voice_timeline(timeline: dict, path: Path) -> None:
|
||||
"""Write the timeline as UTF-8 JSON (accented transcripts stay readable)."""
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
json.dump(timeline, f, ensure_ascii=False, indent=2)
|
||||
|
||||
|
||||
def load_voice_timeline(path: Path) -> Optional[dict]:
|
||||
"""Read a cached voice timeline, or ``None`` when absent/unreadable."""
|
||||
try:
|
||||
with open(path, encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
||||
return None
|
||||
return data if isinstance(data, dict) and "segments" in data else None
|
||||
Reference in New Issue
Block a user