"""Phrase review — the human pass between the AI's decisions and the render. A voice timeline says *how* every line was spoken; a list of voice actions says what the model decided to do about it. Neither is reviewable on its own: the timeline has no editorial intent, and the action list is a set of timecodes with no text attached. This module joins them into the one view an editor can actually judge — the script, phrase by phrase, each carrying the decision that was made about it. The phrase is the unit on purpose. Emphasis, in this pipeline, is not a property of a word but of a line: an emphasized phrase gets a punch-in and a dynamic caption, everything else gets a plain caption. Keeping the same granularity in the review, the JSON, and the render means a toggle in the UI maps to exactly one editorial outcome, with nothing to reconcile in between. Trimming stays inside the phrase for the same reason. A line is rarely wrong as a whole — it has a false start, or a trailing "né" — so each phrase carries a ``trim_start``/``trim_end`` pair that rides on word boundaries. Editing a cut therefore means picking a word, never hunting for a frame, and a partial cut from the model arrives as a trim instead of being rounded away. Round-tripping is the other half of the contract. :func:`build_phrase_review` derives the review from actions, :func:`phrase_review_to_actions` derives actions back from the edited review, and everything the editor touched wins over what was inferred — so re-opening the screen shows what was left there, not a re-derivation that quietly discards the edits. """ import json from pathlib import Path from typing import Any, Dict, List, Optional, Sequence, Tuple from .voice_actions import VoiceAction, merge_cut_ranges, parse_actions PHRASE_REVIEW_VERSION = "1.0" # Emphasis is stored 0-3 rather than as a float so the UI, the JSON and the # render agree on the same discrete decision. The thresholds map the continuous # `peak_emphasis` of the voice timeline onto those levels when the model gave no # explicit direction for a phrase. EMPHASIS_LEVELS = (0, 1, 2, 3) EMPHASIS_THRESHOLDS = (0.25, 0.45, 0.65) # Zoom scale applied per emphasis level when the review is turned back into # actions. Level 0 never produces a zoom. The values stay inside # voice_actions.MIN_ZOOM_SCALE..MAX_ZOOM_SCALE. ZOOM_SCALE_BY_LEVEL = {1: 1.15, 2: 1.3, 3: 1.5} # A phrase only survives if most of it does. Speech boundaries from a transcript # are approximate, so a cut clipping a fraction of a second off the tail is a # trim, not a removal — treating that as "phrase deleted" would grey out lines # that are still fully audible. CUT_COVERAGE_TO_DEACTIVATE = 0.6 # A punch-in shorter than this has no time to ramp in and back out — the writer # rejects the window anyway (see the zoom ease-in/ease-out shape), so refusing # it here turns a silent drop at render time into nothing being placed at all. MIN_ZOOM_DURATION = 0.4 TRACK_SCRIPT = "roteiro" TRACK_BACKSTAGE = "bastidor" TRACKS = (TRACK_SCRIPT, TRACK_BACKSTAGE) def resolve_source( source: str, voice_timeline_path: str, extra_dirs: Sequence[str] = () ) -> str: """The playable path for a timeline's ``source``, or "" when it's gone. The voice timeline stores only the media's *file name* — it is written to be read by a model, where a machine-specific absolute path is noise. That makes it useless for opening a preview, so the file is looked up where it can actually be: beside its own timeline JSON first (that is where ``analyze_voice`` writes it), then in whatever project folders the caller knows about. """ if not source: return "" candidate = Path(source) if candidate.is_absolute() and candidate.is_file(): return str(candidate) directories = [Path(voice_timeline_path).parent] if voice_timeline_path else [] directories += [Path(d) for d in extra_dirs if d] for directory in directories: found = directory / candidate.name if found.is_file(): return str(found) return "" def _overlap(a_start: float, a_end: float, b_start: float, b_end: float) -> float: """Seconds shared by two spans (0.0 when they don't touch).""" return max(0.0, min(a_end, b_end) - max(a_start, b_start)) def _cut_coverage( start: float, end: float, cuts: Sequence[Tuple[float, float]] ) -> float: """Fraction of ``start``-``end`` that falls inside ``cuts`` (0-1).""" span = end - start if span <= 0: return 0.0 removed = sum(_overlap(start, end, c_start, c_end) for c_start, c_end in cuts) return min(1.0, removed / span) def snap_to_words( time: float, words: Sequence[dict], fallback: float, edge: str ) -> float: """Move ``time`` onto the nearest word boundary of this phrase. Trims are expressed by pointing at a word, so a trim handle that landed mid-word would cut a syllable in half. ``edge`` is ``"in"`` (snap to word starts) or ``"out"`` (snap to word ends); with no word timings available the time is left as-is. """ boundaries = [ float(word.get("start" if edge == "in" else "end", 0.0)) for word in words ] boundaries = [b for b in boundaries if b > 0] if not boundaries: return fallback return min(boundaries, key=lambda b: abs(b - time)) def _trim_from_cuts( start: float, end: float, words: Sequence[dict], cuts: Sequence[Tuple[float, float]], ) -> Tuple[float, float]: """Read a partial cut over this phrase as a head/tail trim. Only cuts that touch an edge become trims: a cut carved out of the middle of a line has no representation here (the phrase is the unit), so it is left for the whole-phrase coverage rule to decide. """ trim_start, trim_end = start, end for cut_start, cut_end in cuts: if _overlap(start, end, cut_start, cut_end) <= 0: continue if cut_start <= trim_start < cut_end < end: trim_start = snap_to_words(cut_end, words, cut_end, "in") if start < cut_start < trim_end <= cut_end: trim_end = snap_to_words(cut_start, words, cut_start, "out") if trim_end <= trim_start: return start, end return trim_start, trim_end def _level_from_peak(peak: float) -> int: """Map a 0-1 ``peak_emphasis`` onto a 0-3 level.""" for level, threshold in enumerate(EMPHASIS_THRESHOLDS): if peak < threshold: return level return 3 def _level_from_scale(scale: Optional[float]) -> int: """Map a zoom's scale factor back onto a 0-3 level. The model is free to send any scale inside the allowed range, so this picks the nearest level rather than requiring one of our own three values. """ if scale is None: return 2 best = 1 smallest = None for level, level_scale in ZOOM_SCALE_BY_LEVEL.items(): distance = abs(level_scale - float(scale)) if smallest is None or distance < smallest: smallest, best = distance, level return best def _emphasis_from_actions( start: float, end: float, actions: Sequence[VoiceAction], ) -> Tuple[Optional[int], str]: """The level the model asked for on this phrase, and why. A ``zoom`` or ``text`` action anywhere inside the phrase is read as "this line is the emphasis" — the model places them on the word that carries the point, not on the whole line, so requiring a full-span match would find nothing. Returns ``(None, "")`` when no action touches the phrase. """ level: Optional[int] = None reason = "" for action in actions: if action.kind not in ("zoom", "text"): continue if _overlap(start, end, action.start, action.end) <= 0: continue if action.kind == "zoom": candidate = _level_from_scale(action.params.get("scale")) else: candidate = 2 if level is None or candidate > level: level = candidate reason = action.reason return level, reason def _cut_reason( start: float, end: float, actions: Sequence[VoiceAction] ) -> str: """The reason given for the cut that removes this phrase.""" for action in actions: if action.kind != "cut": continue if _overlap(start, end, action.start, action.end) > 0 and action.reason: return action.reason return "" def build_phrase_review( timeline: dict, actions: Any = None, voice_timeline_path: str = "", extra_dirs: Sequence[str] = (), ) -> dict: """Join a voice timeline with the AI's actions into a reviewable script. ``actions`` accepts whatever :func:`~.voice_actions.parse_actions` accepts — a bare list, ``{"actions": [...]}``, or ``None`` when there is no AI pass and the review starts from the acoustics alone. Malformed rows are skipped and reported in ``errors`` rather than raising, matching the rest of the decision pipeline. """ parsed, errors = parse_actions(actions) if actions else ([], []) cuts = merge_cut_ranges(parsed) phrases: List[dict] = [] for index, segment in enumerate(timeline.get("segments", [])): start = float(segment.get("start", 0.0)) end = float(segment.get("end", 0.0)) peak = float(segment.get("peak_emphasis", 0.0)) take_boundary = bool(segment.get("take_boundary", False)) words = list(segment.get("words", [])) coverage = _cut_coverage(start, end, cuts) active = coverage < CUT_COVERAGE_TO_DEACTIVATE trim_start, trim_end = ( _trim_from_cuts(start, end, words, cuts) if active else (start, end) ) asked_level, asked_reason = _emphasis_from_actions(start, end, parsed) if asked_level is not None: emphasis, reason = asked_level, asked_reason else: emphasis = _level_from_peak(peak) reason = f"ênfase {peak:.2f}" if emphasis else "" if not active: # A removed line carries the reason it was removed; the emphasis it # would have had is kept so re-activating it restores the decision. reason = _cut_reason(start, end, parsed) or reason phrases.append( { "index": index, "start": round(start, 3), "end": round(end, 3), "trim_start": round(trim_start, 3), "trim_end": round(trim_end, 3), "text": str(segment.get("text", "")).strip(), "speaker": str(segment.get("speaker", "")), "active": active, "emphasis": emphasis, "track": TRACK_BACKSTAGE if (not active and take_boundary) else TRACK_SCRIPT, "peak_emphasis": round(peak, 3), # Delivery emotion is a heuristic over the acoustics (see # voice_timeline._emotion_for_word) and only means anything when # the analysis actually ran — `emotion_available` below is what # separates "spoken flat" from "never measured". "emotion": str(segment.get("emotion", "neutral")), "emotion_confidence": round( float(segment.get("emotion_confidence", 0.0)), 3 ), "take_boundary": take_boundary, "gap_before": round(float(segment.get("gap_before", 0.0)), 3), "reason": reason, "words": [ { "text": str(word.get("text", "")), "start": round(float(word.get("start", 0.0)), 3), "end": round(float(word.get("end", 0.0)), 3), "energy": round(float(word.get("energy", 0.0)), 3), "emphasis": round(float(word.get("emphasis", 0.0)), 3), } for word in words ], } ) source = timeline.get("source", "") layers = timeline.get("layers", {}) if isinstance(timeline.get("layers"), dict) else {} return { "version": PHRASE_REVIEW_VERSION, "source": source, "source_path": resolve_source(source, voice_timeline_path, extra_dirs), "duration": round(phrases[-1]["end"], 3) if phrases else 0.0, "speakers": timeline.get("speakers", []), "emotion_available": bool(layers.get("emotion", False)), "phrases": phrases, # Punch-ins the editor places by hand on an arbitrary range, alongside # the whole-phrase zoom that an emphasis level produces. Both end up as # zoom actions; this one exists because the moment worth punching into # is not always a whole sentence. "zooms": [], "errors": errors, } def _coerce_zoom(raw: Any) -> Optional[Dict[str, float]]: """Normalize one manually placed zoom range.""" if not isinstance(raw, dict): return None try: start = float(raw.get("start")) end = float(raw.get("end")) except (TypeError, ValueError): return None if end - start < MIN_ZOOM_DURATION: return None return {"start": start, "end": end} def _coerce_phrase(raw: Any, index: int) -> Optional[Dict[str, Any]]: """Normalize one edited phrase row coming back from the UI.""" if not isinstance(raw, dict): return None try: start = float(raw.get("start")) end = float(raw.get("end")) except (TypeError, ValueError): return None if end <= start: return None try: emphasis = int(raw.get("emphasis", 0)) except (TypeError, ValueError): emphasis = 0 try: trim_start = float(raw.get("trim_start", start)) trim_end = float(raw.get("trim_end", end)) except (TypeError, ValueError): trim_start, trim_end = start, end # A trim that escaped the phrase, or inverted, is treated as no trim at all: # the UI is the only thing that writes these, and silently discarding a bad # pair keeps a rounding slip from deleting material the editor kept. if not (start <= trim_start < trim_end <= end): trim_start, trim_end = start, end track = str(raw.get("track", TRACK_SCRIPT)) return { "index": int(raw.get("index", index)), "start": start, "end": end, "trim_start": trim_start, "trim_end": trim_end, "text": str(raw.get("text", "")).strip(), "speaker": str(raw.get("speaker", "")), "active": bool(raw.get("active", True)), "emphasis": min(3, max(0, emphasis)), "track": track if track in TRACKS else TRACK_SCRIPT, "reason": str(raw.get("reason", "")), } def phrase_review_to_actions(review: dict) -> dict: """Turn an edited review back into the action list the applier consumes. Every deactivated phrase becomes a ``cut``, a trimmed one becomes a cut over the head and/or tail it lost, and every emphasized one becomes a ``zoom`` scaled by its level. The emphasis flags ride along in ``emphasis_spans`` so the caption step can give those lines the dynamic treatment and everything else the plain one, without re-deriving the decision from the acoustics. """ phrases = [ coerced for index, raw in enumerate(review.get("phrases", [])) if (coerced := _coerce_phrase(raw, index)) is not None ] actions: List[dict] = [] emphasis_spans: List[dict] = [] for phrase in phrases: if not phrase["active"]: actions.append( VoiceAction( kind="cut", start=phrase["start"], end=phrase["end"], reason=phrase["reason"] or "desativada na revisão", speaker=phrase["speaker"], ).as_dict() ) continue # Head and tail the editor trimmed off — each becomes its own cut, so a # false start disappears without taking the line with it. for trim_start, trim_end, where in ( (phrase["start"], phrase["trim_start"], "início"), (phrase["trim_end"], phrase["end"], "fim"), ): if trim_end - trim_start <= 0: continue actions.append( VoiceAction( kind="cut", start=trim_start, end=trim_end, reason=f"trecho do {where} da frase removido na revisão", speaker=phrase["speaker"], ).as_dict() ) if phrase["emphasis"] >= 1: actions.append( VoiceAction( kind="zoom", start=phrase["trim_start"], end=phrase["trim_end"], params={"scale": ZOOM_SCALE_BY_LEVEL[phrase["emphasis"]]}, reason=phrase["reason"] or f"ênfase nível {phrase['emphasis']}", speaker=phrase["speaker"], ).as_dict() ) emphasis_spans.append( { "start": phrase["trim_start"], "end": phrase["trim_end"], "level": phrase["emphasis"], "text": phrase["text"], } ) # Hand-placed punch-ins carry no scale on purpose: an omitted scale lets the # applier use the shape configured in "Análise de Voz" (zoom_scale, ease in # and out), so changing that setting restyles every manual zoom instead of # leaving a scale frozen into each one at the moment it was drawn. for raw in review.get("zooms", []): zoom = _coerce_zoom(raw) if zoom is None: continue actions.append( VoiceAction( kind="zoom", start=zoom["start"], end=zoom["end"], reason="zoom marcado na revisão", ).as_dict() ) return { "source": review.get("source", ""), "actions": actions, "emphasis_spans": emphasis_spans, } def merge_saved_decisions(review: dict, saved: Optional[dict]) -> dict: """Lay a previously saved review's decisions over a freshly built one. Only the editorial fields travel — active, emphasis, track, text, trims. Everything else (words, emotion, energy) is re-derived from the current analysis, so re-running the voice pass with better settings improves the screen instead of being masked by a stale copy of itself, and the saved file never has to carry a duplicate of data it does not own. Phrases are matched by index *and* start time: if the analysis changed enough to move a line, the old decision for that slot is dropped rather than applied to a different sentence. """ if not saved: return review review["zooms"] = [ zoom for raw in saved.get("zooms", []) if (zoom := _coerce_zoom(raw)) is not None ] by_index = {} for raw in saved.get("phrases", []): if isinstance(raw, dict) and "index" in raw: by_index[raw["index"]] = raw for phrase in review["phrases"]: previous = by_index.get(phrase["index"]) if previous is None: continue if abs(float(previous.get("start", -1)) - phrase["start"]) > 0.25: continue phrase["active"] = bool(previous.get("active", phrase["active"])) phrase["emphasis"] = min(3, max(0, int(previous.get("emphasis", phrase["emphasis"])))) track = str(previous.get("track", phrase["track"])) phrase["track"] = track if track in TRACKS else phrase["track"] if previous.get("text"): phrase["text"] = str(previous["text"]) trim_start = float(previous.get("trim_start", phrase["trim_start"])) trim_end = float(previous.get("trim_end", phrase["trim_end"])) if phrase["start"] <= trim_start < trim_end <= phrase["end"]: phrase["trim_start"], phrase["trim_end"] = trim_start, trim_end return review def review_paths(voice_timeline_path: str) -> Tuple[Path, Path]: """Where the review and its derived actions live, next to the timeline. Both files sit beside the ``_voice_timeline.json`` they came from and are named after it, so a project folder stays readable and re-running the wizard on the same take overwrites its own files instead of accumulating copies. """ base = Path(voice_timeline_path) stem = base.stem if stem.endswith("_voice_timeline"): stem = stem[: -len("_voice_timeline")] return ( base.with_name(f"{stem}_phrase_review.json"), base.with_name(f"{stem}_phrase_actions.json"), ) def save_phrase_review(voice_timeline_path: str, review: dict) -> Tuple[Path, Path]: """Write the edited review and the actions derived from it. Returns both paths.""" review_path, actions_path = review_paths(voice_timeline_path) review_path.write_text( json.dumps(review, ensure_ascii=False, indent=2), encoding="utf-8" ) actions_path.write_text( json.dumps(phrase_review_to_actions(review), ensure_ascii=False, indent=2), encoding="utf-8", ) return review_path, actions_path def load_phrase_review(voice_timeline_path: str) -> Optional[dict]: """The review saved earlier for this timeline, or ``None`` if there is none.""" review_path, _ = review_paths(voice_timeline_path) if not review_path.is_file(): return None try: data = json.loads(review_path.read_text(encoding="utf-8")) except (OSError, json.JSONDecodeError): return None return data if isinstance(data, dict) else None