Files
gart/code/fcpxml/phrase_review.py
João HenriqueandClaude Sonnet 5 8257155fd3 fix(voz): frases desativadas em sequência deixavam fatias sobrando no corte
phrase_review_to_actions() cortava cada frase desativada isoladamente
(start..end da própria frase) — quando várias seguidas estavam desativadas,
a pausa ENTRE elas não pertencia a nenhuma frase e sobrevivia como um
clipe minúsculo (0,1-0,5s) na timeline final. Confirmado no projeto
Mastopexia real: 29 cuts individuais geravam mais de uma dezena de fatias
sub-segundo; agrupar frases desativadas consecutivas num único cut (do
início da primeira ao fim da última) reduziu para 3 cuts e 4 fatias
residuais (menores, provavelmente do padding do remove_media_silence —
registrado como dívida separada em 09_MANUTENCAO.md §2.5).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-21 18:49:13 -04:00

572 lines
22 KiB
Python

"""Phrase review — the human pass between the AI's decisions and the render.
A voice timeline says *how* every line was spoken; a list of voice actions says
what the model decided to do about it. Neither is reviewable on its own: the
timeline has no editorial intent, and the action list is a set of timecodes with
no text attached. This module joins them into the one view an editor can
actually judge — the script, phrase by phrase, each carrying the decision that
was made about it.
The phrase is the unit on purpose. Emphasis, in this pipeline, is not a property
of a word but of a line: an emphasized phrase gets a punch-in and a dynamic
caption, everything else gets a plain caption. Keeping the same granularity in
the review, the JSON, and the render means a toggle in the UI maps to exactly
one editorial outcome, with nothing to reconcile in between.
Trimming stays inside the phrase for the same reason. A line is rarely wrong as
a whole — it has a false start, or a trailing "né" — so each phrase carries a
``trim_start``/``trim_end`` pair that rides on word boundaries. Editing a cut
therefore means picking a word, never hunting for a frame, and a partial cut
from the model arrives as a trim instead of being rounded away.
Round-tripping is the other half of the contract. :func:`build_phrase_review`
derives the review from actions, :func:`phrase_review_to_actions` derives
actions back from the edited review, and everything the editor touched wins over
what was inferred — so re-opening the screen shows what was left there, not a
re-derivation that quietly discards the edits.
"""
import json
from pathlib import Path
from typing import Any, Dict, List, Optional, Sequence, Tuple
from .voice_actions import VoiceAction, merge_cut_ranges, parse_actions
PHRASE_REVIEW_VERSION = "1.0"
# Emphasis is stored 0-3 rather than as a float so the UI, the JSON and the
# render agree on the same discrete decision. The thresholds map the continuous
# `peak_emphasis` of the voice timeline onto those levels when the model gave no
# explicit direction for a phrase.
EMPHASIS_LEVELS = (0, 1, 2, 3)
EMPHASIS_THRESHOLDS = (0.25, 0.45, 0.65)
# Zoom scale applied per emphasis level when the review is turned back into
# actions. Level 0 never produces a zoom. The values stay inside
# voice_actions.MIN_ZOOM_SCALE..MAX_ZOOM_SCALE.
ZOOM_SCALE_BY_LEVEL = {1: 1.15, 2: 1.3, 3: 1.5}
# A phrase only survives if most of it does. Speech boundaries from a transcript
# are approximate, so a cut clipping a fraction of a second off the tail is a
# trim, not a removal — treating that as "phrase deleted" would grey out lines
# that are still fully audible.
CUT_COVERAGE_TO_DEACTIVATE = 0.6
# A punch-in shorter than this has no time to ramp in and back out — the writer
# rejects the window anyway (see the zoom ease-in/ease-out shape), so refusing
# it here turns a silent drop at render time into nothing being placed at all.
MIN_ZOOM_DURATION = 0.4
TRACK_SCRIPT = "roteiro"
TRACK_BACKSTAGE = "bastidor"
TRACKS = (TRACK_SCRIPT, TRACK_BACKSTAGE)
def resolve_source(
source: str, voice_timeline_path: str, extra_dirs: Sequence[str] = ()
) -> str:
"""The playable path for a timeline's ``source``, or "" when it's gone.
The voice timeline stores only the media's *file name* — it is written to be
read by a model, where a machine-specific absolute path is noise. That makes
it useless for opening a preview, so the file is looked up where it can
actually be: beside its own timeline JSON first (that is where
``analyze_voice`` writes it), then in whatever project folders the caller
knows about.
"""
if not source:
return ""
candidate = Path(source)
if candidate.is_absolute() and candidate.is_file():
return str(candidate)
directories = [Path(voice_timeline_path).parent] if voice_timeline_path else []
directories += [Path(d) for d in extra_dirs if d]
for directory in directories:
found = directory / candidate.name
if found.is_file():
return str(found)
return ""
def _overlap(a_start: float, a_end: float, b_start: float, b_end: float) -> float:
"""Seconds shared by two spans (0.0 when they don't touch)."""
return max(0.0, min(a_end, b_end) - max(a_start, b_start))
def _cut_coverage(
start: float, end: float, cuts: Sequence[Tuple[float, float]]
) -> float:
"""Fraction of ``start``-``end`` that falls inside ``cuts`` (0-1)."""
span = end - start
if span <= 0:
return 0.0
removed = sum(_overlap(start, end, c_start, c_end) for c_start, c_end in cuts)
return min(1.0, removed / span)
def snap_to_words(
time: float, words: Sequence[dict], fallback: float, edge: str
) -> float:
"""Move ``time`` onto the nearest word boundary of this phrase.
Trims are expressed by pointing at a word, so a trim handle that landed
mid-word would cut a syllable in half. ``edge`` is ``"in"`` (snap to word
starts) or ``"out"`` (snap to word ends); with no word timings available the
time is left as-is.
"""
boundaries = [
float(word.get("start" if edge == "in" else "end", 0.0)) for word in words
]
boundaries = [b for b in boundaries if b > 0]
if not boundaries:
return fallback
return min(boundaries, key=lambda b: abs(b - time))
def _trim_from_cuts(
start: float,
end: float,
words: Sequence[dict],
cuts: Sequence[Tuple[float, float]],
) -> Tuple[float, float]:
"""Read a partial cut over this phrase as a head/tail trim.
Only cuts that touch an edge become trims: a cut carved out of the middle of
a line has no representation here (the phrase is the unit), so it is left
for the whole-phrase coverage rule to decide.
"""
trim_start, trim_end = start, end
for cut_start, cut_end in cuts:
if _overlap(start, end, cut_start, cut_end) <= 0:
continue
if cut_start <= trim_start < cut_end < end:
trim_start = snap_to_words(cut_end, words, cut_end, "in")
if start < cut_start < trim_end <= cut_end:
trim_end = snap_to_words(cut_start, words, cut_start, "out")
if trim_end <= trim_start:
return start, end
return trim_start, trim_end
def _level_from_peak(peak: float) -> int:
"""Map a 0-1 ``peak_emphasis`` onto a 0-3 level."""
for level, threshold in enumerate(EMPHASIS_THRESHOLDS):
if peak < threshold:
return level
return 3
def _level_from_scale(scale: Optional[float]) -> int:
"""Map a zoom's scale factor back onto a 0-3 level.
The model is free to send any scale inside the allowed range, so this picks
the nearest level rather than requiring one of our own three values.
"""
if scale is None:
return 2
best = 1
smallest = None
for level, level_scale in ZOOM_SCALE_BY_LEVEL.items():
distance = abs(level_scale - float(scale))
if smallest is None or distance < smallest:
smallest, best = distance, level
return best
def _emphasis_from_actions(
start: float,
end: float,
actions: Sequence[VoiceAction],
) -> Tuple[Optional[int], str]:
"""The level the model asked for on this phrase, and why.
A ``zoom`` or ``text`` action anywhere inside the phrase is read as "this
line is the emphasis" — the model places them on the word that carries the
point, not on the whole line, so requiring a full-span match would find
nothing. Returns ``(None, "")`` when no action touches the phrase.
"""
level: Optional[int] = None
reason = ""
for action in actions:
if action.kind not in ("zoom", "text"):
continue
if _overlap(start, end, action.start, action.end) <= 0:
continue
if action.kind == "zoom":
candidate = _level_from_scale(action.params.get("scale"))
else:
candidate = 2
if level is None or candidate > level:
level = candidate
reason = action.reason
return level, reason
def _cut_reason(
start: float, end: float, actions: Sequence[VoiceAction]
) -> str:
"""The reason given for the cut that removes this phrase."""
for action in actions:
if action.kind != "cut":
continue
if _overlap(start, end, action.start, action.end) > 0 and action.reason:
return action.reason
return ""
def build_phrase_review(
timeline: dict,
actions: Any = None,
voice_timeline_path: str = "",
extra_dirs: Sequence[str] = (),
) -> dict:
"""Join a voice timeline with the AI's actions into a reviewable script.
``actions`` accepts whatever :func:`~.voice_actions.parse_actions` accepts —
a bare list, ``{"actions": [...]}``, or ``None`` when there is no AI pass and
the review starts from the acoustics alone. Malformed rows are skipped and
reported in ``errors`` rather than raising, matching the rest of the
decision pipeline.
"""
parsed, errors = parse_actions(actions) if actions else ([], [])
cuts = merge_cut_ranges(parsed)
phrases: List[dict] = []
for index, segment in enumerate(timeline.get("segments", [])):
start = float(segment.get("start", 0.0))
end = float(segment.get("end", 0.0))
peak = float(segment.get("peak_emphasis", 0.0))
take_boundary = bool(segment.get("take_boundary", False))
words = list(segment.get("words", []))
coverage = _cut_coverage(start, end, cuts)
active = coverage < CUT_COVERAGE_TO_DEACTIVATE
trim_start, trim_end = (
_trim_from_cuts(start, end, words, cuts) if active else (start, end)
)
asked_level, asked_reason = _emphasis_from_actions(start, end, parsed)
if asked_level is not None:
emphasis, reason = asked_level, asked_reason
else:
emphasis = _level_from_peak(peak)
reason = f"ênfase {peak:.2f}" if emphasis else ""
if not active:
# A removed line carries the reason it was removed; the emphasis it
# would have had is kept so re-activating it restores the decision.
reason = _cut_reason(start, end, parsed) or reason
phrases.append(
{
"index": index,
"start": round(start, 3),
"end": round(end, 3),
"trim_start": round(trim_start, 3),
"trim_end": round(trim_end, 3),
"text": str(segment.get("text", "")).strip(),
"speaker": str(segment.get("speaker", "")),
"active": active,
"emphasis": emphasis,
"track": TRACK_BACKSTAGE if (not active and take_boundary) else TRACK_SCRIPT,
"peak_emphasis": round(peak, 3),
# Delivery emotion is a heuristic over the acoustics (see
# voice_timeline._emotion_for_word) and only means anything when
# the analysis actually ran — `emotion_available` below is what
# separates "spoken flat" from "never measured".
"emotion": str(segment.get("emotion", "neutral")),
"emotion_confidence": round(
float(segment.get("emotion_confidence", 0.0)), 3
),
"take_boundary": take_boundary,
"gap_before": round(float(segment.get("gap_before", 0.0)), 3),
"reason": reason,
"words": [
{
"text": str(word.get("text", "")),
"start": round(float(word.get("start", 0.0)), 3),
"end": round(float(word.get("end", 0.0)), 3),
"energy": round(float(word.get("energy", 0.0)), 3),
"emphasis": round(float(word.get("emphasis", 0.0)), 3),
}
for word in words
],
}
)
source = timeline.get("source", "")
layers = timeline.get("layers", {}) if isinstance(timeline.get("layers"), dict) else {}
return {
"version": PHRASE_REVIEW_VERSION,
"source": source,
"source_path": resolve_source(source, voice_timeline_path, extra_dirs),
"rotation": float(timeline.get("rotation", 0.0)),
"duration": round(phrases[-1]["end"], 3) if phrases else 0.0,
"speakers": timeline.get("speakers", []),
"emotion_available": bool(layers.get("emotion", False)),
"phrases": phrases,
# Punch-ins the editor places by hand on an arbitrary range, alongside
# the whole-phrase zoom that an emphasis level produces. Both end up as
# zoom actions; this one exists because the moment worth punching into
# is not always a whole sentence.
"zooms": [],
"errors": errors,
}
def _coerce_zoom(raw: Any) -> Optional[Dict[str, float]]:
"""Normalize one manually placed zoom range."""
if not isinstance(raw, dict):
return None
try:
start = float(raw.get("start"))
end = float(raw.get("end"))
except (TypeError, ValueError):
return None
if end - start < MIN_ZOOM_DURATION:
return None
return {"start": start, "end": end}
def _coerce_phrase(raw: Any, index: int) -> Optional[Dict[str, Any]]:
"""Normalize one edited phrase row coming back from the UI."""
if not isinstance(raw, dict):
return None
try:
start = float(raw.get("start"))
end = float(raw.get("end"))
except (TypeError, ValueError):
return None
if end <= start:
return None
try:
emphasis = int(raw.get("emphasis", 0))
except (TypeError, ValueError):
emphasis = 0
try:
trim_start = float(raw.get("trim_start", start))
trim_end = float(raw.get("trim_end", end))
except (TypeError, ValueError):
trim_start, trim_end = start, end
# A trim that escaped the phrase, or inverted, is treated as no trim at all:
# the UI is the only thing that writes these, and silently discarding a bad
# pair keeps a rounding slip from deleting material the editor kept.
if not (start <= trim_start < trim_end <= end):
trim_start, trim_end = start, end
track = str(raw.get("track", TRACK_SCRIPT))
return {
"index": int(raw.get("index", index)),
"start": start,
"end": end,
"trim_start": trim_start,
"trim_end": trim_end,
"text": str(raw.get("text", "")).strip(),
"speaker": str(raw.get("speaker", "")),
"active": bool(raw.get("active", True)),
"emphasis": min(3, max(0, emphasis)),
"track": track if track in TRACKS else TRACK_SCRIPT,
"reason": str(raw.get("reason", "")),
}
def phrase_review_to_actions(review: dict) -> dict:
"""Turn an edited review back into the action list the applier consumes.
Every deactivated phrase becomes a ``cut``, a trimmed one becomes a cut over
the head and/or tail it lost, and every emphasized one becomes a ``zoom``
scaled by its level. The emphasis flags ride along in ``emphasis_spans`` so
the caption step can give those lines the dynamic treatment and everything
else the plain one, without re-deriving the decision from the acoustics.
"""
phrases = [
coerced
for index, raw in enumerate(review.get("phrases", []))
if (coerced := _coerce_phrase(raw, index)) is not None
]
actions: List[dict] = []
emphasis_spans: List[dict] = []
inactive_run: List[dict] = []
def flush_inactive_run() -> None:
"""One cut per RUN of consecutive deactivated phrases, not one per
phrase. A phrase-by-phrase cut leaves the pause BETWEEN two
deactivated phrases uncut — that gap was never anyone's content, so
nothing asked for it to survive, but it does anyway: a 0.1-0.5s
sliver clip in the final timeline for every such gap. Spanning the
whole run absorbs those gaps into the one cut."""
if not inactive_run:
return
if len(inactive_run) == 1:
reason = inactive_run[0]["reason"] or "desativada na revisão"
else:
reason = (
f"desativadas na revisão ({len(inactive_run)} frases): "
+ "; ".join(p["text"][:40] for p in inactive_run if p["text"])
)
actions.append(
VoiceAction(
kind="cut",
start=inactive_run[0]["start"],
end=inactive_run[-1]["end"],
reason=reason,
speaker=inactive_run[0]["speaker"],
).as_dict()
)
inactive_run.clear()
for phrase in phrases:
if not phrase["active"]:
inactive_run.append(phrase)
continue
flush_inactive_run()
# Head and tail the editor trimmed off — each becomes its own cut, so a
# false start disappears without taking the line with it.
for trim_start, trim_end, where in (
(phrase["start"], phrase["trim_start"], "início"),
(phrase["trim_end"], phrase["end"], "fim"),
):
if trim_end - trim_start <= 0:
continue
actions.append(
VoiceAction(
kind="cut",
start=trim_start,
end=trim_end,
reason=f"trecho do {where} da frase removido na revisão",
speaker=phrase["speaker"],
).as_dict()
)
if phrase["emphasis"] >= 1:
actions.append(
VoiceAction(
kind="zoom",
start=phrase["trim_start"],
end=phrase["trim_end"],
params={"scale": ZOOM_SCALE_BY_LEVEL[phrase["emphasis"]]},
reason=phrase["reason"] or f"ênfase nível {phrase['emphasis']}",
speaker=phrase["speaker"],
).as_dict()
)
emphasis_spans.append(
{
"start": phrase["trim_start"],
"end": phrase["trim_end"],
"level": phrase["emphasis"],
"text": phrase["text"],
}
)
flush_inactive_run()
# Hand-placed punch-ins carry no scale on purpose: an omitted scale lets the
# applier use the shape configured in "Análise de Voz" (zoom_scale, ease in
# and out), so changing that setting restyles every manual zoom instead of
# leaving a scale frozen into each one at the moment it was drawn.
for raw in review.get("zooms", []):
zoom = _coerce_zoom(raw)
if zoom is None:
continue
actions.append(
VoiceAction(
kind="zoom",
start=zoom["start"],
end=zoom["end"],
reason="zoom marcado na revisão",
).as_dict()
)
return {
"source": review.get("source", ""),
"actions": actions,
"emphasis_spans": emphasis_spans,
}
def merge_saved_decisions(review: dict, saved: Optional[dict]) -> dict:
"""Lay a previously saved review's decisions over a freshly built one.
Only the editorial fields travel — active, emphasis, track, text, trims.
Everything else (words, emotion, energy) is re-derived from the current
analysis, so re-running the voice pass with better settings improves the
screen instead of being masked by a stale copy of itself, and the saved file
never has to carry a duplicate of data it does not own.
Phrases are matched by index *and* start time: if the analysis changed
enough to move a line, the old decision for that slot is dropped rather than
applied to a different sentence.
"""
if not saved:
return review
review["zooms"] = [
zoom for raw in saved.get("zooms", []) if (zoom := _coerce_zoom(raw)) is not None
]
by_index = {}
for raw in saved.get("phrases", []):
if isinstance(raw, dict) and "index" in raw:
by_index[raw["index"]] = raw
for phrase in review["phrases"]:
previous = by_index.get(phrase["index"])
if previous is None:
continue
if abs(float(previous.get("start", -1)) - phrase["start"]) > 0.25:
continue
phrase["active"] = bool(previous.get("active", phrase["active"]))
phrase["emphasis"] = min(3, max(0, int(previous.get("emphasis", phrase["emphasis"]))))
track = str(previous.get("track", phrase["track"]))
phrase["track"] = track if track in TRACKS else phrase["track"]
if previous.get("text"):
phrase["text"] = str(previous["text"])
trim_start = float(previous.get("trim_start", phrase["trim_start"]))
trim_end = float(previous.get("trim_end", phrase["trim_end"]))
if phrase["start"] <= trim_start < trim_end <= phrase["end"]:
phrase["trim_start"], phrase["trim_end"] = trim_start, trim_end
return review
def review_paths(voice_timeline_path: str) -> Tuple[Path, Path]:
"""Where the review and its derived actions live, next to the timeline.
Both files sit beside the ``_voice_timeline.json`` they came from and are
named after it, so a project folder stays readable and re-running the wizard
on the same take overwrites its own files instead of accumulating copies.
"""
base = Path(voice_timeline_path)
stem = base.stem
if stem.endswith("_voice_timeline"):
stem = stem[: -len("_voice_timeline")]
return (
base.with_name(f"{stem}_phrase_review.json"),
base.with_name(f"{stem}_phrase_actions.json"),
)
def save_phrase_review(voice_timeline_path: str, review: dict) -> Tuple[Path, Path]:
"""Write the edited review and the actions derived from it. Returns both paths."""
review_path, actions_path = review_paths(voice_timeline_path)
review_path.write_text(
json.dumps(review, ensure_ascii=False, indent=2), encoding="utf-8"
)
actions_path.write_text(
json.dumps(phrase_review_to_actions(review), ensure_ascii=False, indent=2),
encoding="utf-8",
)
return review_path, actions_path
def load_phrase_review(voice_timeline_path: str) -> Optional[dict]:
"""The review saved earlier for this timeline, or ``None`` if there is none."""
review_path, _ = review_paths(voice_timeline_path)
if not review_path.is_file():
return None
try:
data = json.loads(review_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
return None
return data if isinstance(data, dict) else None