feat: etapa 5 do assistente — revisão de ênfases com timeline

Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-19 21:29:27 -04:00
co-authored by Claude Opus 5
parent e7748c2c58
commit 1bebee4359
31 changed files with 4622 additions and 83 deletions
+101 -2
View File
@@ -382,6 +382,10 @@ DEFAULT_VOICE_ANALYSIS_CONFIG: dict = {
"emphasis_floor": 0.25,
"emotion_enabled": False,
"emotion_sensitivity": 0.5,
"zoom_scale": 1.30,
"zoom_mode": "in_out",
"zoom_ease_in": 0.25,
"zoom_ease_out": 0.04,
}
@@ -402,12 +406,28 @@ def load_voice_analysis_config() -> dict:
stored = _load_config().get("voice_analysis")
if not isinstance(stored, dict):
return cfg
for key in ("energy_threshold", "peak_percentile", "emphasis_floor", "emotion_sensitivity"):
for key in (
"energy_threshold", "peak_percentile", "emphasis_floor",
"emotion_sensitivity", "zoom_scale", "zoom_ease_in", "zoom_ease_out",
):
if key in stored:
try:
cfg[key] = max(0.0, min(1.0, float(stored[key])))
value = float(stored[key])
if key == "zoom_scale":
cfg[key] = max(1.0, min(3.0, value))
elif key.startswith("zoom_ease"):
cfg[key] = max(0.01, min(5.0, value))
else:
cfg[key] = max(0.0, min(1.0, value))
except (TypeError, ValueError):
pass
if "emphasis_threshold" in stored and "emphasis_floor" not in stored:
try:
cfg["emphasis_floor"] = max(0.0, min(1.0, float(stored["emphasis_threshold"])))
except (TypeError, ValueError):
pass
if stored.get("zoom_mode") in ("in_out", "in", "out"):
cfg["zoom_mode"] = stored["zoom_mode"]
if "emotion_enabled" in stored:
cfg["emotion_enabled"] = bool(stored["emotion_enabled"])
weights = stored.get("emphasis_weights")
@@ -428,6 +448,10 @@ def save_voice_analysis_config(
emphasis_floor: float | None = None,
emotion_enabled: bool | None = None,
emotion_sensitivity: float | None = None,
zoom_scale: float | None = None,
zoom_mode: str | None = None,
zoom_ease_in: float | None = None,
zoom_ease_out: float | None = None,
) -> dict:
"""Persist voice-analysis thresholds/weights. Only given fields change.
@@ -446,6 +470,14 @@ def save_voice_analysis_config(
cfg["emotion_enabled"] = bool(emotion_enabled)
if emotion_sensitivity is not None:
cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity)))
if zoom_scale is not None:
cfg["zoom_scale"] = max(1.0, min(3.0, float(zoom_scale)))
if zoom_mode in ("in_out", "in", "out"):
cfg["zoom_mode"] = zoom_mode
if zoom_ease_in is not None:
cfg["zoom_ease_in"] = max(0.01, min(5.0, float(zoom_ease_in)))
if zoom_ease_out is not None:
cfg["zoom_ease_out"] = max(0.01, min(5.0, float(zoom_ease_out)))
if emphasis_weights is not None:
for key, value in emphasis_weights.items():
if key in cfg["emphasis_weights"] and value is not None:
@@ -536,6 +568,73 @@ def save_dynamic_subtitle_config(**fields) -> dict:
return cfg
DEFAULT_PLAIN_SUBTITLE_CONFIG: dict = {
"font": "Helvetica Neue",
"font_size": 82,
"font_color": "1 1 1 1",
"max_words": 7,
"position_y": -820.0,
"uppercase": False,
"keep_punctuation": True,
"text_scale": 2.0,
}
def load_plain_subtitle_config() -> dict:
"""Persisted style for simple editable FCPXML title subtitles."""
cfg = dict(DEFAULT_PLAIN_SUBTITLE_CONFIG)
stored = _load_config().get("plain_subtitles")
if not isinstance(stored, dict):
return cfg
for key in ("position_y", "text_scale"):
if key in stored:
try:
cfg[key] = float(stored[key])
except (TypeError, ValueError):
pass
for key in ("font_size", "max_words"):
if key in stored:
try:
cfg[key] = int(stored[key])
except (TypeError, ValueError):
pass
for key in ("font", "font_color"):
if key in stored and isinstance(stored[key], str) and stored[key]:
cfg[key] = stored[key]
for key in ("uppercase", "keep_punctuation"):
if key in stored:
cfg[key] = bool(stored[key])
cfg["max_words"] = max(1, int(cfg["max_words"]))
return cfg
def save_plain_subtitle_config(**fields) -> dict:
"""Persist simple subtitle style fields. Only given fields change."""
cfg = load_plain_subtitle_config()
for key, value in fields.items():
if key not in DEFAULT_PLAIN_SUBTITLE_CONFIG or value is None:
continue
if isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], bool):
cfg[key] = bool(value)
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], float):
try:
cfg[key] = float(value)
except (TypeError, ValueError):
continue
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], int):
try:
cfg[key] = int(value)
except (TypeError, ValueError):
continue
else:
cfg[key] = str(value)
cfg["max_words"] = max(1, int(cfg["max_words"]))
data = _load_config()
data["plain_subtitles"] = cfg
_write_config(data)
return cfg
# Mirrors the silence thresholds the detection/removal handlers use when no
# argument is passed (server_tools/qc.py). Persisted so the app's slider and
# any later run agree without threading three fields through every call.
+547
View File
@@ -0,0 +1,547 @@
"""Phrase review — the human pass between the AI's decisions and the render.
A voice timeline says *how* every line was spoken; a list of voice actions says
what the model decided to do about it. Neither is reviewable on its own: the
timeline has no editorial intent, and the action list is a set of timecodes with
no text attached. This module joins them into the one view an editor can
actually judge — the script, phrase by phrase, each carrying the decision that
was made about it.
The phrase is the unit on purpose. Emphasis, in this pipeline, is not a property
of a word but of a line: an emphasized phrase gets a punch-in and a dynamic
caption, everything else gets a plain caption. Keeping the same granularity in
the review, the JSON, and the render means a toggle in the UI maps to exactly
one editorial outcome, with nothing to reconcile in between.
Trimming stays inside the phrase for the same reason. A line is rarely wrong as
a whole — it has a false start, or a trailing "né" — so each phrase carries a
``trim_start``/``trim_end`` pair that rides on word boundaries. Editing a cut
therefore means picking a word, never hunting for a frame, and a partial cut
from the model arrives as a trim instead of being rounded away.
Round-tripping is the other half of the contract. :func:`build_phrase_review`
derives the review from actions, :func:`phrase_review_to_actions` derives
actions back from the edited review, and everything the editor touched wins over
what was inferred — so re-opening the screen shows what was left there, not a
re-derivation that quietly discards the edits.
"""
import json
from pathlib import Path
from typing import Any, Dict, List, Optional, Sequence, Tuple
from .voice_actions import VoiceAction, merge_cut_ranges, parse_actions
PHRASE_REVIEW_VERSION = "1.0"
# Emphasis is stored 0-3 rather than as a float so the UI, the JSON and the
# render agree on the same discrete decision. The thresholds map the continuous
# `peak_emphasis` of the voice timeline onto those levels when the model gave no
# explicit direction for a phrase.
EMPHASIS_LEVELS = (0, 1, 2, 3)
EMPHASIS_THRESHOLDS = (0.25, 0.45, 0.65)
# Zoom scale applied per emphasis level when the review is turned back into
# actions. Level 0 never produces a zoom. The values stay inside
# voice_actions.MIN_ZOOM_SCALE..MAX_ZOOM_SCALE.
ZOOM_SCALE_BY_LEVEL = {1: 1.15, 2: 1.3, 3: 1.5}
# A phrase only survives if most of it does. Speech boundaries from a transcript
# are approximate, so a cut clipping a fraction of a second off the tail is a
# trim, not a removal — treating that as "phrase deleted" would grey out lines
# that are still fully audible.
CUT_COVERAGE_TO_DEACTIVATE = 0.6
# A punch-in shorter than this has no time to ramp in and back out — the writer
# rejects the window anyway (see the zoom ease-in/ease-out shape), so refusing
# it here turns a silent drop at render time into nothing being placed at all.
MIN_ZOOM_DURATION = 0.4
TRACK_SCRIPT = "roteiro"
TRACK_BACKSTAGE = "bastidor"
TRACKS = (TRACK_SCRIPT, TRACK_BACKSTAGE)
def resolve_source(
source: str, voice_timeline_path: str, extra_dirs: Sequence[str] = ()
) -> str:
"""The playable path for a timeline's ``source``, or "" when it's gone.
The voice timeline stores only the media's *file name* — it is written to be
read by a model, where a machine-specific absolute path is noise. That makes
it useless for opening a preview, so the file is looked up where it can
actually be: beside its own timeline JSON first (that is where
``analyze_voice`` writes it), then in whatever project folders the caller
knows about.
"""
if not source:
return ""
candidate = Path(source)
if candidate.is_absolute() and candidate.is_file():
return str(candidate)
directories = [Path(voice_timeline_path).parent] if voice_timeline_path else []
directories += [Path(d) for d in extra_dirs if d]
for directory in directories:
found = directory / candidate.name
if found.is_file():
return str(found)
return ""
def _overlap(a_start: float, a_end: float, b_start: float, b_end: float) -> float:
"""Seconds shared by two spans (0.0 when they don't touch)."""
return max(0.0, min(a_end, b_end) - max(a_start, b_start))
def _cut_coverage(
start: float, end: float, cuts: Sequence[Tuple[float, float]]
) -> float:
"""Fraction of ``start``-``end`` that falls inside ``cuts`` (0-1)."""
span = end - start
if span <= 0:
return 0.0
removed = sum(_overlap(start, end, c_start, c_end) for c_start, c_end in cuts)
return min(1.0, removed / span)
def snap_to_words(
time: float, words: Sequence[dict], fallback: float, edge: str
) -> float:
"""Move ``time`` onto the nearest word boundary of this phrase.
Trims are expressed by pointing at a word, so a trim handle that landed
mid-word would cut a syllable in half. ``edge`` is ``"in"`` (snap to word
starts) or ``"out"`` (snap to word ends); with no word timings available the
time is left as-is.
"""
boundaries = [
float(word.get("start" if edge == "in" else "end", 0.0)) for word in words
]
boundaries = [b for b in boundaries if b > 0]
if not boundaries:
return fallback
return min(boundaries, key=lambda b: abs(b - time))
def _trim_from_cuts(
start: float,
end: float,
words: Sequence[dict],
cuts: Sequence[Tuple[float, float]],
) -> Tuple[float, float]:
"""Read a partial cut over this phrase as a head/tail trim.
Only cuts that touch an edge become trims: a cut carved out of the middle of
a line has no representation here (the phrase is the unit), so it is left
for the whole-phrase coverage rule to decide.
"""
trim_start, trim_end = start, end
for cut_start, cut_end in cuts:
if _overlap(start, end, cut_start, cut_end) <= 0:
continue
if cut_start <= trim_start < cut_end < end:
trim_start = snap_to_words(cut_end, words, cut_end, "in")
if start < cut_start < trim_end <= cut_end:
trim_end = snap_to_words(cut_start, words, cut_start, "out")
if trim_end <= trim_start:
return start, end
return trim_start, trim_end
def _level_from_peak(peak: float) -> int:
"""Map a 0-1 ``peak_emphasis`` onto a 0-3 level."""
for level, threshold in enumerate(EMPHASIS_THRESHOLDS):
if peak < threshold:
return level
return 3
def _level_from_scale(scale: Optional[float]) -> int:
"""Map a zoom's scale factor back onto a 0-3 level.
The model is free to send any scale inside the allowed range, so this picks
the nearest level rather than requiring one of our own three values.
"""
if scale is None:
return 2
best = 1
smallest = None
for level, level_scale in ZOOM_SCALE_BY_LEVEL.items():
distance = abs(level_scale - float(scale))
if smallest is None or distance < smallest:
smallest, best = distance, level
return best
def _emphasis_from_actions(
start: float,
end: float,
actions: Sequence[VoiceAction],
) -> Tuple[Optional[int], str]:
"""The level the model asked for on this phrase, and why.
A ``zoom`` or ``text`` action anywhere inside the phrase is read as "this
line is the emphasis" — the model places them on the word that carries the
point, not on the whole line, so requiring a full-span match would find
nothing. Returns ``(None, "")`` when no action touches the phrase.
"""
level: Optional[int] = None
reason = ""
for action in actions:
if action.kind not in ("zoom", "text"):
continue
if _overlap(start, end, action.start, action.end) <= 0:
continue
if action.kind == "zoom":
candidate = _level_from_scale(action.params.get("scale"))
else:
candidate = 2
if level is None or candidate > level:
level = candidate
reason = action.reason
return level, reason
def _cut_reason(
start: float, end: float, actions: Sequence[VoiceAction]
) -> str:
"""The reason given for the cut that removes this phrase."""
for action in actions:
if action.kind != "cut":
continue
if _overlap(start, end, action.start, action.end) > 0 and action.reason:
return action.reason
return ""
def build_phrase_review(
timeline: dict,
actions: Any = None,
voice_timeline_path: str = "",
extra_dirs: Sequence[str] = (),
) -> dict:
"""Join a voice timeline with the AI's actions into a reviewable script.
``actions`` accepts whatever :func:`~.voice_actions.parse_actions` accepts —
a bare list, ``{"actions": [...]}``, or ``None`` when there is no AI pass and
the review starts from the acoustics alone. Malformed rows are skipped and
reported in ``errors`` rather than raising, matching the rest of the
decision pipeline.
"""
parsed, errors = parse_actions(actions) if actions else ([], [])
cuts = merge_cut_ranges(parsed)
phrases: List[dict] = []
for index, segment in enumerate(timeline.get("segments", [])):
start = float(segment.get("start", 0.0))
end = float(segment.get("end", 0.0))
peak = float(segment.get("peak_emphasis", 0.0))
take_boundary = bool(segment.get("take_boundary", False))
words = list(segment.get("words", []))
coverage = _cut_coverage(start, end, cuts)
active = coverage < CUT_COVERAGE_TO_DEACTIVATE
trim_start, trim_end = (
_trim_from_cuts(start, end, words, cuts) if active else (start, end)
)
asked_level, asked_reason = _emphasis_from_actions(start, end, parsed)
if asked_level is not None:
emphasis, reason = asked_level, asked_reason
else:
emphasis = _level_from_peak(peak)
reason = f"ênfase {peak:.2f}" if emphasis else ""
if not active:
# A removed line carries the reason it was removed; the emphasis it
# would have had is kept so re-activating it restores the decision.
reason = _cut_reason(start, end, parsed) or reason
phrases.append(
{
"index": index,
"start": round(start, 3),
"end": round(end, 3),
"trim_start": round(trim_start, 3),
"trim_end": round(trim_end, 3),
"text": str(segment.get("text", "")).strip(),
"speaker": str(segment.get("speaker", "")),
"active": active,
"emphasis": emphasis,
"track": TRACK_BACKSTAGE if (not active and take_boundary) else TRACK_SCRIPT,
"peak_emphasis": round(peak, 3),
# Delivery emotion is a heuristic over the acoustics (see
# voice_timeline._emotion_for_word) and only means anything when
# the analysis actually ran — `emotion_available` below is what
# separates "spoken flat" from "never measured".
"emotion": str(segment.get("emotion", "neutral")),
"emotion_confidence": round(
float(segment.get("emotion_confidence", 0.0)), 3
),
"take_boundary": take_boundary,
"gap_before": round(float(segment.get("gap_before", 0.0)), 3),
"reason": reason,
"words": [
{
"text": str(word.get("text", "")),
"start": round(float(word.get("start", 0.0)), 3),
"end": round(float(word.get("end", 0.0)), 3),
"energy": round(float(word.get("energy", 0.0)), 3),
"emphasis": round(float(word.get("emphasis", 0.0)), 3),
}
for word in words
],
}
)
source = timeline.get("source", "")
layers = timeline.get("layers", {}) if isinstance(timeline.get("layers"), dict) else {}
return {
"version": PHRASE_REVIEW_VERSION,
"source": source,
"source_path": resolve_source(source, voice_timeline_path, extra_dirs),
"duration": round(phrases[-1]["end"], 3) if phrases else 0.0,
"speakers": timeline.get("speakers", []),
"emotion_available": bool(layers.get("emotion", False)),
"phrases": phrases,
# Punch-ins the editor places by hand on an arbitrary range, alongside
# the whole-phrase zoom that an emphasis level produces. Both end up as
# zoom actions; this one exists because the moment worth punching into
# is not always a whole sentence.
"zooms": [],
"errors": errors,
}
def _coerce_zoom(raw: Any) -> Optional[Dict[str, float]]:
"""Normalize one manually placed zoom range."""
if not isinstance(raw, dict):
return None
try:
start = float(raw.get("start"))
end = float(raw.get("end"))
except (TypeError, ValueError):
return None
if end - start < MIN_ZOOM_DURATION:
return None
return {"start": start, "end": end}
def _coerce_phrase(raw: Any, index: int) -> Optional[Dict[str, Any]]:
"""Normalize one edited phrase row coming back from the UI."""
if not isinstance(raw, dict):
return None
try:
start = float(raw.get("start"))
end = float(raw.get("end"))
except (TypeError, ValueError):
return None
if end <= start:
return None
try:
emphasis = int(raw.get("emphasis", 0))
except (TypeError, ValueError):
emphasis = 0
try:
trim_start = float(raw.get("trim_start", start))
trim_end = float(raw.get("trim_end", end))
except (TypeError, ValueError):
trim_start, trim_end = start, end
# A trim that escaped the phrase, or inverted, is treated as no trim at all:
# the UI is the only thing that writes these, and silently discarding a bad
# pair keeps a rounding slip from deleting material the editor kept.
if not (start <= trim_start < trim_end <= end):
trim_start, trim_end = start, end
track = str(raw.get("track", TRACK_SCRIPT))
return {
"index": int(raw.get("index", index)),
"start": start,
"end": end,
"trim_start": trim_start,
"trim_end": trim_end,
"text": str(raw.get("text", "")).strip(),
"speaker": str(raw.get("speaker", "")),
"active": bool(raw.get("active", True)),
"emphasis": min(3, max(0, emphasis)),
"track": track if track in TRACKS else TRACK_SCRIPT,
"reason": str(raw.get("reason", "")),
}
def phrase_review_to_actions(review: dict) -> dict:
"""Turn an edited review back into the action list the applier consumes.
Every deactivated phrase becomes a ``cut``, a trimmed one becomes a cut over
the head and/or tail it lost, and every emphasized one becomes a ``zoom``
scaled by its level. The emphasis flags ride along in ``emphasis_spans`` so
the caption step can give those lines the dynamic treatment and everything
else the plain one, without re-deriving the decision from the acoustics.
"""
phrases = [
coerced
for index, raw in enumerate(review.get("phrases", []))
if (coerced := _coerce_phrase(raw, index)) is not None
]
actions: List[dict] = []
emphasis_spans: List[dict] = []
for phrase in phrases:
if not phrase["active"]:
actions.append(
VoiceAction(
kind="cut",
start=phrase["start"],
end=phrase["end"],
reason=phrase["reason"] or "desativada na revisão",
speaker=phrase["speaker"],
).as_dict()
)
continue
# Head and tail the editor trimmed off — each becomes its own cut, so a
# false start disappears without taking the line with it.
for trim_start, trim_end, where in (
(phrase["start"], phrase["trim_start"], "início"),
(phrase["trim_end"], phrase["end"], "fim"),
):
if trim_end - trim_start <= 0:
continue
actions.append(
VoiceAction(
kind="cut",
start=trim_start,
end=trim_end,
reason=f"trecho do {where} da frase removido na revisão",
speaker=phrase["speaker"],
).as_dict()
)
if phrase["emphasis"] >= 1:
actions.append(
VoiceAction(
kind="zoom",
start=phrase["trim_start"],
end=phrase["trim_end"],
params={"scale": ZOOM_SCALE_BY_LEVEL[phrase["emphasis"]]},
reason=phrase["reason"] or f"ênfase nível {phrase['emphasis']}",
speaker=phrase["speaker"],
).as_dict()
)
emphasis_spans.append(
{
"start": phrase["trim_start"],
"end": phrase["trim_end"],
"level": phrase["emphasis"],
"text": phrase["text"],
}
)
# Hand-placed punch-ins carry no scale on purpose: an omitted scale lets the
# applier use the shape configured in "Análise de Voz" (zoom_scale, ease in
# and out), so changing that setting restyles every manual zoom instead of
# leaving a scale frozen into each one at the moment it was drawn.
for raw in review.get("zooms", []):
zoom = _coerce_zoom(raw)
if zoom is None:
continue
actions.append(
VoiceAction(
kind="zoom",
start=zoom["start"],
end=zoom["end"],
reason="zoom marcado na revisão",
).as_dict()
)
return {
"source": review.get("source", ""),
"actions": actions,
"emphasis_spans": emphasis_spans,
}
def merge_saved_decisions(review: dict, saved: Optional[dict]) -> dict:
"""Lay a previously saved review's decisions over a freshly built one.
Only the editorial fields travel — active, emphasis, track, text, trims.
Everything else (words, emotion, energy) is re-derived from the current
analysis, so re-running the voice pass with better settings improves the
screen instead of being masked by a stale copy of itself, and the saved file
never has to carry a duplicate of data it does not own.
Phrases are matched by index *and* start time: if the analysis changed
enough to move a line, the old decision for that slot is dropped rather than
applied to a different sentence.
"""
if not saved:
return review
review["zooms"] = [
zoom for raw in saved.get("zooms", []) if (zoom := _coerce_zoom(raw)) is not None
]
by_index = {}
for raw in saved.get("phrases", []):
if isinstance(raw, dict) and "index" in raw:
by_index[raw["index"]] = raw
for phrase in review["phrases"]:
previous = by_index.get(phrase["index"])
if previous is None:
continue
if abs(float(previous.get("start", -1)) - phrase["start"]) > 0.25:
continue
phrase["active"] = bool(previous.get("active", phrase["active"]))
phrase["emphasis"] = min(3, max(0, int(previous.get("emphasis", phrase["emphasis"]))))
track = str(previous.get("track", phrase["track"]))
phrase["track"] = track if track in TRACKS else phrase["track"]
if previous.get("text"):
phrase["text"] = str(previous["text"])
trim_start = float(previous.get("trim_start", phrase["trim_start"]))
trim_end = float(previous.get("trim_end", phrase["trim_end"]))
if phrase["start"] <= trim_start < trim_end <= phrase["end"]:
phrase["trim_start"], phrase["trim_end"] = trim_start, trim_end
return review
def review_paths(voice_timeline_path: str) -> Tuple[Path, Path]:
"""Where the review and its derived actions live, next to the timeline.
Both files sit beside the ``_voice_timeline.json`` they came from and are
named after it, so a project folder stays readable and re-running the wizard
on the same take overwrites its own files instead of accumulating copies.
"""
base = Path(voice_timeline_path)
stem = base.stem
if stem.endswith("_voice_timeline"):
stem = stem[: -len("_voice_timeline")]
return (
base.with_name(f"{stem}_phrase_review.json"),
base.with_name(f"{stem}_phrase_actions.json"),
)
def save_phrase_review(voice_timeline_path: str, review: dict) -> Tuple[Path, Path]:
"""Write the edited review and the actions derived from it. Returns both paths."""
review_path, actions_path = review_paths(voice_timeline_path)
review_path.write_text(
json.dumps(review, ensure_ascii=False, indent=2), encoding="utf-8"
)
actions_path.write_text(
json.dumps(phrase_review_to_actions(review), ensure_ascii=False, indent=2),
encoding="utf-8",
)
return review_path, actions_path
def load_phrase_review(voice_timeline_path: str) -> Optional[dict]:
"""The review saved earlier for this timeline, or ``None`` if there is none."""
review_path, _ = review_paths(voice_timeline_path)
if not review_path.is_file():
return None
try:
data = json.loads(review_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
return None
return data if isinstance(data, dict) else None
+4 -1
View File
@@ -28,8 +28,11 @@ ALLOWED_MODELS = (
)
# Conservative by default: interjections that are near-universally filler.
# Portuguese "um"/"uma" are usually articles/numerals inside real phrases
# ("de um jeito") rather than discardable hesitations, so only cut them when
# the caller explicitly opts in through the fillers argument.
# "like" / "so" / "actually" are speech, not noise, unless the user opts in.
DEFAULT_FILLERS = ("um", "uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm")
DEFAULT_FILLERS = ("uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm")
_NORM_RE = re.compile(r"[^\w']+")
+24 -9
View File
@@ -82,21 +82,36 @@ def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]:
params = dict(params) if isinstance(params, dict) else {}
if kind == "zoom":
try:
scale = float(params.get("scale", 1.3))
except (TypeError, ValueError):
return None, f"{where}: zoom scale must be a number"
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
return None, (
f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}"
)
params["scale"] = scale
if "scale" in params and params.get("scale") is not None:
try:
scale = float(params["scale"])
except (TypeError, ValueError):
return None, f"{where}: zoom scale must be a number"
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
return None, (
f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}"
)
params["scale"] = scale
if kind == "text":
content = str(params.get("content", "")).strip()
if not content:
return None, f"{where}: text action needs params.content"
params["content"] = content[:MAX_TEXT_LENGTH]
# Style is optional — omitted fields fall back to the "Legendas
# Dinâmicas" emphasis style at apply time (see _apply_placed_action),
# so a callout matches the captions' look without the caller having
# to know or repeat that configuration. Anything given here wins.
for key in ("font", "font_color", "face"):
if key in params and not isinstance(params[key], str):
del params[key]
if "font_size" in params:
try:
params["font_size"] = int(params["font_size"])
except (TypeError, ValueError):
del params["font_size"]
if "bold" in params:
params["bold"] = bool(params["bold"])
return (
VoiceAction(
+89
View File
@@ -56,12 +56,20 @@ VALUE_SCALES = {
"rate_delta": "0-1, how much the local speaking rate departs from the average",
"pause_before": "seconds of silence immediately before the word",
"emphasis": "0-1 combined index; high values are punch-in/highlight candidates",
"emotion": "heuristic label from delivery: neutral, excited, tense, calm, reflective",
"emotion_confidence": "0-1 confidence in the heuristic emotion label",
"arousal": "0-1 vocal activation from energy/rate/pitch movement",
"valence": "0-1 rough positive tone; lower values suggest tension/weight",
},
"segment": {
"gap_before": "seconds of silence before this line",
"take_boundary": "true when the gap is long enough that the take likely restarted here",
"avg_energy": "0-1 mean loudness across the line",
"peak_emphasis": "0-1 highest emphasis of any word in the line",
"emotion": "dominant delivery emotion across the line",
"emotion_confidence": "0-1 confidence in the dominant segment emotion",
"arousal": "0-1 mean vocal activation across the line",
"valence": "0-1 mean rough positive tone across the line",
},
}
@@ -92,11 +100,74 @@ def _round_word(word: dict) -> dict:
"rate_delta": round(word.get("rate_delta", 0.0), 3),
"pause_before": round(word.get("pause_before", 0.0), 3),
"emphasis": round(word.get("emphasis", 0.0), 3),
"emotion": word.get("emotion", "neutral"),
"emotion_confidence": round(word.get("emotion_confidence", 0.0), 3),
"arousal": round(word.get("arousal", 0.0), 3),
"valence": round(word.get("valence", 0.5), 3),
"energy_raw": word.get("energy"),
"pitch_hz": word.get("pitch_hz"),
}
def _emotion_for_word(word: dict, enabled: bool, sensitivity: float) -> dict:
"""Classify delivery emotion from normalized acoustic features.
This is deliberately a local heuristic rather than a claimed clinical
emotion model. It gives the editor a useful signal about delivery shape
while degrading predictably when acoustic extraction is unavailable.
"""
if not enabled:
return {
"emotion": "neutral",
"emotion_confidence": 0.0,
"arousal": 0.0,
"valence": 0.5,
}
energy = float(word.get("energy_norm", 0.0))
pitch = float(word.get("pitch_delta", 0.0))
rate = float(word.get("rate_delta", 0.0))
pause = min(float(word.get("pause_before", 0.0)) / 2.0, 1.0)
emphasis = float(word.get("emphasis", 0.0))
arousal = max(0.0, min(1.0, energy * 0.45 + pitch * 0.25 + rate * 0.20 + emphasis * 0.10))
valence = max(0.0, min(1.0, 0.55 + energy * 0.15 - pause * 0.20 - rate * 0.10))
if arousal >= 0.68 and valence >= 0.50:
label = "excited"
confidence = arousal
elif arousal >= 0.58 and valence < 0.50:
label = "tense"
confidence = max(arousal, 1.0 - valence)
elif arousal <= 0.28 and pause >= 0.25:
label = "reflective"
confidence = max(1.0 - arousal, pause)
elif arousal <= 0.35:
label = "calm"
confidence = 1.0 - arousal
else:
label = "neutral"
confidence = 1.0 - abs(arousal - 0.5) * 2.0
confidence = max(0.0, min(1.0, confidence))
if confidence < sensitivity:
label = "neutral"
return {
"emotion": label,
"emotion_confidence": confidence,
"arousal": arousal,
"valence": valence,
}
def annotate_emotions(words: Sequence[dict], enabled: bool, sensitivity: float) -> List[dict]:
"""Attach heuristic emotion labels to enriched word rows."""
return [
{**w, **_emotion_for_word(w, enabled, sensitivity)}
for w in words
]
def enrich_words(
words: Sequence[dict],
pitch_track: Optional[Sequence] = None,
@@ -166,6 +237,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]
in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end]
energies = [w["energy_norm"] for w in in_seg]
emphases = [w["emphasis"] for w in in_seg]
arousals = [w.get("arousal", 0.0) for w in in_seg]
valences = [w.get("valence", 0.5) for w in in_seg]
emotions = [w.get("emotion", "neutral") for w in in_seg]
dominant = max(set(emotions), key=emotions.count) if emotions else "neutral"
emotion_confidences = [
w.get("emotion_confidence", 0.0) for w in in_seg if w.get("emotion") == dominant
]
gap = max(0.0, start - previous_end)
rows.append(
{
@@ -181,6 +259,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]
"take_boundary": gap >= TAKE_BOUNDARY_GAP,
"avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0,
"peak_emphasis": round(max(emphases), 3) if emphases else 0.0,
"emotion": dominant,
"emotion_confidence": (
round(sum(emotion_confidences) / len(emotion_confidences), 3)
if emotion_confidences else 0.0
),
"arousal": round(sum(arousals) / len(arousals), 3) if arousals else 0.0,
"valence": round(sum(valences) / len(valences), 3) if valences else 0.5,
"words": [_round_word(w) for w in in_seg],
}
)
@@ -425,6 +510,8 @@ def build_voice_timeline(
weights: EmphasisWeights = EmphasisWeights(),
peak_percentile: float = 0.02,
emphasis_floor: float = 0.25,
emotion_enabled: bool = False,
emotion_sensitivity: float = 0.5,
progress_cb: Optional[Callable[[float, str], None]] = None,
) -> dict:
"""Build the consolidated voice timeline for one media file.
@@ -445,6 +532,7 @@ def build_voice_timeline(
report(0.5, "Calculando ênfase...")
words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights)
words = annotate_emotions(words, emotion_enabled, emotion_sensitivity)
report(0.7, "Identificando participantes...")
tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None
@@ -465,6 +553,7 @@ def build_voice_timeline(
"transcript": bool(transcript.get("words")),
"acoustics": pitch_track is not None or energy_track is not None,
"speakers": tracks is not None,
"emotion": bool(emotion_enabled),
},
"scales": VALUE_SCALES,
"summary": _summary(
+63 -2
View File
@@ -2218,13 +2218,23 @@ class FCPXMLModifier:
seg_start: 'TimeValue',
seg_duration: 'TimeValue',
) -> None:
"""Remove markers/keywords from *clip* that fall outside the segment range.
"""Remove markers/keywords/titles from *clip* that fall outside the segment range.
After ``split_clip`` deepcopy's the original clip into each segment, every
segment inherits all child elements. Markers whose ``start`` falls outside
``[seg_start, seg_start + seg_duration)`` are phantom duplicates and must be
removed. Keywords that partially overlap get their ``start``/``duration``
clamped to the segment boundaries.
A lane-nested ``<title>`` (a "text" voice action's on-screen callout,
or a caption from an earlier `generate_dynamic_subtitles` pass) is
the same kind of phantom duplicate, just keyed on ``offset`` instead
of ``start`` — its offset lives in the same source-media coordinate
space as a marker's ``start`` (see ``add_text_title``/``add_marker``,
both anchored at ``parent.start``). Left unfiltered, every further
cut (silence removal, filler removal) duplicates it into every
resulting piece, so the same word shows up several times across the
edited timeline instead of once where it was placed.
"""
seg_end = seg_start + seg_duration
to_remove = []
@@ -2234,6 +2244,10 @@ class FCPXMLModifier:
child_start = TimeValue.from_timecode(child.get('start', '0s'))
if child_start < seg_start or child_start >= seg_end:
to_remove.append(child)
elif tag == 'title':
title_offset = TimeValue.from_timecode(child.get('offset', '0s'))
if title_offset < seg_start or title_offset >= seg_end:
to_remove.append(child)
elif tag == 'keyword':
kw_start = TimeValue.from_timecode(child.get('start', '0s'))
kw_dur = TimeValue.from_timecode(child.get('duration', '0s'))
@@ -2304,6 +2318,7 @@ class FCPXMLModifier:
self._filter_children_for_segment(
new_clip, current_start, segment_duration
)
self._reassign_text_style_ids(new_clip)
spine.insert(clip_index + len(new_clips), new_clip)
new_clips.append(new_clip)
@@ -2404,6 +2419,7 @@ class FCPXMLModifier:
new_clip.set('start', seg_start.to_fcpxml())
new_clip.set('duration', seg_duration.to_fcpxml())
self._filter_children_for_segment(new_clip, seg_start, seg_duration)
self._reassign_text_style_ids(new_clip)
spine.insert(clip_index + len(new_clips), new_clip)
new_clips.append(new_clip)
current_offset = current_offset + seg_duration
@@ -2946,6 +2962,7 @@ class FCPXMLModifier:
('-469658744/1000000000s', '0'),
('12328542033/1000000000s', '1'),
)
_TEXT_SIZE_KEY = '9999/10003/13260/3296672360/5/3296672362/3'
def _ensure_text_title_effect(self, resources: ET.Element) -> str:
"""Return the resource id of the "Text" (Basic Text) effect, creating it if absent."""
@@ -2995,6 +3012,34 @@ class FCPXMLModifier:
self._text_style_ids.add(candidate)
return candidate
def _reassign_text_style_ids(self, clip: ET.Element) -> None:
"""Give every ``<text-style-def>`` inside a just-deepcopy'd *clip* a
fresh document-unique id, repointing any ``<text-style ref="...">``
in the same subtree that pointed at the old one.
``split_clip``/``cut_clip_ranges`` deepcopy the clip once per
resulting segment, so a clip carrying a ``<title>`` (from a "text"
voice action) keeps the exact same ``text-style-def id`` in every
copy. A single cut is harmless — but the batch chain re-cuts the
same clip at each step (silence removal, filler removal, dynamic
subtitles), and every pass multiplies the duplicate, so the DTD
validator eventually rejects the file with "ID ... already
defined". Regenerating here, at the only place copies are made,
fixes it for every caller instead of each one having to remember to.
"""
for style_def in clip.findall('.//text-style-def'):
old_id = style_def.get('id')
if not old_id:
continue
slug = old_id[3:] if old_id.startswith('ts_') else old_id
slug = re.sub(r'_\d+$', '', slug) # drop a prior _<N> counter
new_id = self._unique_text_style_id(slug)
if new_id == old_id:
continue
style_def.set('id', new_id)
for ref_el in clip.findall(f".//text-style[@ref='{old_id}']"):
ref_el.set('ref', new_id)
def _make_text_title_clip(
self,
effect_id: str,
@@ -3012,6 +3057,8 @@ class FCPXMLModifier:
face: Optional[str] = None,
kerning: Optional[float] = None,
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
animated: bool = True,
size_param: Optional[float] = None,
) -> ET.Element:
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
@@ -3042,9 +3089,12 @@ class FCPXMLModifier:
param.set('key', key)
param.set('value', value)
animation_params = {'Opacity', 'Speed', 'Apply Speed'}
for param_name, param_key, param_value in self._TEXT_TITLE_PARAMS:
if not animated and param_name in animation_params:
continue
_add_param(param_name, param_key, param_value)
if param_name == 'Speed':
if animated and param_name == 'Speed':
# "Custom Speed" lands between "Speed" and "Apply Speed" and
# carries a <keyframeAnimation> child instead of a value.
cs = ET.SubElement(elem, 'param')
@@ -3056,6 +3106,9 @@ class FCPXMLModifier:
kf.set('time', kf_time)
kf.set('value', kf_value)
if size_param is not None:
_add_param('Size', self._TEXT_SIZE_KEY, f"{float(size_param):g}")
text_el = ET.SubElement(elem, 'text')
ts_id = self._unique_text_style_id(name)
run = ET.SubElement(text_el, 'text-style')
@@ -3109,6 +3162,10 @@ class FCPXMLModifier:
font_size: int = 196,
font_color: str = '1 1 1 1',
bold: bool = True,
face: Optional[str] = None,
animated: bool = True,
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
size_param: Optional[float] = None,
) -> ET.Element:
"""Add a single static "Text" (Basic Text) title over *parent_clip*.
@@ -3142,6 +3199,10 @@ class FCPXMLModifier:
font_size=font_size,
font_color=font_color,
bold=bold,
face=face,
animated=animated,
font_scale=font_scale,
size_param=size_param,
)
_dtd_insert(parent, title)
return title