feat: etapa 5 do assistente — revisão de ênfases com timeline

Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-19 21:29:27 -04:00
co-authored by Claude Opus 5
parent e7748c2c58
commit 1bebee4359
31 changed files with 4622 additions and 83 deletions
+101 -2
View File
@@ -382,6 +382,10 @@ DEFAULT_VOICE_ANALYSIS_CONFIG: dict = {
"emphasis_floor": 0.25,
"emotion_enabled": False,
"emotion_sensitivity": 0.5,
"zoom_scale": 1.30,
"zoom_mode": "in_out",
"zoom_ease_in": 0.25,
"zoom_ease_out": 0.04,
}
@@ -402,12 +406,28 @@ def load_voice_analysis_config() -> dict:
stored = _load_config().get("voice_analysis")
if not isinstance(stored, dict):
return cfg
for key in ("energy_threshold", "peak_percentile", "emphasis_floor", "emotion_sensitivity"):
for key in (
"energy_threshold", "peak_percentile", "emphasis_floor",
"emotion_sensitivity", "zoom_scale", "zoom_ease_in", "zoom_ease_out",
):
if key in stored:
try:
cfg[key] = max(0.0, min(1.0, float(stored[key])))
value = float(stored[key])
if key == "zoom_scale":
cfg[key] = max(1.0, min(3.0, value))
elif key.startswith("zoom_ease"):
cfg[key] = max(0.01, min(5.0, value))
else:
cfg[key] = max(0.0, min(1.0, value))
except (TypeError, ValueError):
pass
if "emphasis_threshold" in stored and "emphasis_floor" not in stored:
try:
cfg["emphasis_floor"] = max(0.0, min(1.0, float(stored["emphasis_threshold"])))
except (TypeError, ValueError):
pass
if stored.get("zoom_mode") in ("in_out", "in", "out"):
cfg["zoom_mode"] = stored["zoom_mode"]
if "emotion_enabled" in stored:
cfg["emotion_enabled"] = bool(stored["emotion_enabled"])
weights = stored.get("emphasis_weights")
@@ -428,6 +448,10 @@ def save_voice_analysis_config(
emphasis_floor: float | None = None,
emotion_enabled: bool | None = None,
emotion_sensitivity: float | None = None,
zoom_scale: float | None = None,
zoom_mode: str | None = None,
zoom_ease_in: float | None = None,
zoom_ease_out: float | None = None,
) -> dict:
"""Persist voice-analysis thresholds/weights. Only given fields change.
@@ -446,6 +470,14 @@ def save_voice_analysis_config(
cfg["emotion_enabled"] = bool(emotion_enabled)
if emotion_sensitivity is not None:
cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity)))
if zoom_scale is not None:
cfg["zoom_scale"] = max(1.0, min(3.0, float(zoom_scale)))
if zoom_mode in ("in_out", "in", "out"):
cfg["zoom_mode"] = zoom_mode
if zoom_ease_in is not None:
cfg["zoom_ease_in"] = max(0.01, min(5.0, float(zoom_ease_in)))
if zoom_ease_out is not None:
cfg["zoom_ease_out"] = max(0.01, min(5.0, float(zoom_ease_out)))
if emphasis_weights is not None:
for key, value in emphasis_weights.items():
if key in cfg["emphasis_weights"] and value is not None:
@@ -536,6 +568,73 @@ def save_dynamic_subtitle_config(**fields) -> dict:
return cfg
DEFAULT_PLAIN_SUBTITLE_CONFIG: dict = {
"font": "Helvetica Neue",
"font_size": 82,
"font_color": "1 1 1 1",
"max_words": 7,
"position_y": -820.0,
"uppercase": False,
"keep_punctuation": True,
"text_scale": 2.0,
}
def load_plain_subtitle_config() -> dict:
"""Persisted style for simple editable FCPXML title subtitles."""
cfg = dict(DEFAULT_PLAIN_SUBTITLE_CONFIG)
stored = _load_config().get("plain_subtitles")
if not isinstance(stored, dict):
return cfg
for key in ("position_y", "text_scale"):
if key in stored:
try:
cfg[key] = float(stored[key])
except (TypeError, ValueError):
pass
for key in ("font_size", "max_words"):
if key in stored:
try:
cfg[key] = int(stored[key])
except (TypeError, ValueError):
pass
for key in ("font", "font_color"):
if key in stored and isinstance(stored[key], str) and stored[key]:
cfg[key] = stored[key]
for key in ("uppercase", "keep_punctuation"):
if key in stored:
cfg[key] = bool(stored[key])
cfg["max_words"] = max(1, int(cfg["max_words"]))
return cfg
def save_plain_subtitle_config(**fields) -> dict:
"""Persist simple subtitle style fields. Only given fields change."""
cfg = load_plain_subtitle_config()
for key, value in fields.items():
if key not in DEFAULT_PLAIN_SUBTITLE_CONFIG or value is None:
continue
if isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], bool):
cfg[key] = bool(value)
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], float):
try:
cfg[key] = float(value)
except (TypeError, ValueError):
continue
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], int):
try:
cfg[key] = int(value)
except (TypeError, ValueError):
continue
else:
cfg[key] = str(value)
cfg["max_words"] = max(1, int(cfg["max_words"]))
data = _load_config()
data["plain_subtitles"] = cfg
_write_config(data)
return cfg
# Mirrors the silence thresholds the detection/removal handlers use when no
# argument is passed (server_tools/qc.py). Persisted so the app's slider and
# any later run agree without threading three fields through every call.