feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e7748c2c58
commit
1bebee4359
@@ -382,6 +382,10 @@ DEFAULT_VOICE_ANALYSIS_CONFIG: dict = {
|
||||
"emphasis_floor": 0.25,
|
||||
"emotion_enabled": False,
|
||||
"emotion_sensitivity": 0.5,
|
||||
"zoom_scale": 1.30,
|
||||
"zoom_mode": "in_out",
|
||||
"zoom_ease_in": 0.25,
|
||||
"zoom_ease_out": 0.04,
|
||||
}
|
||||
|
||||
|
||||
@@ -402,12 +406,28 @@ def load_voice_analysis_config() -> dict:
|
||||
stored = _load_config().get("voice_analysis")
|
||||
if not isinstance(stored, dict):
|
||||
return cfg
|
||||
for key in ("energy_threshold", "peak_percentile", "emphasis_floor", "emotion_sensitivity"):
|
||||
for key in (
|
||||
"energy_threshold", "peak_percentile", "emphasis_floor",
|
||||
"emotion_sensitivity", "zoom_scale", "zoom_ease_in", "zoom_ease_out",
|
||||
):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = max(0.0, min(1.0, float(stored[key])))
|
||||
value = float(stored[key])
|
||||
if key == "zoom_scale":
|
||||
cfg[key] = max(1.0, min(3.0, value))
|
||||
elif key.startswith("zoom_ease"):
|
||||
cfg[key] = max(0.01, min(5.0, value))
|
||||
else:
|
||||
cfg[key] = max(0.0, min(1.0, value))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if "emphasis_threshold" in stored and "emphasis_floor" not in stored:
|
||||
try:
|
||||
cfg["emphasis_floor"] = max(0.0, min(1.0, float(stored["emphasis_threshold"])))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if stored.get("zoom_mode") in ("in_out", "in", "out"):
|
||||
cfg["zoom_mode"] = stored["zoom_mode"]
|
||||
if "emotion_enabled" in stored:
|
||||
cfg["emotion_enabled"] = bool(stored["emotion_enabled"])
|
||||
weights = stored.get("emphasis_weights")
|
||||
@@ -428,6 +448,10 @@ def save_voice_analysis_config(
|
||||
emphasis_floor: float | None = None,
|
||||
emotion_enabled: bool | None = None,
|
||||
emotion_sensitivity: float | None = None,
|
||||
zoom_scale: float | None = None,
|
||||
zoom_mode: str | None = None,
|
||||
zoom_ease_in: float | None = None,
|
||||
zoom_ease_out: float | None = None,
|
||||
) -> dict:
|
||||
"""Persist voice-analysis thresholds/weights. Only given fields change.
|
||||
|
||||
@@ -446,6 +470,14 @@ def save_voice_analysis_config(
|
||||
cfg["emotion_enabled"] = bool(emotion_enabled)
|
||||
if emotion_sensitivity is not None:
|
||||
cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity)))
|
||||
if zoom_scale is not None:
|
||||
cfg["zoom_scale"] = max(1.0, min(3.0, float(zoom_scale)))
|
||||
if zoom_mode in ("in_out", "in", "out"):
|
||||
cfg["zoom_mode"] = zoom_mode
|
||||
if zoom_ease_in is not None:
|
||||
cfg["zoom_ease_in"] = max(0.01, min(5.0, float(zoom_ease_in)))
|
||||
if zoom_ease_out is not None:
|
||||
cfg["zoom_ease_out"] = max(0.01, min(5.0, float(zoom_ease_out)))
|
||||
if emphasis_weights is not None:
|
||||
for key, value in emphasis_weights.items():
|
||||
if key in cfg["emphasis_weights"] and value is not None:
|
||||
@@ -536,6 +568,73 @@ def save_dynamic_subtitle_config(**fields) -> dict:
|
||||
return cfg
|
||||
|
||||
|
||||
DEFAULT_PLAIN_SUBTITLE_CONFIG: dict = {
|
||||
"font": "Helvetica Neue",
|
||||
"font_size": 82,
|
||||
"font_color": "1 1 1 1",
|
||||
"max_words": 7,
|
||||
"position_y": -820.0,
|
||||
"uppercase": False,
|
||||
"keep_punctuation": True,
|
||||
"text_scale": 2.0,
|
||||
}
|
||||
|
||||
|
||||
def load_plain_subtitle_config() -> dict:
|
||||
"""Persisted style for simple editable FCPXML title subtitles."""
|
||||
cfg = dict(DEFAULT_PLAIN_SUBTITLE_CONFIG)
|
||||
stored = _load_config().get("plain_subtitles")
|
||||
if not isinstance(stored, dict):
|
||||
return cfg
|
||||
for key in ("position_y", "text_scale"):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = float(stored[key])
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
for key in ("font_size", "max_words"):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = int(stored[key])
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
for key in ("font", "font_color"):
|
||||
if key in stored and isinstance(stored[key], str) and stored[key]:
|
||||
cfg[key] = stored[key]
|
||||
for key in ("uppercase", "keep_punctuation"):
|
||||
if key in stored:
|
||||
cfg[key] = bool(stored[key])
|
||||
cfg["max_words"] = max(1, int(cfg["max_words"]))
|
||||
return cfg
|
||||
|
||||
|
||||
def save_plain_subtitle_config(**fields) -> dict:
|
||||
"""Persist simple subtitle style fields. Only given fields change."""
|
||||
cfg = load_plain_subtitle_config()
|
||||
for key, value in fields.items():
|
||||
if key not in DEFAULT_PLAIN_SUBTITLE_CONFIG or value is None:
|
||||
continue
|
||||
if isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], bool):
|
||||
cfg[key] = bool(value)
|
||||
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], float):
|
||||
try:
|
||||
cfg[key] = float(value)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], int):
|
||||
try:
|
||||
cfg[key] = int(value)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
else:
|
||||
cfg[key] = str(value)
|
||||
cfg["max_words"] = max(1, int(cfg["max_words"]))
|
||||
data = _load_config()
|
||||
data["plain_subtitles"] = cfg
|
||||
_write_config(data)
|
||||
return cfg
|
||||
|
||||
|
||||
# Mirrors the silence thresholds the detection/removal handlers use when no
|
||||
# argument is passed (server_tools/qc.py). Persisted so the app's slider and
|
||||
# any later run agree without threading three fields through every call.
|
||||
|
||||
Reference in New Issue
Block a user