feat: etapa 5 do assistente — revisão de ênfases com timeline

Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-19 21:29:27 -04:00
co-authored by Claude Opus 5
parent e7748c2c58
commit 1bebee4359
31 changed files with 4622 additions and 83 deletions
+265 -16
View File
@@ -51,6 +51,23 @@ Commands:
`refine_voice_timeline` never has to reopen the audio later.
-> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."}
build_phrase_review {"voice_timeline": "..._voice_timeline.json",
"actions": {...}|[...]|null, "fresh": false}
The reviewable script for the wizard's emphasis step: every phrase with
the AI's decision already applied (active/emphasis/trim). A review saved
earlier for the same timeline is returned as-is unless `fresh` is true.
-> {"ok": true, "reused": bool, "source", "duration", "speakers",
"phrases": [{index, start, end, trim_start, trim_end, text, speaker,
active, emphasis (0-3), track, peak_emphasis,
take_boundary, gap_before, reason, words}],
"errors": [...]}
save_phrase_review {"voice_timeline": "...", "phrases": [...], "source": "...",
"duration": 0.0, "speakers": [...]}
Writes _phrase_review.json plus the _phrase_actions.json derived from it.
-> {"ok": true, "review_path", "actions_path", "emphasis_count",
"removed_count"}
dynamic_subtitle_config {}
-> {"ok": true, "band_height", "block_center_y", "line_gap", "font",
"font_size", "emphasis_font", "emphasis_face", "emphasis_size",
@@ -119,6 +136,10 @@ Commands:
-> {"ok": true, "diarization": bool, "diarization_message": "...",
"num_speakers": "..."}
acoustics_capability
Whether librosa (pitch/energy for voice analysis) is installed.
-> {"ok": true, "available": bool, "message": "..."}
voice_analysis
-> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85,
"emphasis_weights": {...}, "emotion_enabled": false,
@@ -165,6 +186,7 @@ from fcpxml.model_manager import ( # noqa: E402
load_dynamic_subtitle_config,
load_hf_token,
load_num_speakers,
load_plain_subtitle_config,
load_project_config,
load_selected_model,
load_silence_config,
@@ -175,6 +197,7 @@ from fcpxml.model_manager import ( # noqa: E402
save_hf_token,
save_models_dir,
save_num_speakers,
save_plain_subtitle_config,
save_project_config,
save_selected_model,
save_silence_config,
@@ -200,6 +223,28 @@ def _derived_output(path: str, suffix: str, args: dict) -> str:
from server import generate_output_path
return generate_output_path(path, suffix)
def _is_no_change_message(message: str) -> bool:
"""Whether a tool completed cleanly without needing to save a new file."""
text = message.lower()
return any(
token in text
for token in (
"no cuts to make",
"no silence",
"file unchanged",
"nothing saved",
)
)
def _emit_no_change_or_error(path: str, message: str) -> int:
if _is_no_change_message(message):
_emit({"ok": True, "path": path, "unchanged": True, "message": message})
return 0
_emit({"ok": False, "error": message})
return 1
# Download cancellation events, keyed by model name.
_CANCEL: dict[str, threading.Event] = {}
_LOCK = threading.Lock()
@@ -243,6 +288,42 @@ def _save_json_atomic(path: Path, data: Any) -> None:
json.load(fh)
def _project_media_paths(path: str) -> list[str]:
proj = parse_fcpxml(path)
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
media_paths: list[str] = []
if tl is not None:
for clip in getattr(tl, "clips", []):
mp = media_src_to_path(clip.media_path or "")
if mp and Path(mp).is_file() and mp not in media_paths:
media_paths.append(mp)
return media_paths
def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path:
p = Path(media_path)
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
return directory / f"{p.stem}_voice_timeline.json"
return p.with_name(p.stem + "_voice_timeline.json")
def _load_cached_voice_timeline(json_path: Path, media_path: str) -> dict | None:
try:
with open(json_path, encoding="utf-8") as fh:
data = json.load(fh)
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
return None
if not isinstance(data, dict):
return None
if data.get("source") != Path(media_path).name:
return None
if not isinstance(data.get("segments"), list):
return None
return data
# ── commands ────────────────────────────────────────────────────────────────
@@ -350,8 +431,7 @@ def cmd_remove_silences(args: dict) -> int:
contents = asyncio.run(handle_remove_media_silence({**args, "filepath": path, "output_path": output}))
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists():
_emit({"ok": False, "error": message})
return 1
return _emit_no_change_or_error(path, message)
_emit({"ok": True, "path": output, "message": message})
return 0
except Exception as exc:
@@ -398,8 +478,7 @@ def cmd_remove_filler_words(args: dict) -> int:
contents = asyncio.run(handle_remove_filler_words({**args, "filepath": path, "output_path": output}))
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists():
_emit({"ok": False, "error": message})
return 1
return _emit_no_change_or_error(path, message)
_emit({"ok": True, "path": output, "message": message})
return 0
except Exception as exc:
@@ -454,6 +533,29 @@ def cmd_generate_dynamic_subtitles(args: dict) -> int:
return 1
def cmd_generate_plain_subtitles(args: dict) -> int:
"""Generate simple static editable subtitle title clips."""
path = str(args.get("path", ""))
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
try:
from server import handle_generate_plain_subtitles
output = _derived_output(path, "_plain_subtitles", args)
contents = asyncio.run(
handle_generate_plain_subtitles({**args, "filepath": path, "output_path": output})
)
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists():
return _emit_no_change_or_error(path, message)
_emit({"ok": True, "path": output, "message": message})
return 0
except Exception as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_add_zoom(args: dict) -> int:
"""Add an ease-in/ease-out punch-in zoom to one clip."""
path = str(args.get("path", ""))
@@ -720,17 +822,10 @@ def cmd_analyze_voice(args: dict) -> int:
num_speakers = str(args.get("num_speakers") or load_num_speakers() or "")
try:
proj = parse_fcpxml(path)
media_paths = _project_media_paths(path)
except Exception as exc:
_emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
return 1
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
media_paths: list[str] = []
if tl is not None:
for clip in getattr(tl, "clips", []):
mp = media_src_to_path(clip.media_path or "")
if mp and Path(mp).is_file() and mp not in media_paths:
media_paths.append(mp)
if not media_paths:
_emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."})
return 1
@@ -738,17 +833,41 @@ def cmd_analyze_voice(args: dict) -> int:
from server import handle_build_voice_timeline
messages: list[str] = []
output_dir = str(args.get("output_dir") or "").strip()
existing: list[Path] = []
for mp in media_paths:
timeline_path = _voice_timeline_json_path(mp, output_dir)
if _load_cached_voice_timeline(timeline_path, mp) is not None:
existing.append(timeline_path)
if existing and len(existing) == len(media_paths) and not bool(args.get("force_reprocess", False)):
message = "# Voice Timeline Cache\n\n"
message += "Reaproveitando análise de voz existente. Nada foi reprocessado.\n\n"
for timeline_path in existing:
message += f"- **Timeline JSON**: {timeline_path}\n"
_emit({
"ok": True,
"path": path,
"reused": True,
"timelines": [str(p) for p in existing],
"message": message,
})
return 0
for mp in media_paths:
transcript_path = _transcript_json_path(mp, output_dir)
reused_prefix = ""
if _load_cached_transcript(transcript_path) is not None:
reused_prefix = f"# Cache\n\nReaproveitando transcrição existente: `{transcript_path}`\n\n"
try:
contents = asyncio.run(handle_build_voice_timeline({
"media_path": mp, "model": model, "language": language,
"hf_token": token, "num_speakers": num_speakers,
"output_dir": args.get("output_dir"),
"output_dir": output_dir,
}))
except Exception as exc:
_emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"})
return 1
messages.append("\n".join(getattr(c, "text", str(c)) for c in contents))
messages.append(reused_prefix + "\n".join(getattr(c, "text", str(c)) for c in contents))
_emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)})
return 0
@@ -938,9 +1057,23 @@ def cmd_set_diarization(args: dict) -> int:
return 0
def cmd_acoustics_capability(args: dict) -> int:
"""Whether librosa (pitch/energy extraction) is installed in this venv.
Surfaces `features_capability()` — previously computed but never
exposed to the app, so `layers.acoustics: false` in a voice timeline
had no explanation the user could act on.
"""
from fcpxml.voice_features import features_capability
ok, msg = features_capability()
_emit({"ok": True, "available": ok, "message": msg})
return 0
def cmd_voice_analysis(args: dict) -> int:
"""Read the persisted voice-analysis settings (energy/emphasis/emotion)."""
_emit({"ok": True, **load_voice_analysis_config()})
config = load_voice_analysis_config()
_emit({"ok": True, **config, "emphasis_threshold": config["emphasis_floor"]})
return 0
@@ -950,9 +1083,13 @@ def cmd_set_voice_analysis(args: dict) -> int:
config = save_voice_analysis_config(
energy_threshold=args.get("energy_threshold"),
emphasis_weights=weights if isinstance(weights, dict) else None,
emphasis_threshold=args.get("emphasis_threshold"),
emphasis_floor=args.get("emphasis_threshold"),
emotion_enabled=args.get("emotion_enabled"),
emotion_sensitivity=args.get("emotion_sensitivity"),
zoom_scale=args.get("zoom_scale"),
zoom_mode=args.get("zoom_mode"),
zoom_ease_in=args.get("zoom_ease_in"),
zoom_ease_out=args.get("zoom_ease_out"),
)
_emit({"ok": True, **config})
return 0
@@ -977,6 +1114,24 @@ def cmd_set_dynamic_subtitle_config(args: dict) -> int:
return 0
def cmd_plain_subtitle_config(args: dict) -> int:
"""Read the persisted simple subtitle style."""
_emit({"ok": True, **load_plain_subtitle_config()})
return 0
def cmd_set_plain_subtitle_config(args: dict) -> int:
"""Persist simple subtitle style fields. Only the given fields change."""
config = save_plain_subtitle_config(**{
k: args.get(k) for k in (
"font", "font_size", "font_color", "max_words",
"position_y", "uppercase", "keep_punctuation", "text_scale",
)
})
_emit({"ok": True, **config})
return 0
def cmd_silence_config(args: dict) -> int:
"""Read the persisted silence thresholds (noise floor, duration, padding)."""
_emit({"ok": True, **load_silence_config()})
@@ -1026,6 +1181,14 @@ def cmd_apply_voice_actions(args: dict) -> int:
return 1
actions = loaded.get("actions") if isinstance(loaded, dict) else loaded
# The documented output format is {"source": ..., "actions": [...]} —
# callers passing that whole object inline (e.g. the wizard pasting the
# skill's JSON verbatim) need the same unwrap the actions_path branch
# above already does, or a well-formed payload gets rejected as
# "malformed" for having one extra layer of nesting.
if isinstance(actions, dict):
actions = actions.get("actions")
if not isinstance(actions, list) or not actions:
_emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."})
return 1
@@ -1054,6 +1217,86 @@ def cmd_apply_voice_actions(args: dict) -> int:
return 0
def cmd_build_phrase_review(args: dict) -> int:
"""Build the reviewable script (phrases + the AI's decisions) for the wizard.
`voice_timeline` points at the _voice_timeline.json; `actions` carries the
decision list the model returned (inline, in any of the shapes the skill
emits). The review is always rebuilt from the current analysis, then the
decisions saved on a previous visit are laid back over it — reopening the
step must show the edits the user left there without freezing the acoustics
as they were when they left.
"""
from fcpxml.phrase_review import (
build_phrase_review,
load_phrase_review,
merge_saved_decisions,
)
timeline_path = str(args.get("voice_timeline", ""))
if not timeline_path or not Path(timeline_path).exists():
_emit({"ok": False, "error": "Análise de voz (voice_timeline.json) não encontrada."})
return 1
try:
with open(timeline_path, encoding="utf-8") as fh:
timeline = json.load(fh)
except (OSError, ValueError) as exc:
_emit({"ok": False, "error": f"Erro ao ler a análise de voz: {exc}"})
return 1
extra = [d for d in (args.get("output_dir"), args.get("media_dir")) if d]
review = build_phrase_review(
timeline,
args.get("actions"),
voice_timeline_path=timeline_path,
extra_dirs=extra,
)
saved = None if args.get("fresh") else load_phrase_review(timeline_path)
review = merge_saved_decisions(review, saved)
_emit({"ok": True, "reused": saved is not None, **review})
return 0
def cmd_save_phrase_review(args: dict) -> int:
"""Persist the edited review and the actions derived from it."""
from fcpxml.phrase_review import save_phrase_review
timeline_path = str(args.get("voice_timeline", ""))
if not timeline_path:
_emit({"ok": False, "error": "Caminho da análise de voz não informado."})
return 1
phrases = args.get("phrases")
if not isinstance(phrases, list):
_emit({"ok": False, "error": "Nenhuma frase para salvar."})
return 1
review = {
"version": args.get("version", "1.0"),
"source": args.get("source", ""),
"duration": args.get("duration", 0.0),
"speakers": args.get("speakers", []),
"phrases": phrases,
"zooms": args.get("zooms", []),
}
try:
review_path, actions_path = save_phrase_review(timeline_path, review)
except OSError as exc:
_emit({"ok": False, "error": f"Erro ao salvar a revisão: {exc}"})
return 1
_emit({
"ok": True,
"review_path": str(review_path),
"actions_path": str(actions_path),
"emphasis_count": sum(1 for p in phrases if int(p.get("emphasis", 0) or 0) >= 1),
"removed_count": sum(1 for p in phrases if not p.get("active", True)),
})
return 0
def cmd_project_config(args: dict) -> int:
"""Read the last project folder/file the app was working on."""
_emit({"ok": True, **load_project_config()})
@@ -1115,17 +1358,23 @@ def main() -> int:
"remove_filler_words": cmd_remove_filler_words,
"transcript_markers": cmd_transcript_markers,
"generate_dynamic_subtitles": cmd_generate_dynamic_subtitles,
"generate_plain_subtitles": cmd_generate_plain_subtitles,
"add_zoom": cmd_add_zoom,
"zoom_clips": cmd_zoom_clips,
"zoom_segments": cmd_zoom_segments,
"rename_speakers": cmd_rename_speakers,
"set_diarization": cmd_set_diarization,
"acoustics_capability": cmd_acoustics_capability,
"voice_analysis": cmd_voice_analysis,
"set_voice_analysis": cmd_set_voice_analysis,
"analyze_voice": cmd_analyze_voice,
"dynamic_subtitle_config": cmd_dynamic_subtitle_config,
"set_dynamic_subtitle_config": cmd_set_dynamic_subtitle_config,
"plain_subtitle_config": cmd_plain_subtitle_config,
"set_plain_subtitle_config": cmd_set_plain_subtitle_config,
"apply_voice_actions": cmd_apply_voice_actions,
"build_phrase_review": cmd_build_phrase_review,
"save_phrase_review": cmd_save_phrase_review,
"project_config": cmd_project_config,
"set_project_config": cmd_set_project_config,
"silence_config": cmd_silence_config,