feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e7748c2c58
commit
1bebee4359
+265
-16
@@ -51,6 +51,23 @@ Commands:
|
||||
`refine_voice_timeline` never has to reopen the audio later.
|
||||
-> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."}
|
||||
|
||||
build_phrase_review {"voice_timeline": "..._voice_timeline.json",
|
||||
"actions": {...}|[...]|null, "fresh": false}
|
||||
The reviewable script for the wizard's emphasis step: every phrase with
|
||||
the AI's decision already applied (active/emphasis/trim). A review saved
|
||||
earlier for the same timeline is returned as-is unless `fresh` is true.
|
||||
-> {"ok": true, "reused": bool, "source", "duration", "speakers",
|
||||
"phrases": [{index, start, end, trim_start, trim_end, text, speaker,
|
||||
active, emphasis (0-3), track, peak_emphasis,
|
||||
take_boundary, gap_before, reason, words}],
|
||||
"errors": [...]}
|
||||
|
||||
save_phrase_review {"voice_timeline": "...", "phrases": [...], "source": "...",
|
||||
"duration": 0.0, "speakers": [...]}
|
||||
Writes _phrase_review.json plus the _phrase_actions.json derived from it.
|
||||
-> {"ok": true, "review_path", "actions_path", "emphasis_count",
|
||||
"removed_count"}
|
||||
|
||||
dynamic_subtitle_config {}
|
||||
-> {"ok": true, "band_height", "block_center_y", "line_gap", "font",
|
||||
"font_size", "emphasis_font", "emphasis_face", "emphasis_size",
|
||||
@@ -119,6 +136,10 @@ Commands:
|
||||
-> {"ok": true, "diarization": bool, "diarization_message": "...",
|
||||
"num_speakers": "..."}
|
||||
|
||||
acoustics_capability
|
||||
Whether librosa (pitch/energy for voice analysis) is installed.
|
||||
-> {"ok": true, "available": bool, "message": "..."}
|
||||
|
||||
voice_analysis
|
||||
-> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85,
|
||||
"emphasis_weights": {...}, "emotion_enabled": false,
|
||||
@@ -165,6 +186,7 @@ from fcpxml.model_manager import ( # noqa: E402
|
||||
load_dynamic_subtitle_config,
|
||||
load_hf_token,
|
||||
load_num_speakers,
|
||||
load_plain_subtitle_config,
|
||||
load_project_config,
|
||||
load_selected_model,
|
||||
load_silence_config,
|
||||
@@ -175,6 +197,7 @@ from fcpxml.model_manager import ( # noqa: E402
|
||||
save_hf_token,
|
||||
save_models_dir,
|
||||
save_num_speakers,
|
||||
save_plain_subtitle_config,
|
||||
save_project_config,
|
||||
save_selected_model,
|
||||
save_silence_config,
|
||||
@@ -200,6 +223,28 @@ def _derived_output(path: str, suffix: str, args: dict) -> str:
|
||||
from server import generate_output_path
|
||||
return generate_output_path(path, suffix)
|
||||
|
||||
|
||||
def _is_no_change_message(message: str) -> bool:
|
||||
"""Whether a tool completed cleanly without needing to save a new file."""
|
||||
text = message.lower()
|
||||
return any(
|
||||
token in text
|
||||
for token in (
|
||||
"no cuts to make",
|
||||
"no silence",
|
||||
"file unchanged",
|
||||
"nothing saved",
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def _emit_no_change_or_error(path: str, message: str) -> int:
|
||||
if _is_no_change_message(message):
|
||||
_emit({"ok": True, "path": path, "unchanged": True, "message": message})
|
||||
return 0
|
||||
_emit({"ok": False, "error": message})
|
||||
return 1
|
||||
|
||||
# Download cancellation events, keyed by model name.
|
||||
_CANCEL: dict[str, threading.Event] = {}
|
||||
_LOCK = threading.Lock()
|
||||
@@ -243,6 +288,42 @@ def _save_json_atomic(path: Path, data: Any) -> None:
|
||||
json.load(fh)
|
||||
|
||||
|
||||
def _project_media_paths(path: str) -> list[str]:
|
||||
proj = parse_fcpxml(path)
|
||||
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
|
||||
media_paths: list[str] = []
|
||||
if tl is not None:
|
||||
for clip in getattr(tl, "clips", []):
|
||||
mp = media_src_to_path(clip.media_path or "")
|
||||
if mp and Path(mp).is_file() and mp not in media_paths:
|
||||
media_paths.append(mp)
|
||||
return media_paths
|
||||
|
||||
|
||||
def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path:
|
||||
p = Path(media_path)
|
||||
if output_dir:
|
||||
directory = Path(output_dir).expanduser()
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
return directory / f"{p.stem}_voice_timeline.json"
|
||||
return p.with_name(p.stem + "_voice_timeline.json")
|
||||
|
||||
|
||||
def _load_cached_voice_timeline(json_path: Path, media_path: str) -> dict | None:
|
||||
try:
|
||||
with open(json_path, encoding="utf-8") as fh:
|
||||
data = json.load(fh)
|
||||
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
||||
return None
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
if data.get("source") != Path(media_path).name:
|
||||
return None
|
||||
if not isinstance(data.get("segments"), list):
|
||||
return None
|
||||
return data
|
||||
|
||||
|
||||
# ── commands ────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@@ -350,8 +431,7 @@ def cmd_remove_silences(args: dict) -> int:
|
||||
contents = asyncio.run(handle_remove_media_silence({**args, "filepath": path, "output_path": output}))
|
||||
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
|
||||
if not Path(output).exists():
|
||||
_emit({"ok": False, "error": message})
|
||||
return 1
|
||||
return _emit_no_change_or_error(path, message)
|
||||
_emit({"ok": True, "path": output, "message": message})
|
||||
return 0
|
||||
except Exception as exc:
|
||||
@@ -398,8 +478,7 @@ def cmd_remove_filler_words(args: dict) -> int:
|
||||
contents = asyncio.run(handle_remove_filler_words({**args, "filepath": path, "output_path": output}))
|
||||
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
|
||||
if not Path(output).exists():
|
||||
_emit({"ok": False, "error": message})
|
||||
return 1
|
||||
return _emit_no_change_or_error(path, message)
|
||||
_emit({"ok": True, "path": output, "message": message})
|
||||
return 0
|
||||
except Exception as exc:
|
||||
@@ -454,6 +533,29 @@ def cmd_generate_dynamic_subtitles(args: dict) -> int:
|
||||
return 1
|
||||
|
||||
|
||||
def cmd_generate_plain_subtitles(args: dict) -> int:
|
||||
"""Generate simple static editable subtitle title clips."""
|
||||
path = str(args.get("path", ""))
|
||||
if not path or not Path(path).exists():
|
||||
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
|
||||
return 1
|
||||
try:
|
||||
from server import handle_generate_plain_subtitles
|
||||
|
||||
output = _derived_output(path, "_plain_subtitles", args)
|
||||
contents = asyncio.run(
|
||||
handle_generate_plain_subtitles({**args, "filepath": path, "output_path": output})
|
||||
)
|
||||
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
|
||||
if not Path(output).exists():
|
||||
return _emit_no_change_or_error(path, message)
|
||||
_emit({"ok": True, "path": output, "message": message})
|
||||
return 0
|
||||
except Exception as exc:
|
||||
_emit({"ok": False, "error": str(exc)})
|
||||
return 1
|
||||
|
||||
|
||||
def cmd_add_zoom(args: dict) -> int:
|
||||
"""Add an ease-in/ease-out punch-in zoom to one clip."""
|
||||
path = str(args.get("path", ""))
|
||||
@@ -720,17 +822,10 @@ def cmd_analyze_voice(args: dict) -> int:
|
||||
num_speakers = str(args.get("num_speakers") or load_num_speakers() or "")
|
||||
|
||||
try:
|
||||
proj = parse_fcpxml(path)
|
||||
media_paths = _project_media_paths(path)
|
||||
except Exception as exc:
|
||||
_emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
|
||||
return 1
|
||||
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
|
||||
media_paths: list[str] = []
|
||||
if tl is not None:
|
||||
for clip in getattr(tl, "clips", []):
|
||||
mp = media_src_to_path(clip.media_path or "")
|
||||
if mp and Path(mp).is_file() and mp not in media_paths:
|
||||
media_paths.append(mp)
|
||||
if not media_paths:
|
||||
_emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."})
|
||||
return 1
|
||||
@@ -738,17 +833,41 @@ def cmd_analyze_voice(args: dict) -> int:
|
||||
from server import handle_build_voice_timeline
|
||||
|
||||
messages: list[str] = []
|
||||
output_dir = str(args.get("output_dir") or "").strip()
|
||||
existing: list[Path] = []
|
||||
for mp in media_paths:
|
||||
timeline_path = _voice_timeline_json_path(mp, output_dir)
|
||||
if _load_cached_voice_timeline(timeline_path, mp) is not None:
|
||||
existing.append(timeline_path)
|
||||
if existing and len(existing) == len(media_paths) and not bool(args.get("force_reprocess", False)):
|
||||
message = "# Voice Timeline Cache\n\n"
|
||||
message += "Reaproveitando análise de voz existente. Nada foi reprocessado.\n\n"
|
||||
for timeline_path in existing:
|
||||
message += f"- **Timeline JSON**: {timeline_path}\n"
|
||||
_emit({
|
||||
"ok": True,
|
||||
"path": path,
|
||||
"reused": True,
|
||||
"timelines": [str(p) for p in existing],
|
||||
"message": message,
|
||||
})
|
||||
return 0
|
||||
|
||||
for mp in media_paths:
|
||||
transcript_path = _transcript_json_path(mp, output_dir)
|
||||
reused_prefix = ""
|
||||
if _load_cached_transcript(transcript_path) is not None:
|
||||
reused_prefix = f"# Cache\n\nReaproveitando transcrição existente: `{transcript_path}`\n\n"
|
||||
try:
|
||||
contents = asyncio.run(handle_build_voice_timeline({
|
||||
"media_path": mp, "model": model, "language": language,
|
||||
"hf_token": token, "num_speakers": num_speakers,
|
||||
"output_dir": args.get("output_dir"),
|
||||
"output_dir": output_dir,
|
||||
}))
|
||||
except Exception as exc:
|
||||
_emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"})
|
||||
return 1
|
||||
messages.append("\n".join(getattr(c, "text", str(c)) for c in contents))
|
||||
messages.append(reused_prefix + "\n".join(getattr(c, "text", str(c)) for c in contents))
|
||||
|
||||
_emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)})
|
||||
return 0
|
||||
@@ -938,9 +1057,23 @@ def cmd_set_diarization(args: dict) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_acoustics_capability(args: dict) -> int:
|
||||
"""Whether librosa (pitch/energy extraction) is installed in this venv.
|
||||
|
||||
Surfaces `features_capability()` — previously computed but never
|
||||
exposed to the app, so `layers.acoustics: false` in a voice timeline
|
||||
had no explanation the user could act on.
|
||||
"""
|
||||
from fcpxml.voice_features import features_capability
|
||||
ok, msg = features_capability()
|
||||
_emit({"ok": True, "available": ok, "message": msg})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_voice_analysis(args: dict) -> int:
|
||||
"""Read the persisted voice-analysis settings (energy/emphasis/emotion)."""
|
||||
_emit({"ok": True, **load_voice_analysis_config()})
|
||||
config = load_voice_analysis_config()
|
||||
_emit({"ok": True, **config, "emphasis_threshold": config["emphasis_floor"]})
|
||||
return 0
|
||||
|
||||
|
||||
@@ -950,9 +1083,13 @@ def cmd_set_voice_analysis(args: dict) -> int:
|
||||
config = save_voice_analysis_config(
|
||||
energy_threshold=args.get("energy_threshold"),
|
||||
emphasis_weights=weights if isinstance(weights, dict) else None,
|
||||
emphasis_threshold=args.get("emphasis_threshold"),
|
||||
emphasis_floor=args.get("emphasis_threshold"),
|
||||
emotion_enabled=args.get("emotion_enabled"),
|
||||
emotion_sensitivity=args.get("emotion_sensitivity"),
|
||||
zoom_scale=args.get("zoom_scale"),
|
||||
zoom_mode=args.get("zoom_mode"),
|
||||
zoom_ease_in=args.get("zoom_ease_in"),
|
||||
zoom_ease_out=args.get("zoom_ease_out"),
|
||||
)
|
||||
_emit({"ok": True, **config})
|
||||
return 0
|
||||
@@ -977,6 +1114,24 @@ def cmd_set_dynamic_subtitle_config(args: dict) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_plain_subtitle_config(args: dict) -> int:
|
||||
"""Read the persisted simple subtitle style."""
|
||||
_emit({"ok": True, **load_plain_subtitle_config()})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_set_plain_subtitle_config(args: dict) -> int:
|
||||
"""Persist simple subtitle style fields. Only the given fields change."""
|
||||
config = save_plain_subtitle_config(**{
|
||||
k: args.get(k) for k in (
|
||||
"font", "font_size", "font_color", "max_words",
|
||||
"position_y", "uppercase", "keep_punctuation", "text_scale",
|
||||
)
|
||||
})
|
||||
_emit({"ok": True, **config})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_silence_config(args: dict) -> int:
|
||||
"""Read the persisted silence thresholds (noise floor, duration, padding)."""
|
||||
_emit({"ok": True, **load_silence_config()})
|
||||
@@ -1026,6 +1181,14 @@ def cmd_apply_voice_actions(args: dict) -> int:
|
||||
return 1
|
||||
actions = loaded.get("actions") if isinstance(loaded, dict) else loaded
|
||||
|
||||
# The documented output format is {"source": ..., "actions": [...]} —
|
||||
# callers passing that whole object inline (e.g. the wizard pasting the
|
||||
# skill's JSON verbatim) need the same unwrap the actions_path branch
|
||||
# above already does, or a well-formed payload gets rejected as
|
||||
# "malformed" for having one extra layer of nesting.
|
||||
if isinstance(actions, dict):
|
||||
actions = actions.get("actions")
|
||||
|
||||
if not isinstance(actions, list) or not actions:
|
||||
_emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."})
|
||||
return 1
|
||||
@@ -1054,6 +1217,86 @@ def cmd_apply_voice_actions(args: dict) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_build_phrase_review(args: dict) -> int:
|
||||
"""Build the reviewable script (phrases + the AI's decisions) for the wizard.
|
||||
|
||||
`voice_timeline` points at the _voice_timeline.json; `actions` carries the
|
||||
decision list the model returned (inline, in any of the shapes the skill
|
||||
emits). The review is always rebuilt from the current analysis, then the
|
||||
decisions saved on a previous visit are laid back over it — reopening the
|
||||
step must show the edits the user left there without freezing the acoustics
|
||||
as they were when they left.
|
||||
"""
|
||||
from fcpxml.phrase_review import (
|
||||
build_phrase_review,
|
||||
load_phrase_review,
|
||||
merge_saved_decisions,
|
||||
)
|
||||
|
||||
timeline_path = str(args.get("voice_timeline", ""))
|
||||
if not timeline_path or not Path(timeline_path).exists():
|
||||
_emit({"ok": False, "error": "Análise de voz (voice_timeline.json) não encontrada."})
|
||||
return 1
|
||||
|
||||
try:
|
||||
with open(timeline_path, encoding="utf-8") as fh:
|
||||
timeline = json.load(fh)
|
||||
except (OSError, ValueError) as exc:
|
||||
_emit({"ok": False, "error": f"Erro ao ler a análise de voz: {exc}"})
|
||||
return 1
|
||||
|
||||
extra = [d for d in (args.get("output_dir"), args.get("media_dir")) if d]
|
||||
review = build_phrase_review(
|
||||
timeline,
|
||||
args.get("actions"),
|
||||
voice_timeline_path=timeline_path,
|
||||
extra_dirs=extra,
|
||||
)
|
||||
|
||||
saved = None if args.get("fresh") else load_phrase_review(timeline_path)
|
||||
review = merge_saved_decisions(review, saved)
|
||||
_emit({"ok": True, "reused": saved is not None, **review})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_save_phrase_review(args: dict) -> int:
|
||||
"""Persist the edited review and the actions derived from it."""
|
||||
from fcpxml.phrase_review import save_phrase_review
|
||||
|
||||
timeline_path = str(args.get("voice_timeline", ""))
|
||||
if not timeline_path:
|
||||
_emit({"ok": False, "error": "Caminho da análise de voz não informado."})
|
||||
return 1
|
||||
|
||||
phrases = args.get("phrases")
|
||||
if not isinstance(phrases, list):
|
||||
_emit({"ok": False, "error": "Nenhuma frase para salvar."})
|
||||
return 1
|
||||
|
||||
review = {
|
||||
"version": args.get("version", "1.0"),
|
||||
"source": args.get("source", ""),
|
||||
"duration": args.get("duration", 0.0),
|
||||
"speakers": args.get("speakers", []),
|
||||
"phrases": phrases,
|
||||
"zooms": args.get("zooms", []),
|
||||
}
|
||||
try:
|
||||
review_path, actions_path = save_phrase_review(timeline_path, review)
|
||||
except OSError as exc:
|
||||
_emit({"ok": False, "error": f"Erro ao salvar a revisão: {exc}"})
|
||||
return 1
|
||||
|
||||
_emit({
|
||||
"ok": True,
|
||||
"review_path": str(review_path),
|
||||
"actions_path": str(actions_path),
|
||||
"emphasis_count": sum(1 for p in phrases if int(p.get("emphasis", 0) or 0) >= 1),
|
||||
"removed_count": sum(1 for p in phrases if not p.get("active", True)),
|
||||
})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_project_config(args: dict) -> int:
|
||||
"""Read the last project folder/file the app was working on."""
|
||||
_emit({"ok": True, **load_project_config()})
|
||||
@@ -1115,17 +1358,23 @@ def main() -> int:
|
||||
"remove_filler_words": cmd_remove_filler_words,
|
||||
"transcript_markers": cmd_transcript_markers,
|
||||
"generate_dynamic_subtitles": cmd_generate_dynamic_subtitles,
|
||||
"generate_plain_subtitles": cmd_generate_plain_subtitles,
|
||||
"add_zoom": cmd_add_zoom,
|
||||
"zoom_clips": cmd_zoom_clips,
|
||||
"zoom_segments": cmd_zoom_segments,
|
||||
"rename_speakers": cmd_rename_speakers,
|
||||
"set_diarization": cmd_set_diarization,
|
||||
"acoustics_capability": cmd_acoustics_capability,
|
||||
"voice_analysis": cmd_voice_analysis,
|
||||
"set_voice_analysis": cmd_set_voice_analysis,
|
||||
"analyze_voice": cmd_analyze_voice,
|
||||
"dynamic_subtitle_config": cmd_dynamic_subtitle_config,
|
||||
"set_dynamic_subtitle_config": cmd_set_dynamic_subtitle_config,
|
||||
"plain_subtitle_config": cmd_plain_subtitle_config,
|
||||
"set_plain_subtitle_config": cmd_set_plain_subtitle_config,
|
||||
"apply_voice_actions": cmd_apply_voice_actions,
|
||||
"build_phrase_review": cmd_build_phrase_review,
|
||||
"save_phrase_review": cmd_save_phrase_review,
|
||||
"project_config": cmd_project_config,
|
||||
"set_project_config": cmd_set_project_config,
|
||||
"silence_config": cmd_silence_config,
|
||||
|
||||
Reference in New Issue
Block a user