feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e7748c2c58
commit
1bebee4359
+265
-16
@@ -51,6 +51,23 @@ Commands:
|
||||
`refine_voice_timeline` never has to reopen the audio later.
|
||||
-> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."}
|
||||
|
||||
build_phrase_review {"voice_timeline": "..._voice_timeline.json",
|
||||
"actions": {...}|[...]|null, "fresh": false}
|
||||
The reviewable script for the wizard's emphasis step: every phrase with
|
||||
the AI's decision already applied (active/emphasis/trim). A review saved
|
||||
earlier for the same timeline is returned as-is unless `fresh` is true.
|
||||
-> {"ok": true, "reused": bool, "source", "duration", "speakers",
|
||||
"phrases": [{index, start, end, trim_start, trim_end, text, speaker,
|
||||
active, emphasis (0-3), track, peak_emphasis,
|
||||
take_boundary, gap_before, reason, words}],
|
||||
"errors": [...]}
|
||||
|
||||
save_phrase_review {"voice_timeline": "...", "phrases": [...], "source": "...",
|
||||
"duration": 0.0, "speakers": [...]}
|
||||
Writes _phrase_review.json plus the _phrase_actions.json derived from it.
|
||||
-> {"ok": true, "review_path", "actions_path", "emphasis_count",
|
||||
"removed_count"}
|
||||
|
||||
dynamic_subtitle_config {}
|
||||
-> {"ok": true, "band_height", "block_center_y", "line_gap", "font",
|
||||
"font_size", "emphasis_font", "emphasis_face", "emphasis_size",
|
||||
@@ -119,6 +136,10 @@ Commands:
|
||||
-> {"ok": true, "diarization": bool, "diarization_message": "...",
|
||||
"num_speakers": "..."}
|
||||
|
||||
acoustics_capability
|
||||
Whether librosa (pitch/energy for voice analysis) is installed.
|
||||
-> {"ok": true, "available": bool, "message": "..."}
|
||||
|
||||
voice_analysis
|
||||
-> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85,
|
||||
"emphasis_weights": {...}, "emotion_enabled": false,
|
||||
@@ -165,6 +186,7 @@ from fcpxml.model_manager import ( # noqa: E402
|
||||
load_dynamic_subtitle_config,
|
||||
load_hf_token,
|
||||
load_num_speakers,
|
||||
load_plain_subtitle_config,
|
||||
load_project_config,
|
||||
load_selected_model,
|
||||
load_silence_config,
|
||||
@@ -175,6 +197,7 @@ from fcpxml.model_manager import ( # noqa: E402
|
||||
save_hf_token,
|
||||
save_models_dir,
|
||||
save_num_speakers,
|
||||
save_plain_subtitle_config,
|
||||
save_project_config,
|
||||
save_selected_model,
|
||||
save_silence_config,
|
||||
@@ -200,6 +223,28 @@ def _derived_output(path: str, suffix: str, args: dict) -> str:
|
||||
from server import generate_output_path
|
||||
return generate_output_path(path, suffix)
|
||||
|
||||
|
||||
def _is_no_change_message(message: str) -> bool:
|
||||
"""Whether a tool completed cleanly without needing to save a new file."""
|
||||
text = message.lower()
|
||||
return any(
|
||||
token in text
|
||||
for token in (
|
||||
"no cuts to make",
|
||||
"no silence",
|
||||
"file unchanged",
|
||||
"nothing saved",
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def _emit_no_change_or_error(path: str, message: str) -> int:
|
||||
if _is_no_change_message(message):
|
||||
_emit({"ok": True, "path": path, "unchanged": True, "message": message})
|
||||
return 0
|
||||
_emit({"ok": False, "error": message})
|
||||
return 1
|
||||
|
||||
# Download cancellation events, keyed by model name.
|
||||
_CANCEL: dict[str, threading.Event] = {}
|
||||
_LOCK = threading.Lock()
|
||||
@@ -243,6 +288,42 @@ def _save_json_atomic(path: Path, data: Any) -> None:
|
||||
json.load(fh)
|
||||
|
||||
|
||||
def _project_media_paths(path: str) -> list[str]:
|
||||
proj = parse_fcpxml(path)
|
||||
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
|
||||
media_paths: list[str] = []
|
||||
if tl is not None:
|
||||
for clip in getattr(tl, "clips", []):
|
||||
mp = media_src_to_path(clip.media_path or "")
|
||||
if mp and Path(mp).is_file() and mp not in media_paths:
|
||||
media_paths.append(mp)
|
||||
return media_paths
|
||||
|
||||
|
||||
def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path:
|
||||
p = Path(media_path)
|
||||
if output_dir:
|
||||
directory = Path(output_dir).expanduser()
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
return directory / f"{p.stem}_voice_timeline.json"
|
||||
return p.with_name(p.stem + "_voice_timeline.json")
|
||||
|
||||
|
||||
def _load_cached_voice_timeline(json_path: Path, media_path: str) -> dict | None:
|
||||
try:
|
||||
with open(json_path, encoding="utf-8") as fh:
|
||||
data = json.load(fh)
|
||||
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
||||
return None
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
if data.get("source") != Path(media_path).name:
|
||||
return None
|
||||
if not isinstance(data.get("segments"), list):
|
||||
return None
|
||||
return data
|
||||
|
||||
|
||||
# ── commands ────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@@ -350,8 +431,7 @@ def cmd_remove_silences(args: dict) -> int:
|
||||
contents = asyncio.run(handle_remove_media_silence({**args, "filepath": path, "output_path": output}))
|
||||
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
|
||||
if not Path(output).exists():
|
||||
_emit({"ok": False, "error": message})
|
||||
return 1
|
||||
return _emit_no_change_or_error(path, message)
|
||||
_emit({"ok": True, "path": output, "message": message})
|
||||
return 0
|
||||
except Exception as exc:
|
||||
@@ -398,8 +478,7 @@ def cmd_remove_filler_words(args: dict) -> int:
|
||||
contents = asyncio.run(handle_remove_filler_words({**args, "filepath": path, "output_path": output}))
|
||||
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
|
||||
if not Path(output).exists():
|
||||
_emit({"ok": False, "error": message})
|
||||
return 1
|
||||
return _emit_no_change_or_error(path, message)
|
||||
_emit({"ok": True, "path": output, "message": message})
|
||||
return 0
|
||||
except Exception as exc:
|
||||
@@ -454,6 +533,29 @@ def cmd_generate_dynamic_subtitles(args: dict) -> int:
|
||||
return 1
|
||||
|
||||
|
||||
def cmd_generate_plain_subtitles(args: dict) -> int:
|
||||
"""Generate simple static editable subtitle title clips."""
|
||||
path = str(args.get("path", ""))
|
||||
if not path or not Path(path).exists():
|
||||
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
|
||||
return 1
|
||||
try:
|
||||
from server import handle_generate_plain_subtitles
|
||||
|
||||
output = _derived_output(path, "_plain_subtitles", args)
|
||||
contents = asyncio.run(
|
||||
handle_generate_plain_subtitles({**args, "filepath": path, "output_path": output})
|
||||
)
|
||||
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
|
||||
if not Path(output).exists():
|
||||
return _emit_no_change_or_error(path, message)
|
||||
_emit({"ok": True, "path": output, "message": message})
|
||||
return 0
|
||||
except Exception as exc:
|
||||
_emit({"ok": False, "error": str(exc)})
|
||||
return 1
|
||||
|
||||
|
||||
def cmd_add_zoom(args: dict) -> int:
|
||||
"""Add an ease-in/ease-out punch-in zoom to one clip."""
|
||||
path = str(args.get("path", ""))
|
||||
@@ -720,17 +822,10 @@ def cmd_analyze_voice(args: dict) -> int:
|
||||
num_speakers = str(args.get("num_speakers") or load_num_speakers() or "")
|
||||
|
||||
try:
|
||||
proj = parse_fcpxml(path)
|
||||
media_paths = _project_media_paths(path)
|
||||
except Exception as exc:
|
||||
_emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
|
||||
return 1
|
||||
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
|
||||
media_paths: list[str] = []
|
||||
if tl is not None:
|
||||
for clip in getattr(tl, "clips", []):
|
||||
mp = media_src_to_path(clip.media_path or "")
|
||||
if mp and Path(mp).is_file() and mp not in media_paths:
|
||||
media_paths.append(mp)
|
||||
if not media_paths:
|
||||
_emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."})
|
||||
return 1
|
||||
@@ -738,17 +833,41 @@ def cmd_analyze_voice(args: dict) -> int:
|
||||
from server import handle_build_voice_timeline
|
||||
|
||||
messages: list[str] = []
|
||||
output_dir = str(args.get("output_dir") or "").strip()
|
||||
existing: list[Path] = []
|
||||
for mp in media_paths:
|
||||
timeline_path = _voice_timeline_json_path(mp, output_dir)
|
||||
if _load_cached_voice_timeline(timeline_path, mp) is not None:
|
||||
existing.append(timeline_path)
|
||||
if existing and len(existing) == len(media_paths) and not bool(args.get("force_reprocess", False)):
|
||||
message = "# Voice Timeline Cache\n\n"
|
||||
message += "Reaproveitando análise de voz existente. Nada foi reprocessado.\n\n"
|
||||
for timeline_path in existing:
|
||||
message += f"- **Timeline JSON**: {timeline_path}\n"
|
||||
_emit({
|
||||
"ok": True,
|
||||
"path": path,
|
||||
"reused": True,
|
||||
"timelines": [str(p) for p in existing],
|
||||
"message": message,
|
||||
})
|
||||
return 0
|
||||
|
||||
for mp in media_paths:
|
||||
transcript_path = _transcript_json_path(mp, output_dir)
|
||||
reused_prefix = ""
|
||||
if _load_cached_transcript(transcript_path) is not None:
|
||||
reused_prefix = f"# Cache\n\nReaproveitando transcrição existente: `{transcript_path}`\n\n"
|
||||
try:
|
||||
contents = asyncio.run(handle_build_voice_timeline({
|
||||
"media_path": mp, "model": model, "language": language,
|
||||
"hf_token": token, "num_speakers": num_speakers,
|
||||
"output_dir": args.get("output_dir"),
|
||||
"output_dir": output_dir,
|
||||
}))
|
||||
except Exception as exc:
|
||||
_emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"})
|
||||
return 1
|
||||
messages.append("\n".join(getattr(c, "text", str(c)) for c in contents))
|
||||
messages.append(reused_prefix + "\n".join(getattr(c, "text", str(c)) for c in contents))
|
||||
|
||||
_emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)})
|
||||
return 0
|
||||
@@ -938,9 +1057,23 @@ def cmd_set_diarization(args: dict) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_acoustics_capability(args: dict) -> int:
|
||||
"""Whether librosa (pitch/energy extraction) is installed in this venv.
|
||||
|
||||
Surfaces `features_capability()` — previously computed but never
|
||||
exposed to the app, so `layers.acoustics: false` in a voice timeline
|
||||
had no explanation the user could act on.
|
||||
"""
|
||||
from fcpxml.voice_features import features_capability
|
||||
ok, msg = features_capability()
|
||||
_emit({"ok": True, "available": ok, "message": msg})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_voice_analysis(args: dict) -> int:
|
||||
"""Read the persisted voice-analysis settings (energy/emphasis/emotion)."""
|
||||
_emit({"ok": True, **load_voice_analysis_config()})
|
||||
config = load_voice_analysis_config()
|
||||
_emit({"ok": True, **config, "emphasis_threshold": config["emphasis_floor"]})
|
||||
return 0
|
||||
|
||||
|
||||
@@ -950,9 +1083,13 @@ def cmd_set_voice_analysis(args: dict) -> int:
|
||||
config = save_voice_analysis_config(
|
||||
energy_threshold=args.get("energy_threshold"),
|
||||
emphasis_weights=weights if isinstance(weights, dict) else None,
|
||||
emphasis_threshold=args.get("emphasis_threshold"),
|
||||
emphasis_floor=args.get("emphasis_threshold"),
|
||||
emotion_enabled=args.get("emotion_enabled"),
|
||||
emotion_sensitivity=args.get("emotion_sensitivity"),
|
||||
zoom_scale=args.get("zoom_scale"),
|
||||
zoom_mode=args.get("zoom_mode"),
|
||||
zoom_ease_in=args.get("zoom_ease_in"),
|
||||
zoom_ease_out=args.get("zoom_ease_out"),
|
||||
)
|
||||
_emit({"ok": True, **config})
|
||||
return 0
|
||||
@@ -977,6 +1114,24 @@ def cmd_set_dynamic_subtitle_config(args: dict) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_plain_subtitle_config(args: dict) -> int:
|
||||
"""Read the persisted simple subtitle style."""
|
||||
_emit({"ok": True, **load_plain_subtitle_config()})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_set_plain_subtitle_config(args: dict) -> int:
|
||||
"""Persist simple subtitle style fields. Only the given fields change."""
|
||||
config = save_plain_subtitle_config(**{
|
||||
k: args.get(k) for k in (
|
||||
"font", "font_size", "font_color", "max_words",
|
||||
"position_y", "uppercase", "keep_punctuation", "text_scale",
|
||||
)
|
||||
})
|
||||
_emit({"ok": True, **config})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_silence_config(args: dict) -> int:
|
||||
"""Read the persisted silence thresholds (noise floor, duration, padding)."""
|
||||
_emit({"ok": True, **load_silence_config()})
|
||||
@@ -1026,6 +1181,14 @@ def cmd_apply_voice_actions(args: dict) -> int:
|
||||
return 1
|
||||
actions = loaded.get("actions") if isinstance(loaded, dict) else loaded
|
||||
|
||||
# The documented output format is {"source": ..., "actions": [...]} —
|
||||
# callers passing that whole object inline (e.g. the wizard pasting the
|
||||
# skill's JSON verbatim) need the same unwrap the actions_path branch
|
||||
# above already does, or a well-formed payload gets rejected as
|
||||
# "malformed" for having one extra layer of nesting.
|
||||
if isinstance(actions, dict):
|
||||
actions = actions.get("actions")
|
||||
|
||||
if not isinstance(actions, list) or not actions:
|
||||
_emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."})
|
||||
return 1
|
||||
@@ -1054,6 +1217,86 @@ def cmd_apply_voice_actions(args: dict) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_build_phrase_review(args: dict) -> int:
|
||||
"""Build the reviewable script (phrases + the AI's decisions) for the wizard.
|
||||
|
||||
`voice_timeline` points at the _voice_timeline.json; `actions` carries the
|
||||
decision list the model returned (inline, in any of the shapes the skill
|
||||
emits). The review is always rebuilt from the current analysis, then the
|
||||
decisions saved on a previous visit are laid back over it — reopening the
|
||||
step must show the edits the user left there without freezing the acoustics
|
||||
as they were when they left.
|
||||
"""
|
||||
from fcpxml.phrase_review import (
|
||||
build_phrase_review,
|
||||
load_phrase_review,
|
||||
merge_saved_decisions,
|
||||
)
|
||||
|
||||
timeline_path = str(args.get("voice_timeline", ""))
|
||||
if not timeline_path or not Path(timeline_path).exists():
|
||||
_emit({"ok": False, "error": "Análise de voz (voice_timeline.json) não encontrada."})
|
||||
return 1
|
||||
|
||||
try:
|
||||
with open(timeline_path, encoding="utf-8") as fh:
|
||||
timeline = json.load(fh)
|
||||
except (OSError, ValueError) as exc:
|
||||
_emit({"ok": False, "error": f"Erro ao ler a análise de voz: {exc}"})
|
||||
return 1
|
||||
|
||||
extra = [d for d in (args.get("output_dir"), args.get("media_dir")) if d]
|
||||
review = build_phrase_review(
|
||||
timeline,
|
||||
args.get("actions"),
|
||||
voice_timeline_path=timeline_path,
|
||||
extra_dirs=extra,
|
||||
)
|
||||
|
||||
saved = None if args.get("fresh") else load_phrase_review(timeline_path)
|
||||
review = merge_saved_decisions(review, saved)
|
||||
_emit({"ok": True, "reused": saved is not None, **review})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_save_phrase_review(args: dict) -> int:
|
||||
"""Persist the edited review and the actions derived from it."""
|
||||
from fcpxml.phrase_review import save_phrase_review
|
||||
|
||||
timeline_path = str(args.get("voice_timeline", ""))
|
||||
if not timeline_path:
|
||||
_emit({"ok": False, "error": "Caminho da análise de voz não informado."})
|
||||
return 1
|
||||
|
||||
phrases = args.get("phrases")
|
||||
if not isinstance(phrases, list):
|
||||
_emit({"ok": False, "error": "Nenhuma frase para salvar."})
|
||||
return 1
|
||||
|
||||
review = {
|
||||
"version": args.get("version", "1.0"),
|
||||
"source": args.get("source", ""),
|
||||
"duration": args.get("duration", 0.0),
|
||||
"speakers": args.get("speakers", []),
|
||||
"phrases": phrases,
|
||||
"zooms": args.get("zooms", []),
|
||||
}
|
||||
try:
|
||||
review_path, actions_path = save_phrase_review(timeline_path, review)
|
||||
except OSError as exc:
|
||||
_emit({"ok": False, "error": f"Erro ao salvar a revisão: {exc}"})
|
||||
return 1
|
||||
|
||||
_emit({
|
||||
"ok": True,
|
||||
"review_path": str(review_path),
|
||||
"actions_path": str(actions_path),
|
||||
"emphasis_count": sum(1 for p in phrases if int(p.get("emphasis", 0) or 0) >= 1),
|
||||
"removed_count": sum(1 for p in phrases if not p.get("active", True)),
|
||||
})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_project_config(args: dict) -> int:
|
||||
"""Read the last project folder/file the app was working on."""
|
||||
_emit({"ok": True, **load_project_config()})
|
||||
@@ -1115,17 +1358,23 @@ def main() -> int:
|
||||
"remove_filler_words": cmd_remove_filler_words,
|
||||
"transcript_markers": cmd_transcript_markers,
|
||||
"generate_dynamic_subtitles": cmd_generate_dynamic_subtitles,
|
||||
"generate_plain_subtitles": cmd_generate_plain_subtitles,
|
||||
"add_zoom": cmd_add_zoom,
|
||||
"zoom_clips": cmd_zoom_clips,
|
||||
"zoom_segments": cmd_zoom_segments,
|
||||
"rename_speakers": cmd_rename_speakers,
|
||||
"set_diarization": cmd_set_diarization,
|
||||
"acoustics_capability": cmd_acoustics_capability,
|
||||
"voice_analysis": cmd_voice_analysis,
|
||||
"set_voice_analysis": cmd_set_voice_analysis,
|
||||
"analyze_voice": cmd_analyze_voice,
|
||||
"dynamic_subtitle_config": cmd_dynamic_subtitle_config,
|
||||
"set_dynamic_subtitle_config": cmd_set_dynamic_subtitle_config,
|
||||
"plain_subtitle_config": cmd_plain_subtitle_config,
|
||||
"set_plain_subtitle_config": cmd_set_plain_subtitle_config,
|
||||
"apply_voice_actions": cmd_apply_voice_actions,
|
||||
"build_phrase_review": cmd_build_phrase_review,
|
||||
"save_phrase_review": cmd_save_phrase_review,
|
||||
"project_config": cmd_project_config,
|
||||
"set_project_config": cmd_set_project_config,
|
||||
"silence_config": cmd_silence_config,
|
||||
|
||||
@@ -1183,6 +1183,51 @@ o outro; percentil entrega um punhado útil nos dois casos.
|
||||
|
||||
---
|
||||
|
||||
## 21 — 2026-08-19 — Teste travado no default antigo de `zoom scale`
|
||||
|
||||
- **Sintoma:** `tests/test_voice_actions.py::test_default_scale_when_absent`
|
||||
quebrando com `KeyError: 'scale'`, sem relação com a alteração em curso.
|
||||
- **Causa raiz:** `parse_actions` deixou de carimbar `scale=1.3` quando o
|
||||
parâmetro vem ausente, justamente para que
|
||||
`server_tools/_shared.py` use o `zoom_scale` configurado pelo usuário. O
|
||||
teste continuou afirmando o default antigo, então passou a acusar como erro
|
||||
exatamente o comportamento desejado.
|
||||
- **Solução adotada:** teste reescrito para o contrato novo — um `scale`
|
||||
omitido tem que chegar ausente ao aplicador (`test_absent_scale_is_left_absent`).
|
||||
- **Aprendizado:** quando um default sai do parser e vira configuração, o teste
|
||||
que afirmava o valor antigo passa a defender o bug. Ao remover um default,
|
||||
procure o teste que o fixava no mesmo commit — senão ele fica dizendo o
|
||||
contrário do código, e a próxima pessoa perde tempo achando que quebrou algo.
|
||||
- **Estado:** `resolvido`
|
||||
|
||||
---
|
||||
|
||||
## 22 — 2026-08-19 — `VideoPlayer` (AVKit) derruba o app compilado por `swiftc`
|
||||
|
||||
- **Sintoma:** "G-ART encerrou inesperadamente" (SIGABRT) toda vez que o
|
||||
assistente entrava na etapa 5. Nada aparecia na tela antes do crash.
|
||||
- **Causa raiz:** o app é montado invocando `swiftc` direto
|
||||
(`MacApp/build_app.sh`), não pelo Xcode. Nesse modo o runtime não consegue
|
||||
resolver a superclasse Objective-C de `VideoPlayer`:
|
||||
`failed to demangle superclass of VideoPlayerView from mangled name
|
||||
'So12AVPlayerViewC'` → `getSuperclassMetadata` chama `fatalError`. É erro de
|
||||
runtime, então a compilação passa limpa e o problema só aparece ao abrir a
|
||||
view.
|
||||
- **Solução adotada:** trocar `VideoPlayer` por um `AVPlayerLayer` dentro de um
|
||||
`NSViewRepresentable` (`PlayerSurface`/`PlayerLayerView` em
|
||||
`PhraseReviewView.swift`). Só depende de AVFoundation, que linka normalmente.
|
||||
Os controles de transporte já viviam na barra da timeline, então não se perde
|
||||
nada com a chrome do AVKit.
|
||||
- **Aprendizado:** compilar limpo não prova que um componente de framework
|
||||
existe em runtime neste build. Ao usar uma view SwiftUI que embrulha uma
|
||||
classe AppKit/ObjC (AVKit, WebKit, MapKit), abra a tela de fato antes de
|
||||
concluir. Um harness pequeno (`swiftc` com os mesmos fontes + um `@main` que
|
||||
monta só aquela view e sai) reproduz o crash em segundos, sem precisar
|
||||
navegar o app inteiro até lá.
|
||||
- **Estado:** `resolvido`
|
||||
|
||||
---
|
||||
|
||||
## Resumo rápido (índice)
|
||||
|
||||
| # | Data | Problema | Estado |
|
||||
@@ -1205,5 +1250,7 @@ o outro; percentil entrega um punhado útil nos dois casos.
|
||||
| 18 | 2026-08-19 | Legendas dinâmicas geradas com `bold="0" fontFace="Bold"` não renderizam no FCP — negrito deve ser `bold="1"` (atributo) e itálico `fontFace`+`italic="1"` | `resolvido` |
|
||||
| 19 | 2026-08-19 | `output_dir` usado só como cerca de validação e nunca como destino — toda chamada entre pastas falhava acusando o caminho que ela mesma gerou | `resolvido` |
|
||||
| 20 | 2026-08-19 | `apply_voice_actions` ausente da ponte e do encadeamento do app — dava para analisar e legendar, não para cortar | `resolvido` |
|
||||
| 21 | 2026-08-19 | Teste ainda afirmava o default `zoom scale=1.3` removido do parser (agora vem do `zoom_scale` do usuário) | `resolvido` |
|
||||
| 22 | 2026-08-19 | `VideoPlayer` (AVKit) aborta em runtime no app compilado por `swiftc` — etapa 5 fechava o app; trocado por `AVPlayerLayer` | `resolvido` |
|
||||
|
||||
> Mantenha o índice acima sempre sincronizado com as entradas mais recentes.
|
||||
|
||||
@@ -12,6 +12,7 @@ struct GArtApp: App {
|
||||
}
|
||||
|
||||
enum ActiveTab: Hashable {
|
||||
case wizard
|
||||
case project
|
||||
case captions
|
||||
case voiceAnalysis
|
||||
@@ -20,19 +21,23 @@ enum ActiveTab: Hashable {
|
||||
}
|
||||
|
||||
struct ContentView: View {
|
||||
@State private var activeTab: ActiveTab? = .project
|
||||
@State private var activeTab: ActiveTab? = .wizard
|
||||
|
||||
var body: some View {
|
||||
NavigationSplitView {
|
||||
List(selection: $activeTab) {
|
||||
Label("Assistente", systemImage: "wand.and.stars")
|
||||
.tag(ActiveTab.wizard)
|
||||
Section("Avançado") {
|
||||
Label("Projeto", systemImage: "film")
|
||||
.tag(ActiveTab.project)
|
||||
Label("Legendas Dinâmicas", systemImage: "captions.bubble")
|
||||
Label("Legendas", systemImage: "captions.bubble")
|
||||
.tag(ActiveTab.captions)
|
||||
Label("Análise de Voz", systemImage: "waveform")
|
||||
.tag(ActiveTab.voiceAnalysis)
|
||||
Label("Modelos", systemImage: "tray.and.arrow.down")
|
||||
.tag(ActiveTab.models)
|
||||
}
|
||||
Label("Sobre", systemImage: "info.circle")
|
||||
.tag(ActiveTab.about)
|
||||
}
|
||||
@@ -40,19 +45,22 @@ struct ContentView: View {
|
||||
.navigationSplitViewColumnWidth(min: 180, ideal: 200)
|
||||
} detail: {
|
||||
switch activeTab {
|
||||
case .wizard, nil:
|
||||
WizardView().id(UUID())
|
||||
.navigationTitle("Assistente")
|
||||
case .project:
|
||||
ProjectView().id(UUID())
|
||||
.navigationTitle("Projeto")
|
||||
case .captions:
|
||||
CaptionsView().id(UUID())
|
||||
.navigationTitle("Legendas Dinâmicas")
|
||||
.navigationTitle("Legendas")
|
||||
case .voiceAnalysis:
|
||||
VoiceAnalysisView().id(UUID())
|
||||
.navigationTitle("Análise de Voz")
|
||||
case .models:
|
||||
ModelDownloadView().id(UUID())
|
||||
.navigationTitle("Modelos")
|
||||
case .about, nil:
|
||||
case .about:
|
||||
AboutView()
|
||||
.navigationTitle("Sobre")
|
||||
}
|
||||
|
||||
@@ -19,6 +19,7 @@ import UniformTypeIdentifiers
|
||||
/// assunto.
|
||||
struct CaptionsView: View {
|
||||
@State private var config = CaptionStyleConfig.defaults
|
||||
@State private var plainConfig = PlainSubtitleConfig.defaults
|
||||
@State private var isLoading = true
|
||||
@State private var errorMessage: String?
|
||||
|
||||
@@ -53,6 +54,13 @@ struct CaptionsView: View {
|
||||
)
|
||||
}
|
||||
|
||||
private func plainBound<T>(_ keyPath: WritableKeyPath<PlainSubtitleConfig, T>) -> Binding<T> {
|
||||
Binding(
|
||||
get: { plainConfig[keyPath: keyPath] },
|
||||
set: { plainConfig[keyPath: keyPath] = $0; savePlain() }
|
||||
)
|
||||
}
|
||||
|
||||
private func colorBound(_ keyPath: WritableKeyPath<CaptionStyleConfig, String>) -> Binding<Color> {
|
||||
Binding(
|
||||
get: { Color(rgbaString: config[keyPath: keyPath]) },
|
||||
@@ -60,6 +68,13 @@ struct CaptionsView: View {
|
||||
)
|
||||
}
|
||||
|
||||
private func plainColorBound(_ keyPath: WritableKeyPath<PlainSubtitleConfig, String>) -> Binding<Color> {
|
||||
Binding(
|
||||
get: { Color(rgbaString: plainConfig[keyPath: keyPath]) },
|
||||
set: { plainConfig[keyPath: keyPath] = $0.fcpxmlColorString; savePlain() }
|
||||
)
|
||||
}
|
||||
|
||||
var body: some View {
|
||||
HSplitView {
|
||||
previewColumn
|
||||
@@ -152,6 +167,7 @@ struct CaptionsView: View {
|
||||
positionSection
|
||||
bodySection
|
||||
emphasisSection
|
||||
plainSubtitleSection
|
||||
calibrationSection
|
||||
}
|
||||
if let errorMessage {
|
||||
@@ -225,6 +241,35 @@ struct CaptionsView: View {
|
||||
}
|
||||
}
|
||||
|
||||
private var plainSubtitleSection: some View {
|
||||
Section("Legenda comum") {
|
||||
Picker("Fonte", selection: plainBound(\.font)) {
|
||||
ForEach(fontChoices, id: \.self) { Text($0).tag($0) }
|
||||
}
|
||||
slider(
|
||||
"Tamanho",
|
||||
value: plainBound(\.fontSize), in: 28...300, step: 1,
|
||||
readout: "\(Int(plainConfig.fontSize))pt",
|
||||
help: "Tamanho da legenda comum editável no Final Cut."
|
||||
)
|
||||
slider(
|
||||
"Máximo de palavras",
|
||||
value: plainBound(\.maxWords), in: 1...14, step: 1,
|
||||
readout: "\(Int(plainConfig.maxWords))",
|
||||
help: "Quantidade máxima de palavras por bloco de legenda."
|
||||
)
|
||||
slider(
|
||||
"Altura",
|
||||
value: plainBound(\.positionY), in: -1200...300, step: 1,
|
||||
readout: "\(Int(plainConfig.positionY))",
|
||||
help: "Posição vertical da legenda comum no quadro; valores mais negativos descem."
|
||||
)
|
||||
ColorPicker("Cor", selection: plainColorBound(\.fontColor), supportsOpacity: true)
|
||||
Toggle("Usar letra maiúscula", isOn: plainBound(\.uppercase))
|
||||
Toggle("Manter vírgula e ponto", isOn: plainBound(\.keepPunctuation))
|
||||
}
|
||||
}
|
||||
|
||||
private var calibrationSection: some View {
|
||||
Section {
|
||||
slider(
|
||||
@@ -305,18 +350,33 @@ struct CaptionsView: View {
|
||||
} else if let error {
|
||||
errorMessage = error
|
||||
}
|
||||
PythonBridge.call(command: "plain_subtitle_config") { plainResult, plainError in
|
||||
DispatchQueue.main.async {
|
||||
if let plainResult {
|
||||
plainConfig = PlainSubtitleConfig(from: plainResult)
|
||||
} else if let plainError {
|
||||
errorMessage = plainError
|
||||
}
|
||||
isLoading = false
|
||||
continuation.resume()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func save() {
|
||||
PythonBridge.call(command: "set_dynamic_subtitle_config", arguments: config.arguments()) { _, error in
|
||||
DispatchQueue.main.async { errorMessage = error }
|
||||
}
|
||||
}
|
||||
|
||||
private func savePlain() {
|
||||
PythonBridge.call(command: "set_plain_subtitle_config", arguments: plainConfig.arguments()) { _, error in
|
||||
DispatchQueue.main.async { errorMessage = error }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// O estilo das legendas dinâmicas, no formato que a tela edita e o bridge
|
||||
@@ -402,6 +462,69 @@ struct CaptionStyleConfig {
|
||||
}
|
||||
}
|
||||
|
||||
struct PlainSubtitleConfig {
|
||||
var font: String
|
||||
var fontSize: Double
|
||||
var fontColor: String
|
||||
var maxWords: Double
|
||||
var positionY: Double
|
||||
var uppercase: Bool
|
||||
var keepPunctuation: Bool
|
||||
var textScale: Double
|
||||
|
||||
static let defaults = PlainSubtitleConfig(
|
||||
font: "Helvetica Neue",
|
||||
fontSize: 82,
|
||||
fontColor: "1 1 1 1",
|
||||
maxWords: 7,
|
||||
positionY: -820,
|
||||
uppercase: false,
|
||||
keepPunctuation: true,
|
||||
textScale: 2.0
|
||||
)
|
||||
|
||||
init(from json: [String: Any]) {
|
||||
let d = PlainSubtitleConfig.defaults
|
||||
self.init(
|
||||
font: json["font"] as? String ?? d.font,
|
||||
fontSize: (json["font_size"] as? NSNumber)?.doubleValue ?? d.fontSize,
|
||||
fontColor: json["font_color"] as? String ?? d.fontColor,
|
||||
maxWords: (json["max_words"] as? NSNumber)?.doubleValue ?? d.maxWords,
|
||||
positionY: (json["position_y"] as? NSNumber)?.doubleValue ?? d.positionY,
|
||||
uppercase: json["uppercase"] as? Bool ?? d.uppercase,
|
||||
keepPunctuation: json["keep_punctuation"] as? Bool ?? d.keepPunctuation,
|
||||
textScale: (json["text_scale"] as? NSNumber)?.doubleValue ?? d.textScale
|
||||
)
|
||||
}
|
||||
|
||||
init(
|
||||
font: String, fontSize: Double, fontColor: String, maxWords: Double,
|
||||
positionY: Double, uppercase: Bool, keepPunctuation: Bool, textScale: Double
|
||||
) {
|
||||
self.font = font
|
||||
self.fontSize = fontSize
|
||||
self.fontColor = fontColor
|
||||
self.maxWords = maxWords
|
||||
self.positionY = positionY
|
||||
self.uppercase = uppercase
|
||||
self.keepPunctuation = keepPunctuation
|
||||
self.textScale = textScale
|
||||
}
|
||||
|
||||
func arguments() -> [String: Any] {
|
||||
[
|
||||
"font": font,
|
||||
"font_size": Int(fontSize),
|
||||
"font_color": fontColor,
|
||||
"max_words": Int(maxWords),
|
||||
"position_y": positionY,
|
||||
"uppercase": uppercase,
|
||||
"keep_punctuation": keepPunctuation,
|
||||
"text_scale": textScale,
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
extension Color {
|
||||
/// Parses an FCPXML "R G B A" space-separated 0-1 string into a Color.
|
||||
init(rgbaString: String) {
|
||||
|
||||
@@ -15,6 +15,11 @@ struct ModelDownloadView: View {
|
||||
@State private var hfTokenText: String = ""
|
||||
@State private var numSpeakersText: String = ""
|
||||
@State private var language: String = "auto"
|
||||
@State private var acousticsAvailable: Bool?
|
||||
@State private var acousticsMessage: String = ""
|
||||
@State private var isInstallingAcoustics = false
|
||||
@State private var acousticsInstallLog: String = ""
|
||||
@State private var acousticsInstallError: String?
|
||||
|
||||
private let languages: [(String, String)] = [
|
||||
("auto", "Detectar automaticamente"),
|
||||
@@ -33,6 +38,7 @@ struct ModelDownloadView: View {
|
||||
var body: some View {
|
||||
Form {
|
||||
storageSection
|
||||
acousticsSection
|
||||
diarizationSection
|
||||
if let errorMessage {
|
||||
Section {
|
||||
@@ -65,7 +71,7 @@ struct ModelDownloadView: View {
|
||||
}
|
||||
}
|
||||
.formStyle(.grouped)
|
||||
.task { await refresh() }
|
||||
.task { await refresh(); checkAcoustics() }
|
||||
}
|
||||
|
||||
// MARK: - Transcription language
|
||||
@@ -95,6 +101,98 @@ struct ModelDownloadView: View {
|
||||
PythonBridge.call(command: "set_language", arguments: ["language": code]) { _, _ in }
|
||||
}
|
||||
|
||||
// MARK: - Acoustic analysis (librosa)
|
||||
|
||||
/// A ênfase de voz (pitch/energia) precisa do `librosa`, que é uma
|
||||
/// dependência opcional — sem ela `layers.acoustics` vem `false` na
|
||||
/// análise e a decisão de zoom fica sem base real. Antes disso só dava
|
||||
/// pra descobrir lendo o JSON exportado; agora o app já diz e resolve.
|
||||
private var acousticsSection: some View {
|
||||
Section {
|
||||
VStack(alignment: .leading, spacing: 10) {
|
||||
if let acousticsAvailable {
|
||||
Label(
|
||||
acousticsMessage.isEmpty
|
||||
? (acousticsAvailable ? "Disponível" : "Indisponível")
|
||||
: acousticsMessage,
|
||||
systemImage: acousticsAvailable ? "checkmark.circle.fill" : "exclamationmark.triangle.fill"
|
||||
)
|
||||
.font(.caption)
|
||||
.foregroundStyle(acousticsAvailable ? Color.green : Color.orange)
|
||||
} else {
|
||||
Label("Verificando…", systemImage: "hourglass")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
|
||||
if acousticsAvailable == false {
|
||||
Button {
|
||||
installAcoustics()
|
||||
} label: {
|
||||
if isInstallingAcoustics {
|
||||
HStack { ProgressView().controlSize(.small); Text("Instalando…") }
|
||||
} else {
|
||||
Label("Instalar (uv sync --all-extras)", systemImage: "arrow.down.circle")
|
||||
}
|
||||
}
|
||||
.disabled(isInstallingAcoustics)
|
||||
|
||||
if !acousticsInstallLog.isEmpty {
|
||||
ScrollView {
|
||||
Text(acousticsInstallLog)
|
||||
.font(.system(.caption2, design: .monospaced))
|
||||
.foregroundStyle(.secondary)
|
||||
.frame(maxWidth: .infinity, alignment: .leading)
|
||||
}
|
||||
.frame(height: 90)
|
||||
.background(RoundedRectangle(cornerRadius: 6).fill(Color.secondary.opacity(0.06)))
|
||||
}
|
||||
if let acousticsInstallError {
|
||||
Label(acousticsInstallError, systemImage: "xmark.circle.fill")
|
||||
.font(.caption).foregroundStyle(.red)
|
||||
}
|
||||
}
|
||||
}
|
||||
} header: {
|
||||
Text("Análise Acústica (zoom por voz)")
|
||||
} footer: {
|
||||
Text("Mede a energia e o tom de voz de verdade, para os candidatos a zoom da edição por voz. Sem isso, a análise ainda transcreve e decide cortes pelo texto — só o zoom fica sem base acústica.")
|
||||
.font(.caption)
|
||||
.foregroundStyle(.secondary)
|
||||
}
|
||||
}
|
||||
|
||||
private func checkAcoustics() {
|
||||
PythonBridge.call(command: "acoustics_capability") { result, err in
|
||||
DispatchQueue.main.async {
|
||||
guard let result, result["ok"] as? Bool == true else { return }
|
||||
acousticsAvailable = result["available"] as? Bool
|
||||
acousticsMessage = result["message"] as? String ?? ""
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func installAcoustics() {
|
||||
isInstallingAcoustics = true
|
||||
acousticsInstallLog = ""
|
||||
acousticsInstallError = nil
|
||||
// --all-extras, não só "intelligence": `uv sync` substitui o
|
||||
// ambiente pelos extras pedidos em vez de somar, então um sync
|
||||
// parcial aqui derrubaria dev/transcribe/diarização já instalados.
|
||||
PythonBridge.runUV(arguments: ["sync", "--all-extras"]) { line in
|
||||
DispatchQueue.main.async {
|
||||
acousticsInstallLog += (acousticsInstallLog.isEmpty ? "" : "\n") + line
|
||||
}
|
||||
} completion: { code, err in
|
||||
DispatchQueue.main.async {
|
||||
isInstallingAcoustics = false
|
||||
if code != 0 {
|
||||
acousticsInstallError = err ?? "Falha ao instalar."
|
||||
}
|
||||
checkAcoustics()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Diarization
|
||||
|
||||
private var diarizationSection: some View {
|
||||
|
||||
@@ -116,6 +116,145 @@ struct ZoomClip: Identifiable {
|
||||
}
|
||||
}
|
||||
|
||||
/// One word inside a phrase, with the acoustics that justify an emphasis.
|
||||
struct ReviewWord: Identifiable {
|
||||
let id: Int
|
||||
let text: String
|
||||
let start: Double
|
||||
let end: Double
|
||||
let energy: Double
|
||||
let emphasis: Double
|
||||
|
||||
init(id: Int, json: [String: Any]) {
|
||||
self.id = id
|
||||
text = json["text"] as? String ?? ""
|
||||
start = json["start"] as? Double ?? 0
|
||||
end = json["end"] as? Double ?? 0
|
||||
energy = json["energy"] as? Double ?? 0
|
||||
emphasis = json["emphasis"] as? Double ?? 0
|
||||
}
|
||||
}
|
||||
|
||||
/// A phrase in the review step — one spoken line plus the decision made about
|
||||
/// it. Mirrors `fcpxml/phrase_review.py`; `emphasis` is 0–3 and everything
|
||||
/// mutable here is what the editor is allowed to change.
|
||||
struct ReviewPhrase: Identifiable {
|
||||
let id: Int
|
||||
let start: Double
|
||||
let end: Double
|
||||
var trimStart: Double
|
||||
var trimEnd: Double
|
||||
var text: String
|
||||
let speaker: String
|
||||
var active: Bool
|
||||
var emphasis: Int
|
||||
var track: String
|
||||
let peakEmphasis: Double
|
||||
let emotion: String
|
||||
let emotionConfidence: Double
|
||||
let takeBoundary: Bool
|
||||
let gapBefore: Double
|
||||
let reason: String
|
||||
let words: [ReviewWord]
|
||||
|
||||
static let trackScript = "roteiro"
|
||||
static let trackBackstage = "bastidor"
|
||||
|
||||
/// Delivery emotion as the analysis names it, in the user's language plus a
|
||||
/// glyph — the label alone is too easy to skim past in a dense list.
|
||||
static func emotionLabel(_ emotion: String) -> (String, String) {
|
||||
switch emotion {
|
||||
case "excited": return ("Empolgado", "flame")
|
||||
case "tense": return ("Tenso", "bolt")
|
||||
case "calm": return ("Calmo", "leaf")
|
||||
case "reflective": return ("Reflexivo", "moon")
|
||||
default: return ("Neutro", "circle")
|
||||
}
|
||||
}
|
||||
|
||||
init(json: [String: Any]) {
|
||||
id = json["index"] as? Int ?? 0
|
||||
start = json["start"] as? Double ?? 0
|
||||
end = json["end"] as? Double ?? 0
|
||||
trimStart = json["trim_start"] as? Double ?? (json["start"] as? Double ?? 0)
|
||||
trimEnd = json["trim_end"] as? Double ?? (json["end"] as? Double ?? 0)
|
||||
text = json["text"] as? String ?? ""
|
||||
speaker = json["speaker"] as? String ?? ""
|
||||
active = json["active"] as? Bool ?? true
|
||||
emphasis = json["emphasis"] as? Int ?? 0
|
||||
track = json["track"] as? String ?? ReviewPhrase.trackScript
|
||||
peakEmphasis = json["peak_emphasis"] as? Double ?? 0
|
||||
emotion = json["emotion"] as? String ?? "neutral"
|
||||
emotionConfidence = json["emotion_confidence"] as? Double ?? 0
|
||||
takeBoundary = json["take_boundary"] as? Bool ?? false
|
||||
gapBefore = json["gap_before"] as? Double ?? 0
|
||||
reason = json["reason"] as? String ?? ""
|
||||
words = (json["words"] as? [[String: Any]] ?? [])
|
||||
.enumerated().map { ReviewWord(id: $0.offset, json: $0.element) }
|
||||
}
|
||||
|
||||
var asJSON: [String: Any] {
|
||||
[
|
||||
"index": id,
|
||||
"start": start,
|
||||
"end": end,
|
||||
"trim_start": trimStart,
|
||||
"trim_end": trimEnd,
|
||||
"text": text,
|
||||
"speaker": speaker,
|
||||
"active": active,
|
||||
"emphasis": emphasis,
|
||||
"track": track,
|
||||
"reason": reason,
|
||||
]
|
||||
}
|
||||
|
||||
var isBackstage: Bool { track == ReviewPhrase.trackBackstage }
|
||||
var isTrimmed: Bool { trimStart > start + 0.001 || trimEnd < end - 0.001 }
|
||||
var timecode: String {
|
||||
String(format: "%02d:%02d", Int(start) / 60, Int(start) % 60)
|
||||
}
|
||||
|
||||
/// The word boundaries a trim handle is allowed to land on.
|
||||
func snap(_ time: Double, edge: TrimEdge) -> Double {
|
||||
let boundaries = words.map { edge == .start ? $0.start : $0.end }.filter { $0 > 0 }
|
||||
guard let nearest = boundaries.min(by: { abs($0 - time) < abs($1 - time) }) else {
|
||||
return time
|
||||
}
|
||||
return nearest
|
||||
}
|
||||
}
|
||||
|
||||
enum TrimEdge { case start, end }
|
||||
|
||||
/// A punch-in the editor placed by hand over an arbitrary range, next to the
|
||||
/// whole-phrase zoom that an emphasis level produces. It stores only *when* —
|
||||
/// the scale and the ramp come from the Voice Analysis settings at render time.
|
||||
struct ManualZoom: Identifiable {
|
||||
let id = UUID()
|
||||
var start: Double
|
||||
var end: Double
|
||||
|
||||
/// Below this a punch-in has no room to ramp in and back out; the writer
|
||||
/// rejects the window, so offering it would place nothing.
|
||||
static let minimumDuration: Double = 0.4
|
||||
|
||||
init(start: Double, end: Double) {
|
||||
self.start = start
|
||||
self.end = end
|
||||
}
|
||||
|
||||
init?(json: [String: Any]) {
|
||||
guard let start = json["start"] as? Double, let end = json["end"] as? Double,
|
||||
end - start >= ManualZoom.minimumDuration
|
||||
else { return nil }
|
||||
self.start = start
|
||||
self.end = end
|
||||
}
|
||||
|
||||
var asJSON: [String: Any] { ["start": start, "end": end] }
|
||||
}
|
||||
|
||||
struct ZoomSegment: Identifiable {
|
||||
let id: Int
|
||||
let start: Double
|
||||
|
||||
@@ -0,0 +1,429 @@
|
||||
import AVFoundation
|
||||
import Combine
|
||||
import Foundation
|
||||
|
||||
/// State behind the wizard's emphasis-review step.
|
||||
///
|
||||
/// Holds the phrases, the selection, and the player — together, because they
|
||||
/// are one thing to the user: clicking a phrase moves the playhead, playing
|
||||
/// moves the selection, and skipping a removed line only works if whoever owns
|
||||
/// playback also knows which lines are removed.
|
||||
///
|
||||
/// The preview deliberately plays the *original* media and jumps over whatever
|
||||
/// the edit removes, instead of rendering a cut first. Rendering to check a
|
||||
/// toggle would put minutes between a decision and its result; jumping gives
|
||||
/// the same reading instantly, and the real cut is generated later from the
|
||||
/// exact same phrase list.
|
||||
@MainActor
|
||||
final class PhraseReviewModel: ObservableObject {
|
||||
@Published var phrases: [ReviewPhrase] = []
|
||||
@Published var selection: Int?
|
||||
@Published var isLoading = false
|
||||
@Published var errorMessage: String?
|
||||
@Published var currentTime: Double = 0
|
||||
@Published var isPlaying = false
|
||||
@Published var pixelsPerSecond: Double = 40
|
||||
@Published var skipRemoved = true
|
||||
@Published var zooms: [ManualZoom] = []
|
||||
/// In/out the editor dragged on the timeline, in source seconds.
|
||||
@Published var rangeStart: Double?
|
||||
@Published var rangeEnd: Double?
|
||||
/// Aspect ratio of the footage as recorded.
|
||||
@Published var videoAspect: Double = 16.0 / 9.0
|
||||
/// Aspect ratio the project delivers in, read from the .fcpxml. It is
|
||||
/// routinely *not* the footage's: these takes are shot horizontal and
|
||||
/// delivered vertical, so previewing the raw frame would show a crop the
|
||||
/// audience never sees — and the emphasis decisions are about what lands on
|
||||
/// screen. Nil until the project is known.
|
||||
@Published var projectAspect: Double?
|
||||
/// Whether the preview crops to the delivery frame. On by default whenever
|
||||
/// the two aspects disagree.
|
||||
@Published var matchProjectFraming = true
|
||||
|
||||
/// What the preview should actually draw.
|
||||
var previewAspect: Double {
|
||||
guard matchProjectFraming, let projectAspect else { return videoAspect }
|
||||
return projectAspect
|
||||
}
|
||||
|
||||
/// True when the delivery frame differs enough from the footage that the
|
||||
/// preview is showing a crop rather than the whole take.
|
||||
var isCropping: Bool {
|
||||
guard matchProjectFraming, let projectAspect else { return false }
|
||||
return abs(projectAspect - videoAspect) > 0.01
|
||||
}
|
||||
|
||||
private(set) var source = ""
|
||||
private(set) var sourcePath = ""
|
||||
private(set) var duration: Double = 0
|
||||
private(set) var speakers: [String] = []
|
||||
private(set) var emotionAvailable = false
|
||||
private(set) var player: AVPlayer?
|
||||
|
||||
private var voiceTimelinePath = ""
|
||||
private var timeObserver: Any?
|
||||
private var playbackLimit: Double?
|
||||
|
||||
let minPixelsPerSecond: Double = 8
|
||||
let maxPixelsPerSecond: Double = 400
|
||||
|
||||
deinit {
|
||||
if let timeObserver, let player {
|
||||
player.removeTimeObserver(timeObserver)
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Carregar
|
||||
|
||||
/// Builds the review from the voice timeline plus whatever the AI decided.
|
||||
/// A review saved on a previous visit wins — see `cmd_build_phrase_review`.
|
||||
/// Reads the delivery format from the project so the preview can frame the
|
||||
/// take the way it will actually be seen.
|
||||
func loadProjectFormat(projectPath: String) {
|
||||
PythonBridge.call(command: "inspect", arguments: ["path": projectPath]) { [weak self] result, _ in
|
||||
Task { @MainActor in
|
||||
guard let self,
|
||||
let timelines = result?["timelines"] as? [[String: Any]],
|
||||
let first = timelines.first,
|
||||
let width = first["width"] as? Int, let height = first["height"] as? Int,
|
||||
width > 0, height > 0
|
||||
else { return }
|
||||
self.projectAspect = Double(width) / Double(height)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func load(voiceTimelinePath: String, decisionsJSON: String,
|
||||
outputFolder: String? = nil, mediaFolder: String? = nil) {
|
||||
self.voiceTimelinePath = voiceTimelinePath
|
||||
isLoading = true
|
||||
errorMessage = nil
|
||||
|
||||
var arguments: [String: Any] = ["voice_timeline": voiceTimelinePath]
|
||||
if let outputFolder { arguments["output_dir"] = outputFolder }
|
||||
if let mediaFolder { arguments["media_dir"] = mediaFolder }
|
||||
if let data = decisionsJSON.data(using: .utf8),
|
||||
let parsed = try? JSONSerialization.jsonObject(with: data) {
|
||||
arguments["actions"] = parsed
|
||||
}
|
||||
|
||||
PythonBridge.call(command: "build_phrase_review", arguments: arguments) { [weak self] result, error in
|
||||
Task { @MainActor in
|
||||
guard let self else { return }
|
||||
self.isLoading = false
|
||||
if let error {
|
||||
self.errorMessage = error
|
||||
return
|
||||
}
|
||||
guard let result, result["ok"] as? Bool == true else {
|
||||
self.errorMessage = result?["error"] as? String ?? "Não foi possível montar a revisão."
|
||||
return
|
||||
}
|
||||
self.apply(result)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func apply(_ result: [String: Any]) {
|
||||
source = result["source"] as? String ?? ""
|
||||
// The timeline JSON stores only the media's file name; the bridge
|
||||
// resolves it to something openable (see phrase_review.resolve_source).
|
||||
sourcePath = result["source_path"] as? String ?? ""
|
||||
duration = result["duration"] as? Double ?? 0
|
||||
speakers = result["speakers"] as? [String] ?? []
|
||||
emotionAvailable = result["emotion_available"] as? Bool ?? false
|
||||
phrases = (result["phrases"] as? [[String: Any]] ?? []).map { ReviewPhrase(json: $0) }
|
||||
zooms = (result["zooms"] as? [[String: Any]] ?? []).compactMap { ManualZoom(json: $0) }
|
||||
selection = phrases.first?.id
|
||||
if let errors = result["errors"] as? [String], !errors.isEmpty {
|
||||
errorMessage = "A IA mandou \(errors.count) decisão(ões) que não deu para ler — o resto foi aplicado."
|
||||
}
|
||||
preparePlayer()
|
||||
}
|
||||
|
||||
/// Point the preview at a media file the user chose by hand — the way out
|
||||
/// when the footage moved somewhere the automatic lookup can't reach.
|
||||
func useMedia(at path: String) {
|
||||
sourcePath = path
|
||||
preparePlayer()
|
||||
}
|
||||
|
||||
private func preparePlayer() {
|
||||
guard !sourcePath.isEmpty, FileManager.default.fileExists(atPath: sourcePath) else {
|
||||
player = nil
|
||||
return
|
||||
}
|
||||
if let timeObserver, let player {
|
||||
player.removeTimeObserver(timeObserver)
|
||||
self.timeObserver = nil
|
||||
}
|
||||
let asset = AVURLAsset(url: URL(fileURLWithPath: sourcePath))
|
||||
let player = AVPlayer(playerItem: AVPlayerItem(asset: asset))
|
||||
self.player = player
|
||||
readAspect(from: asset)
|
||||
// 60 Hz: the same observer drives the playhead *and* decides when to
|
||||
// jump a removed stretch, so its period is the worst-case amount of cut
|
||||
// material that can be heard before the skip lands. At 20 Hz that was an
|
||||
// audible blip on every join.
|
||||
let interval = CMTime(seconds: 1.0 / 60.0, preferredTimescale: 600)
|
||||
timeObserver = player.addPeriodicTimeObserver(forInterval: interval, queue: .main) { [weak self] time in
|
||||
Task { @MainActor in
|
||||
self?.tick(time.seconds)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The displayed aspect ratio, honouring the rotation the camera recorded.
|
||||
/// A phone take is stored 1920×1080 with a 90° transform: reading
|
||||
/// `naturalSize` alone would call a vertical video horizontal.
|
||||
private func readAspect(from asset: AVURLAsset) {
|
||||
Task { [weak self] in
|
||||
guard let track = try? await asset.loadTracks(withMediaType: .video).first,
|
||||
let size = try? await track.load(.naturalSize),
|
||||
let transform = try? await track.load(.preferredTransform)
|
||||
else { return }
|
||||
let displayed = size.applying(transform)
|
||||
let width = abs(displayed.width), height = abs(displayed.height)
|
||||
guard width > 0, height > 0 else { return }
|
||||
await MainActor.run { self?.videoAspect = width / height }
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Reprodução
|
||||
|
||||
private func tick(_ time: Double) {
|
||||
currentTime = time
|
||||
guard isPlaying else { return }
|
||||
|
||||
// Playing a single phrase or a marked range stops at its out point
|
||||
// instead of running on into the rest of the take.
|
||||
if let limit = playbackLimit, time >= limit {
|
||||
pause()
|
||||
seek(to: limit)
|
||||
return
|
||||
}
|
||||
|
||||
if skipRemoved, let jump = nextKeptTime(after: time), jump > time {
|
||||
seek(to: jump)
|
||||
}
|
||||
if let phrase = phrase(at: time), selection != phrase.id {
|
||||
selection = phrase.id
|
||||
}
|
||||
}
|
||||
|
||||
/// Where playback should resume when `time` lands on removed material.
|
||||
/// Returns nil when the time is on material that survives.
|
||||
func nextKeptTime(after time: Double) -> Double? {
|
||||
for phrase in phrases where time >= phrase.start - 0.001 && time < phrase.end {
|
||||
if !phrase.active { return phrase.end }
|
||||
if time < phrase.trimStart { return phrase.trimStart }
|
||||
if time >= phrase.trimEnd { return phrase.end }
|
||||
return nil
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func togglePlay() {
|
||||
if isPlaying {
|
||||
pause()
|
||||
} else {
|
||||
playbackLimit = nil
|
||||
play()
|
||||
}
|
||||
}
|
||||
|
||||
private func play() {
|
||||
guard let player else { return }
|
||||
if skipRemoved, let jump = nextKeptTime(after: currentTime) { seek(to: jump) }
|
||||
player.play()
|
||||
isPlaying = true
|
||||
}
|
||||
|
||||
func pause() {
|
||||
player?.pause()
|
||||
isPlaying = false
|
||||
playbackLimit = nil
|
||||
}
|
||||
|
||||
/// Play exactly one span and stop — how a cut is judged: in context, at
|
||||
/// speed, without hunting for the out point by hand.
|
||||
func playRange(from start: Double, to end: Double) {
|
||||
guard end > start else { return }
|
||||
seek(to: start)
|
||||
playbackLimit = end
|
||||
player?.play()
|
||||
isPlaying = true
|
||||
}
|
||||
|
||||
func playSelectedPhrase() {
|
||||
guard let selection, let phrase = phrases.first(where: { $0.id == selection })
|
||||
else { return }
|
||||
playRange(from: phrase.active ? phrase.trimStart : phrase.start,
|
||||
to: phrase.active ? phrase.trimEnd : phrase.end)
|
||||
}
|
||||
|
||||
func seek(to time: Double) {
|
||||
currentTime = max(0, time)
|
||||
player?.seek(to: CMTime(seconds: max(0, time), preferredTimescale: 600),
|
||||
toleranceBefore: .zero, toleranceAfter: .zero)
|
||||
}
|
||||
|
||||
/// Move the playhead to a phrase and select it.
|
||||
func goTo(phraseID: Int) {
|
||||
guard let phrase = phrases.first(where: { $0.id == phraseID }) else { return }
|
||||
selection = phraseID
|
||||
seek(to: phrase.active ? phrase.trimStart : phrase.start)
|
||||
}
|
||||
|
||||
func phrase(at time: Double) -> ReviewPhrase? {
|
||||
phrases.first { time >= $0.start && time < $0.end }
|
||||
}
|
||||
|
||||
func selectNeighbour(_ delta: Int) {
|
||||
guard let selection, let index = phrases.firstIndex(where: { $0.id == selection }) else {
|
||||
if let first = phrases.first { goTo(phraseID: first.id) }
|
||||
return
|
||||
}
|
||||
let next = min(max(0, index + delta), phrases.count - 1)
|
||||
goTo(phraseID: phrases[next].id)
|
||||
}
|
||||
|
||||
// MARK: - Edições
|
||||
|
||||
private func update(_ id: Int, _ change: (inout ReviewPhrase) -> Void) {
|
||||
guard let index = phrases.firstIndex(where: { $0.id == id }) else { return }
|
||||
change(&phrases[index])
|
||||
}
|
||||
|
||||
func setEmphasis(_ level: Int, for id: Int) {
|
||||
update(id) { $0.emphasis = min(3, max(0, level)) }
|
||||
}
|
||||
|
||||
func toggleActive(_ id: Int) {
|
||||
update(id) { $0.active.toggle() }
|
||||
}
|
||||
|
||||
func setTrack(_ track: String, for id: Int) {
|
||||
update(id) { $0.track = track }
|
||||
}
|
||||
|
||||
func setText(_ text: String, for id: Int) {
|
||||
update(id) { $0.text = text }
|
||||
}
|
||||
|
||||
/// Trim a phrase's head or tail, landing on a word boundary.
|
||||
/// A trim that would swallow the whole line is refused — deactivating the
|
||||
/// phrase is the way to remove it, and doing it by accident with a drag
|
||||
/// would lose the emphasis decision along with the line.
|
||||
func trim(_ id: Int, edge: TrimEdge, to time: Double) {
|
||||
update(id) { phrase in
|
||||
let snapped = phrase.snap(time, edge: edge)
|
||||
switch edge {
|
||||
case .start:
|
||||
let value = min(max(phrase.start, snapped), phrase.trimEnd - 0.1)
|
||||
if value < phrase.trimEnd { phrase.trimStart = value }
|
||||
case .end:
|
||||
let value = max(min(phrase.end, snapped), phrase.trimStart + 0.1)
|
||||
if value > phrase.trimStart { phrase.trimEnd = value }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func resetTrim(_ id: Int) {
|
||||
update(id) { $0.trimStart = $0.start; $0.trimEnd = $0.end }
|
||||
}
|
||||
|
||||
/// Trim everything before/after a given word — the text-first way to cut,
|
||||
/// since the editor reads the line and points at where it should begin.
|
||||
func trimToWord(_ word: ReviewWord, edge: TrimEdge, in id: Int) {
|
||||
trim(id, edge: edge, to: edge == .start ? word.start : word.end)
|
||||
}
|
||||
|
||||
// MARK: - Trecho marcado e zooms
|
||||
|
||||
var hasRange: Bool {
|
||||
guard let rangeStart, let rangeEnd else { return false }
|
||||
return rangeEnd - rangeStart >= ManualZoom.minimumDuration
|
||||
}
|
||||
|
||||
var rangeSpan: (start: Double, end: Double)? {
|
||||
guard let rangeStart, let rangeEnd, rangeEnd > rangeStart else { return nil }
|
||||
return (rangeStart, rangeEnd)
|
||||
}
|
||||
|
||||
func setRange(from start: Double, to end: Double) {
|
||||
rangeStart = min(start, end)
|
||||
rangeEnd = max(start, end)
|
||||
}
|
||||
|
||||
func clearRange() {
|
||||
rangeStart = nil
|
||||
rangeEnd = nil
|
||||
}
|
||||
|
||||
/// Add a punch-in over the marked range. Scale and ramp are not stored:
|
||||
/// they come from the "Análise de Voz" settings when the edit is rendered,
|
||||
/// so changing the look there restyles every zoom at once.
|
||||
func addZoomForRange() {
|
||||
guard let span = rangeSpan, span.end - span.start >= ManualZoom.minimumDuration
|
||||
else { return }
|
||||
zooms.append(ManualZoom(start: span.start, end: span.end))
|
||||
zooms.sort { $0.start < $1.start }
|
||||
clearRange()
|
||||
}
|
||||
|
||||
func addZoomForPhrase(_ id: Int) {
|
||||
guard let phrase = phrases.first(where: { $0.id == id }) else { return }
|
||||
zooms.append(ManualZoom(start: phrase.trimStart, end: phrase.trimEnd))
|
||||
zooms.sort { $0.start < $1.start }
|
||||
}
|
||||
|
||||
func removeZoom(_ id: UUID) {
|
||||
zooms.removeAll { $0.id == id }
|
||||
}
|
||||
|
||||
func zoom(at time: Double) -> ManualZoom? {
|
||||
zooms.first { time >= $0.start && time <= $0.end }
|
||||
}
|
||||
|
||||
func setEmphasisForAll(_ level: Int) {
|
||||
for index in phrases.indices where phrases[index].active {
|
||||
phrases[index].emphasis = level
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Resumo e gravação
|
||||
|
||||
var emphasisCount: Int { phrases.filter { $0.active && $0.emphasis >= 1 }.count }
|
||||
var removedCount: Int { phrases.filter { !$0.active }.count }
|
||||
var keptDuration: Double {
|
||||
phrases.filter { $0.active }.reduce(0) { $0 + ($1.trimEnd - $1.trimStart) }
|
||||
}
|
||||
|
||||
/// Persists the edited review plus the actions derived from it. Called when
|
||||
/// the wizard advances — the render itself happens in the next step.
|
||||
func save(completion: @escaping (String?) -> Void) {
|
||||
guard !voiceTimelinePath.isEmpty, !phrases.isEmpty else {
|
||||
completion(nil)
|
||||
return
|
||||
}
|
||||
let arguments: [String: Any] = [
|
||||
"voice_timeline": voiceTimelinePath,
|
||||
"source": source,
|
||||
"duration": duration,
|
||||
"speakers": speakers,
|
||||
"phrases": phrases.map { $0.asJSON },
|
||||
"zooms": zooms.map { $0.asJSON },
|
||||
]
|
||||
PythonBridge.call(command: "save_phrase_review", arguments: arguments) { result, error in
|
||||
Task { @MainActor in
|
||||
if let error {
|
||||
completion(nil)
|
||||
_ = error
|
||||
return
|
||||
}
|
||||
completion(result?["review_path"] as? String)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,413 @@
|
||||
import AVFoundation
|
||||
import SwiftUI
|
||||
|
||||
/// The video surface, as a plain `AVPlayerLayer` in an `NSView`.
|
||||
///
|
||||
/// AVKit's `VideoPlayer` would be the obvious choice and is a trap here: this
|
||||
/// app is built by invoking `swiftc` directly (see `MacApp/build_app.sh`), and
|
||||
/// `_AVKit_SwiftUI` aborts at launch instantiating its generic metadata under
|
||||
/// that build. A player layer needs only AVFoundation, which links cleanly —
|
||||
/// and the transport controls live in the timeline's own toolbar anyway, so
|
||||
/// nothing is lost by dropping AVKit's chrome.
|
||||
private struct PlayerSurface: NSViewRepresentable {
|
||||
let player: AVPlayer
|
||||
/// When true the frame is filled and cropped instead of letterboxed — used
|
||||
/// to preview horizontal footage inside a vertical delivery frame.
|
||||
var fills: Bool
|
||||
|
||||
func makeNSView(context: Context) -> PlayerLayerView {
|
||||
let view = PlayerLayerView()
|
||||
view.player = player
|
||||
view.fills = fills
|
||||
return view
|
||||
}
|
||||
|
||||
func updateNSView(_ view: PlayerLayerView, context: Context) {
|
||||
if view.player !== player { view.player = player }
|
||||
view.fills = fills
|
||||
}
|
||||
}
|
||||
|
||||
final class PlayerLayerView: NSView {
|
||||
private let playerLayer = AVPlayerLayer()
|
||||
|
||||
var player: AVPlayer? {
|
||||
get { playerLayer.player }
|
||||
set { playerLayer.player = newValue }
|
||||
}
|
||||
|
||||
var fills: Bool = false {
|
||||
didSet { playerLayer.videoGravity = fills ? .resizeAspectFill : .resizeAspect }
|
||||
}
|
||||
|
||||
override init(frame frameRect: NSRect) {
|
||||
super.init(frame: frameRect)
|
||||
wantsLayer = true
|
||||
layer = CALayer()
|
||||
layer?.backgroundColor = NSColor.black.cgColor
|
||||
playerLayer.videoGravity = .resizeAspect
|
||||
layer?.addSublayer(playerLayer)
|
||||
}
|
||||
|
||||
required init?(coder: NSCoder) {
|
||||
super.init(coder: coder)
|
||||
wantsLayer = true
|
||||
layer = CALayer()
|
||||
playerLayer.videoGravity = .resizeAspect
|
||||
layer?.addSublayer(playerLayer)
|
||||
}
|
||||
|
||||
override func layout() {
|
||||
super.layout()
|
||||
playerLayer.frame = bounds
|
||||
}
|
||||
}
|
||||
|
||||
/// The wizard's emphasis-review step, laid out like an editing room: preview on
|
||||
/// top, timeline across the bottom, and the script as an inspector down the
|
||||
/// right side.
|
||||
///
|
||||
/// The arrangement is the point. Every decision here is about a *sentence*, so
|
||||
/// the same phrase has to be legible in all three places at once — a block on
|
||||
/// the timeline, a line of text in the inspector, and a moment in the preview.
|
||||
/// Selecting in any one of them selects in the other two.
|
||||
struct PhraseReviewView: View {
|
||||
@ObservedObject var model: PhraseReviewModel
|
||||
|
||||
var body: some View {
|
||||
VSplitView {
|
||||
HSplitView {
|
||||
previewPane
|
||||
.frame(minWidth: 320, idealWidth: 640)
|
||||
inspectorPane
|
||||
.frame(minWidth: 300, idealWidth: 360, maxWidth: 520)
|
||||
}
|
||||
.frame(minHeight: 240)
|
||||
|
||||
TimelineTracksView(model: model)
|
||||
.frame(minHeight: 190, idealHeight: 210)
|
||||
}
|
||||
.overlay { if model.isLoading { loadingOverlay } }
|
||||
.focusable()
|
||||
.onKeyPress(.space) { model.togglePlay(); return .handled }
|
||||
.onKeyPress(.return) { model.playSelectedPhrase(); return .handled }
|
||||
.onKeyPress(.leftArrow) { model.selectNeighbour(-1); return .handled }
|
||||
.onKeyPress(.rightArrow) { model.selectNeighbour(1); return .handled }
|
||||
.onKeyPress(characters: .decimalDigits) { press in
|
||||
guard let level = Int(press.characters), (0...3).contains(level),
|
||||
let selection = model.selection else { return .ignored }
|
||||
model.setEmphasis(level, for: selection)
|
||||
return .handled
|
||||
}
|
||||
}
|
||||
|
||||
private var loadingOverlay: some View {
|
||||
ZStack {
|
||||
Color(nsColor: .windowBackgroundColor).opacity(0.85)
|
||||
VStack(spacing: 10) {
|
||||
ProgressView()
|
||||
Text("Montando a revisão…").font(.callout).foregroundStyle(.secondary)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Preview
|
||||
|
||||
private var previewPane: some View {
|
||||
VStack(spacing: 0) {
|
||||
if let player = model.player {
|
||||
// The footage here is usually vertical. Sizing the surface to
|
||||
// the take's own aspect keeps a 9:16 frame as tall as the pane
|
||||
// allows instead of shrinking it to fit a horizontal box.
|
||||
// Framed to what the project delivers, not to what the camera
|
||||
// recorded: these takes are shot horizontal and cut vertical,
|
||||
// so the raw frame would show material the audience never sees.
|
||||
ZStack {
|
||||
Color.black
|
||||
PlayerSurface(player: player, fills: model.isCropping)
|
||||
.aspectRatio(model.previewAspect, contentMode: .fit)
|
||||
.clipped()
|
||||
}
|
||||
.overlay(alignment: .topTrailing) { framingBadge }
|
||||
} else {
|
||||
ZStack {
|
||||
Color.black.opacity(0.85)
|
||||
VStack(spacing: 10) {
|
||||
Image(systemName: "film.stack")
|
||||
.font(.system(size: 28)).foregroundStyle(.secondary)
|
||||
Text(model.source.isEmpty
|
||||
? "A análise de voz não registrou qual mídia foi usada."
|
||||
: "Não achei \(model.source) na pasta do projeto.")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
Text("A revisão funciona igual sem o preview — ele só ajuda a conferir o corte.")
|
||||
.font(.caption).foregroundStyle(.tertiary)
|
||||
Button("Localizar a mídia…") { pickMedia() }
|
||||
.buttonStyle(.bordered)
|
||||
}
|
||||
.multilineTextAlignment(.center)
|
||||
.padding(.horizontal, 24)
|
||||
}
|
||||
}
|
||||
Divider()
|
||||
summaryBar
|
||||
}
|
||||
}
|
||||
|
||||
private var summaryBar: some View {
|
||||
HStack(spacing: 16) {
|
||||
summaryItem("text.quote", "\(model.phrases.count) frases")
|
||||
summaryItem("sparkles", "\(model.emphasisCount) com ênfase")
|
||||
summaryItem("scissors", "\(model.removedCount) fora do corte")
|
||||
summaryItem("clock", durationLabel(model.keptDuration))
|
||||
if !model.zooms.isEmpty {
|
||||
summaryItem("plus.magnifyingglass", "\(model.zooms.count) zooms")
|
||||
}
|
||||
Spacer()
|
||||
if let phrase = selectedPhrase, !phrase.reason.isEmpty {
|
||||
Label(phrase.reason, systemImage: "brain")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
.lineLimit(1).truncationMode(.tail)
|
||||
}
|
||||
}
|
||||
.padding(.horizontal, 14)
|
||||
.padding(.vertical, 8)
|
||||
}
|
||||
|
||||
private func summaryItem(_ icon: String, _ text: String) -> some View {
|
||||
Label(text, systemImage: icon).font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
|
||||
private func durationLabel(_ seconds: Double) -> String {
|
||||
String(format: "%02d:%02d finais", Int(seconds) / 60, Int(seconds) % 60)
|
||||
}
|
||||
|
||||
/// Says which frame is on screen, and lets the editor flip to the raw take.
|
||||
/// Without it a centred crop looks like the footage itself, and someone
|
||||
/// would judge framing on an approximation without knowing it.
|
||||
@ViewBuilder
|
||||
private var framingBadge: some View {
|
||||
if model.projectAspect != nil, abs((model.projectAspect ?? 0) - model.videoAspect) > 0.01 {
|
||||
Button {
|
||||
model.matchProjectFraming.toggle()
|
||||
} label: {
|
||||
Label(model.matchProjectFraming ? "Enquadramento do projeto" : "Mídia original",
|
||||
systemImage: model.matchProjectFraming ? "crop" : "rectangle.expand.vertical")
|
||||
.font(.caption2)
|
||||
}
|
||||
.buttonStyle(.borderless)
|
||||
.padding(6)
|
||||
.background(Capsule().fill(.black.opacity(0.45)))
|
||||
.foregroundStyle(.white)
|
||||
.padding(8)
|
||||
.help("A fonte é horizontal e o projeto é vertical — o preview mostra o corte central aproximado. O enquadramento real de cada clipe vem do Final Cut.")
|
||||
}
|
||||
}
|
||||
|
||||
private func pickMedia() {
|
||||
let panel = NSOpenPanel()
|
||||
panel.canChooseFiles = true
|
||||
panel.canChooseDirectories = false
|
||||
panel.allowsMultipleSelection = false
|
||||
panel.prompt = "Usar esta mídia"
|
||||
panel.message = model.source.isEmpty
|
||||
? "Escolha o arquivo de vídeo desta gravação."
|
||||
: "Escolha onde está \(model.source)."
|
||||
if panel.runModal() == .OK, let url = panel.url {
|
||||
model.useMedia(at: url.path)
|
||||
}
|
||||
}
|
||||
|
||||
private var selectedPhrase: ReviewPhrase? {
|
||||
guard let selection = model.selection else { return nil }
|
||||
return model.phrases.first { $0.id == selection }
|
||||
}
|
||||
|
||||
// MARK: - Inspector de frases
|
||||
|
||||
private var inspectorPane: some View {
|
||||
VStack(spacing: 0) {
|
||||
inspectorHeader
|
||||
Divider()
|
||||
List(selection: $model.selection) {
|
||||
ForEach($model.phrases) { $phrase in
|
||||
PhraseRow(phrase: $phrase, model: model)
|
||||
.tag(phrase.id)
|
||||
}
|
||||
}
|
||||
.listStyle(.inset)
|
||||
.onChange(of: model.selection) { _, newValue in
|
||||
if let newValue { model.goTo(phraseID: newValue) }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var inspectorHeader: some View {
|
||||
VStack(alignment: .leading, spacing: 6) {
|
||||
Text("Frases").font(.headline)
|
||||
Text("Só as frases com ênfase recebem zoom e legenda dinâmica. O resto fica com legenda comum.")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
if !model.emotionAvailable {
|
||||
Label("Emoção da fala não foi detectada nesta análise — ligue em Avançado → Análise de Voz e refaça o passo 3.",
|
||||
systemImage: "waveform.path.ecg")
|
||||
.font(.caption2).foregroundStyle(.secondary)
|
||||
}
|
||||
HStack(spacing: 8) {
|
||||
Button("Limpar ênfases") { model.setEmphasisForAll(0) }
|
||||
.buttonStyle(.link).font(.caption)
|
||||
Spacer()
|
||||
Text("0–3 no teclado · ← → navega")
|
||||
.font(.caption2).foregroundStyle(.secondary)
|
||||
}
|
||||
}
|
||||
.padding(12)
|
||||
}
|
||||
}
|
||||
|
||||
/// One phrase in the inspector: the line as it will be said, plus every
|
||||
/// decision attached to it. Kept in one row on purpose — jumping to a separate
|
||||
/// detail pane to set a toggle would double the clicks on the most repeated
|
||||
/// action in the screen.
|
||||
private struct PhraseRow: View {
|
||||
@Binding var phrase: ReviewPhrase
|
||||
@ObservedObject var model: PhraseReviewModel
|
||||
@State private var isEditing = false
|
||||
|
||||
var body: some View {
|
||||
VStack(alignment: .leading, spacing: 6) {
|
||||
HStack(spacing: 6) {
|
||||
Text(phrase.timecode)
|
||||
.font(.system(.caption2, design: .monospaced))
|
||||
.foregroundStyle(.secondary)
|
||||
if phrase.takeBoundary {
|
||||
Image(systemName: "scissors.badge.ellipsis")
|
||||
.font(.caption2).foregroundStyle(.orange)
|
||||
.help("Nova tomada começa aqui")
|
||||
}
|
||||
if phrase.isTrimmed {
|
||||
Image(systemName: "arrow.left.and.right.square")
|
||||
.font(.caption2).foregroundStyle(.blue)
|
||||
.help("Frase cortada nas pontas")
|
||||
}
|
||||
if model.emotionAvailable {
|
||||
emotionChip
|
||||
}
|
||||
Spacer()
|
||||
Toggle("", isOn: $phrase.active)
|
||||
.toggleStyle(.switch)
|
||||
.controlSize(.mini)
|
||||
.labelsHidden()
|
||||
.help(phrase.active ? "No corte" : "Fora do corte")
|
||||
}
|
||||
|
||||
if isEditing {
|
||||
TextField("Texto da frase", text: $phrase.text, axis: .vertical)
|
||||
.textFieldStyle(.roundedBorder)
|
||||
.font(.callout)
|
||||
.onSubmit { isEditing = false }
|
||||
} else {
|
||||
Text(phrase.text.isEmpty ? "(sem texto)" : phrase.text)
|
||||
.font(.callout)
|
||||
.foregroundStyle(phrase.active ? .primary : .secondary)
|
||||
.strikethrough(!phrase.active)
|
||||
.onTapGesture(count: 2) { isEditing = true }
|
||||
}
|
||||
|
||||
HStack(spacing: 8) {
|
||||
Picker("", selection: $phrase.emphasis) {
|
||||
ForEach(0..<4, id: \.self) { level in
|
||||
Text(EmphasisPalette.label(level)).tag(level)
|
||||
}
|
||||
}
|
||||
.pickerStyle(.segmented)
|
||||
.controlSize(.mini)
|
||||
.labelsHidden()
|
||||
.disabled(!phrase.active)
|
||||
|
||||
Picker("", selection: $phrase.track) {
|
||||
Text("Roteiro").tag(ReviewPhrase.trackScript)
|
||||
Text("Bastidor").tag(ReviewPhrase.trackBackstage)
|
||||
}
|
||||
.pickerStyle(.menu)
|
||||
.controlSize(.mini)
|
||||
.labelsHidden()
|
||||
.frame(width: 92)
|
||||
}
|
||||
|
||||
if model.selection == phrase.id && !phrase.words.isEmpty {
|
||||
wordTrimmer
|
||||
}
|
||||
}
|
||||
.padding(.vertical, 4)
|
||||
.opacity(phrase.active ? 1 : 0.55)
|
||||
}
|
||||
|
||||
/// The delivery emotion the acoustics suggest. Shown faded below its own
|
||||
/// confidence: a guess the analysis is unsure about should not compete for
|
||||
/// attention with the emphasis decision, which is the point of the row.
|
||||
private var emotionChip: some View {
|
||||
let (label, icon) = ReviewPhrase.emotionLabel(phrase.emotion)
|
||||
return Label(label, systemImage: icon)
|
||||
.font(.caption2)
|
||||
.padding(.horizontal, 5)
|
||||
.padding(.vertical, 1)
|
||||
.background(
|
||||
Capsule().fill(Color.secondary.opacity(0.12))
|
||||
)
|
||||
.foregroundStyle(phrase.emotionConfidence >= 0.5 ? .secondary : .tertiary)
|
||||
.help("Emoção da entrega: \(label) — confiança \(Int(phrase.emotionConfidence * 100))%")
|
||||
}
|
||||
|
||||
/// Trimming by pointing at the transcript: click a word to start the phrase
|
||||
/// there, option-click to end it there. Same edit as dragging the block's
|
||||
/// edge on the timeline, but reachable while reading the line.
|
||||
private var wordTrimmer: some View {
|
||||
VStack(alignment: .leading, spacing: 4) {
|
||||
HStack(spacing: 4) {
|
||||
Text("Cortar pelas palavras").font(.caption2).foregroundStyle(.secondary)
|
||||
Spacer()
|
||||
if phrase.isTrimmed {
|
||||
Button("Inteira") { model.resetTrim(phrase.id) }
|
||||
.buttonStyle(.link).font(.caption2)
|
||||
}
|
||||
}
|
||||
FlowWords(words: phrase.words, phrase: phrase) { word, edge in
|
||||
model.trimToWord(word, edge: edge, in: phrase.id)
|
||||
}
|
||||
Text("Clique = começa aqui · ⌥clique = termina aqui")
|
||||
.font(.caption2).foregroundStyle(.tertiary)
|
||||
}
|
||||
.padding(.top, 2)
|
||||
}
|
||||
}
|
||||
|
||||
/// The phrase's words as wrapping chips, dimmed where they fall outside the trim.
|
||||
private struct FlowWords: View {
|
||||
let words: [ReviewWord]
|
||||
let phrase: ReviewPhrase
|
||||
let onTrim: (ReviewWord, TrimEdge) -> Void
|
||||
|
||||
var body: some View {
|
||||
// A LazyVGrid with adaptive columns wraps chips without a custom layout;
|
||||
// phrases are short enough that the slight raggedness beats the cost of
|
||||
// hand-rolling a flow layout here.
|
||||
LazyVGrid(columns: [GridItem(.adaptive(minimum: 44), spacing: 3)],
|
||||
alignment: .leading, spacing: 3) {
|
||||
ForEach(words) { word in
|
||||
let kept = word.start >= phrase.trimStart - 0.001 && word.end <= phrase.trimEnd + 0.001
|
||||
Text(word.text)
|
||||
.font(.caption2)
|
||||
.padding(.horizontal, 4)
|
||||
.padding(.vertical, 2)
|
||||
.background(
|
||||
RoundedRectangle(cornerRadius: 3)
|
||||
.fill(kept ? Color.accentColor.opacity(0.12) : Color.secondary.opacity(0.08))
|
||||
)
|
||||
.foregroundStyle(kept ? .primary : .secondary)
|
||||
.strikethrough(!kept)
|
||||
.onTapGesture {
|
||||
onTrim(word, NSEvent.modifierFlags.contains(.option) ? .end : .start)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -44,12 +44,26 @@ enum PythonBridge {
|
||||
return ["python3", scriptURL.path]
|
||||
}
|
||||
|
||||
/// `admin/models_api.py` lives outside `code/`, but its dependencies
|
||||
/// (`pyproject.toml`, `.venv`) live inside it. `uv run` picks the
|
||||
/// environment from the process's cwd, not from the script path — so
|
||||
/// running with cwd at the repo root made `uv` create/use a second,
|
||||
/// empty `.venv` there, silently ignoring everything installed into
|
||||
/// `code/.venv` (this cost a real debugging session: librosa/pyannote
|
||||
/// installed successfully but the app kept reporting them missing).
|
||||
/// Every `uv run` must share the same cwd as `uv sync` to see the same
|
||||
/// environment.
|
||||
static var workingDirectory: URL {
|
||||
projectRoot
|
||||
codeDirectory
|
||||
}
|
||||
|
||||
/// Directory containing `pyproject.toml` — where `uv sync` must run from.
|
||||
static var codeDirectory: URL {
|
||||
projectRoot.appendingPathComponent("code")
|
||||
}
|
||||
|
||||
/// Locate `uv` on PATH or in common install locations.
|
||||
private static func findUV() -> String? {
|
||||
static func findUV() -> String? {
|
||||
if let onPath = which("uv") { return onPath }
|
||||
let candidates = [
|
||||
"/usr/local/bin/uv",
|
||||
@@ -148,6 +162,59 @@ enum PythonBridge {
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - uv sync (installing optional extras, e.g. acoustic analysis)
|
||||
|
||||
/// Runs `uv <arguments>` from `codeDirectory` (where `pyproject.toml`
|
||||
/// lives), streaming each output line as plain text — used for
|
||||
/// `sync --extra intelligence` so "Modelos" can install the librosa
|
||||
/// extra without the user opening a terminal.
|
||||
static func runUV(arguments: [String],
|
||||
onLine: @escaping (String) -> Void,
|
||||
completion: @escaping (Int, String?) -> Void) {
|
||||
guard let uv = findUV() else {
|
||||
completion(1, "uv não encontrado. Instale com: curl -LsSf https://astral.sh/uv/install.sh | sh")
|
||||
return
|
||||
}
|
||||
let process = Process()
|
||||
process.executableURL = URL(fileURLWithPath: "/usr/bin/env")
|
||||
process.arguments = [uv] + arguments
|
||||
process.currentDirectoryURL = codeDirectory
|
||||
|
||||
let pipe = Pipe()
|
||||
process.standardOutput = pipe
|
||||
process.standardError = pipe
|
||||
|
||||
var buffer = ""
|
||||
let lock = NSLock()
|
||||
pipe.fileHandleForReading.readabilityHandler = { handle in
|
||||
let data = handle.availableData
|
||||
guard !data.isEmpty, let s = String(data: data, encoding: .utf8) else { return }
|
||||
lock.lock()
|
||||
buffer += s
|
||||
let parts = buffer.split(separator: "\n", omittingEmptySubsequences: false)
|
||||
buffer = String(parts.last ?? "")
|
||||
let lines = parts.dropLast()
|
||||
lock.unlock()
|
||||
for line in lines where !line.isEmpty { onLine(String(line)) }
|
||||
}
|
||||
|
||||
process.terminationHandler = { p in
|
||||
pipe.fileHandleForReading.readabilityHandler = nil
|
||||
lock.lock()
|
||||
let last = buffer.trimmingCharacters(in: .whitespacesAndNewlines)
|
||||
buffer = ""
|
||||
lock.unlock()
|
||||
if !last.isEmpty { onLine(last) }
|
||||
completion(Int(p.terminationStatus), p.terminationStatus == 0 ? nil : "uv sync terminou com erro (código \(p.terminationStatus)).")
|
||||
}
|
||||
|
||||
do {
|
||||
try process.run()
|
||||
} catch {
|
||||
completion(1, error.localizedDescription)
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Convenience: single JSON result
|
||||
|
||||
/// Runs a command and delivers the first parsed JSON document as the result.
|
||||
|
||||
@@ -0,0 +1,506 @@
|
||||
import SwiftUI
|
||||
|
||||
/// Colors shared by the timeline and the inspector, so a block and its row in
|
||||
/// the list always read as the same thing.
|
||||
enum EmphasisPalette {
|
||||
static func color(_ level: Int) -> Color {
|
||||
switch level {
|
||||
case 1: return Color.blue
|
||||
case 2: return Color.orange
|
||||
case 3: return Color.pink
|
||||
default: return Color.secondary
|
||||
}
|
||||
}
|
||||
|
||||
static func label(_ level: Int) -> String {
|
||||
switch level {
|
||||
case 1: return "Leve"
|
||||
case 2: return "Média"
|
||||
case 3: return "Forte"
|
||||
default: return "Sem"
|
||||
}
|
||||
}
|
||||
|
||||
static func speakerColor(_ speaker: String, among speakers: [String]) -> Color {
|
||||
let palette: [Color] = [.teal, .purple, .green, .indigo, .brown, .cyan]
|
||||
guard let index = speakers.firstIndex(of: speaker) else { return .gray }
|
||||
return palette[index % palette.count]
|
||||
}
|
||||
}
|
||||
|
||||
/// The timeline strip: four stacked tracks over one shared time axis.
|
||||
///
|
||||
/// Phrases are laid out as real views rather than drawn into a Canvas, because
|
||||
/// every one of them is a target — click to select, drag its edge to trim,
|
||||
/// right-click to change emphasis. The dense per-word energy track *is* a
|
||||
/// Canvas: it has thousands of bars and nothing to hit.
|
||||
struct TimelineTracksView: View {
|
||||
@ObservedObject var model: PhraseReviewModel
|
||||
|
||||
private let rulerHeight: CGFloat = 18
|
||||
private let phraseHeight: CGFloat = 46
|
||||
private let energyHeight: CGFloat = 34
|
||||
private let stripHeight: CGFloat = 12
|
||||
private let handleWidth: CGFloat = 8
|
||||
|
||||
private let gutterWidth: CGFloat = 92
|
||||
private let trackSpacing: CGFloat = 4
|
||||
|
||||
private var pps: CGFloat { CGFloat(model.pixelsPerSecond) }
|
||||
private var contentWidth: CGFloat { max(320, CGFloat(model.duration) * pps) }
|
||||
|
||||
/// Name, icon and height of each lane, in the order they stack. The gutter
|
||||
/// and the tracks are built from this one list so a label can never drift
|
||||
/// off the lane it names.
|
||||
private var lanes: [(label: String, icon: String, height: CGFloat)] {
|
||||
[
|
||||
("", "", rulerHeight),
|
||||
("Zooms", "plus.magnifyingglass", stripHeight + 6),
|
||||
("Frases", "text.quote", phraseHeight),
|
||||
("Energia", "waveform", energyHeight),
|
||||
("Emoção", "face.smiling", stripHeight),
|
||||
("Locutor", "person.wave.2", stripHeight),
|
||||
("Roteiro", "list.bullet.rectangle", stripHeight),
|
||||
]
|
||||
}
|
||||
|
||||
var body: some View {
|
||||
VStack(spacing: 0) {
|
||||
toolbar
|
||||
Divider()
|
||||
HStack(alignment: .top, spacing: 0) {
|
||||
gutter
|
||||
Divider()
|
||||
timelineScroller
|
||||
}
|
||||
}
|
||||
.background(Color(nsColor: .underPageBackgroundColor))
|
||||
}
|
||||
|
||||
/// Fixed column naming each lane. Without it the stripes are six colours
|
||||
/// with no way to tell which one is emotion and which one is the speaker.
|
||||
private var gutter: some View {
|
||||
VStack(alignment: .leading, spacing: trackSpacing) {
|
||||
ForEach(lanes.indices, id: \.self) { index in
|
||||
let lane = lanes[index]
|
||||
HStack(spacing: 4) {
|
||||
if !lane.icon.isEmpty {
|
||||
Image(systemName: lane.icon).font(.system(size: 9))
|
||||
}
|
||||
Text(lane.label).font(.system(size: 10))
|
||||
Spacer(minLength: 0)
|
||||
}
|
||||
.foregroundStyle(.secondary)
|
||||
.frame(height: lane.height, alignment: .center)
|
||||
}
|
||||
}
|
||||
.padding(.horizontal, 8)
|
||||
.padding(.vertical, 8)
|
||||
.frame(width: gutterWidth, alignment: .leading)
|
||||
}
|
||||
|
||||
private var timelineScroller: some View {
|
||||
ScrollViewReader { proxy in
|
||||
ScrollView([.horizontal]) {
|
||||
ZStack(alignment: .topLeading) {
|
||||
VStack(alignment: .leading, spacing: trackSpacing) {
|
||||
ruler
|
||||
zoomTrack
|
||||
phraseTrack
|
||||
energyTrack
|
||||
emotionTrack
|
||||
speakerTrack
|
||||
scriptTrack
|
||||
}
|
||||
.frame(width: contentWidth, alignment: .leading)
|
||||
rangeOverlay
|
||||
playhead
|
||||
// Anchors the auto-scroll: one invisible marker per
|
||||
// phrase, so selecting a line off-screen brings it in.
|
||||
ForEach(model.phrases) { phrase in
|
||||
Color.clear
|
||||
.frame(width: 1, height: 1)
|
||||
.offset(x: x(phrase.start))
|
||||
.id(phrase.id)
|
||||
}
|
||||
}
|
||||
.padding(.vertical, 8)
|
||||
.contentShape(Rectangle())
|
||||
.gesture(scrubGesture)
|
||||
.contextMenu { timelineMenu }
|
||||
}
|
||||
.onChange(of: model.selection) { _, newValue in
|
||||
guard let newValue else { return }
|
||||
withAnimation(.easeOut(duration: 0.2)) {
|
||||
proxy.scrollTo(newValue, anchor: .center)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Barra de controles
|
||||
|
||||
private var toolbar: some View {
|
||||
HStack(spacing: 12) {
|
||||
Button {
|
||||
model.togglePlay()
|
||||
} label: {
|
||||
Image(systemName: model.isPlaying ? "pause.fill" : "play.fill")
|
||||
}
|
||||
.buttonStyle(.borderless)
|
||||
.help("Reproduzir (espaço)")
|
||||
.disabled(model.player == nil)
|
||||
|
||||
Text(timecode(model.currentTime))
|
||||
.font(.system(.caption, design: .monospaced))
|
||||
.foregroundStyle(.secondary)
|
||||
|
||||
Button {
|
||||
model.playSelectedPhrase()
|
||||
} label: {
|
||||
Image(systemName: "play.rectangle")
|
||||
}
|
||||
.buttonStyle(.borderless)
|
||||
.help("Tocar só a frase selecionada (⏎)")
|
||||
.disabled(model.player == nil || model.selection == nil)
|
||||
|
||||
Toggle("Pular removidos", isOn: $model.skipRemoved)
|
||||
.toggleStyle(.checkbox)
|
||||
.font(.caption)
|
||||
.help("Durante a reprodução, salta os trechos desativados — mostra como o corte ficou.")
|
||||
|
||||
Button {
|
||||
model.addZoomForRange()
|
||||
} label: {
|
||||
Label("Zoom no trecho", systemImage: "plus.magnifyingglass")
|
||||
}
|
||||
.buttonStyle(.borderless)
|
||||
.font(.caption)
|
||||
.disabled(!model.hasRange)
|
||||
.help("Arraste na timeline para marcar um trecho e crie um zoom nele. A escala vem de Análise de Voz.")
|
||||
|
||||
Spacer()
|
||||
|
||||
legend
|
||||
|
||||
Spacer()
|
||||
|
||||
Image(systemName: "minus.magnifyingglass").foregroundStyle(.secondary)
|
||||
Slider(value: $model.pixelsPerSecond,
|
||||
in: model.minPixelsPerSecond...model.maxPixelsPerSecond)
|
||||
.frame(width: 130)
|
||||
Image(systemName: "plus.magnifyingglass").foregroundStyle(.secondary)
|
||||
}
|
||||
.padding(.horizontal, 12)
|
||||
.padding(.vertical, 8)
|
||||
}
|
||||
|
||||
private var legend: some View {
|
||||
HStack(spacing: 10) {
|
||||
ForEach(0..<4, id: \.self) { level in
|
||||
HStack(spacing: 4) {
|
||||
RoundedRectangle(cornerRadius: 2)
|
||||
.fill(EmphasisPalette.color(level))
|
||||
.frame(width: 10, height: 10)
|
||||
Text(EmphasisPalette.label(level)).font(.caption2)
|
||||
}
|
||||
}
|
||||
}
|
||||
.foregroundStyle(.secondary)
|
||||
}
|
||||
|
||||
// MARK: - Trilhas
|
||||
|
||||
private var ruler: some View {
|
||||
Canvas { context, size in
|
||||
let step = tickStep()
|
||||
var time = 0.0
|
||||
while time <= model.duration {
|
||||
let position = x(time)
|
||||
context.stroke(
|
||||
Path { $0.move(to: CGPoint(x: position, y: size.height - 6))
|
||||
$0.addLine(to: CGPoint(x: position, y: size.height)) },
|
||||
with: .color(.secondary.opacity(0.5))
|
||||
)
|
||||
context.draw(
|
||||
Text(timecode(time)).font(.system(size: 9, design: .monospaced))
|
||||
.foregroundColor(.secondary),
|
||||
at: CGPoint(x: position + 18, y: 6)
|
||||
)
|
||||
time += step
|
||||
}
|
||||
}
|
||||
.frame(width: contentWidth, height: rulerHeight)
|
||||
}
|
||||
|
||||
private var phraseTrack: some View {
|
||||
ZStack(alignment: .topLeading) {
|
||||
RoundedRectangle(cornerRadius: 4)
|
||||
.fill(Color.secondary.opacity(0.06))
|
||||
.frame(width: contentWidth, height: phraseHeight)
|
||||
ForEach(model.phrases) { phrase in
|
||||
phraseBlock(phrase)
|
||||
}
|
||||
}
|
||||
.frame(width: contentWidth, height: phraseHeight, alignment: .topLeading)
|
||||
}
|
||||
|
||||
@ViewBuilder
|
||||
private func phraseBlock(_ phrase: ReviewPhrase) -> some View {
|
||||
let isSelected = model.selection == phrase.id
|
||||
let color = EmphasisPalette.color(phrase.emphasis)
|
||||
let fullWidth = max(2, width(from: phrase.start, to: phrase.end))
|
||||
let keptWidth = max(1, width(from: phrase.trimStart, to: phrase.trimEnd))
|
||||
|
||||
ZStack(alignment: .topLeading) {
|
||||
// The whole line, dim — what is there before the edit.
|
||||
RoundedRectangle(cornerRadius: 4)
|
||||
.fill(color.opacity(phrase.active ? 0.15 : 0.10))
|
||||
.frame(width: fullWidth, height: phraseHeight)
|
||||
|
||||
// What survives: the kept span, drawn solid over it.
|
||||
RoundedRectangle(cornerRadius: 4)
|
||||
.fill(color.opacity(phrase.active ? 0.55 : 0.12))
|
||||
.frame(width: keptWidth, height: phraseHeight)
|
||||
.offset(x: width(from: phrase.start, to: phrase.trimStart))
|
||||
|
||||
Text(phrase.text)
|
||||
.font(.system(size: 10))
|
||||
.lineLimit(2)
|
||||
.padding(.horizontal, 4)
|
||||
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
|
||||
.foregroundStyle(phrase.active ? .primary : .secondary)
|
||||
.strikethrough(!phrase.active)
|
||||
|
||||
RoundedRectangle(cornerRadius: 4)
|
||||
.stroke(isSelected ? Color.accentColor : color.opacity(0.4),
|
||||
lineWidth: isSelected ? 2 : 1)
|
||||
.frame(width: fullWidth, height: phraseHeight)
|
||||
|
||||
if isSelected && phrase.active {
|
||||
trimHandle(phrase, edge: .start)
|
||||
trimHandle(phrase, edge: .end)
|
||||
}
|
||||
}
|
||||
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
|
||||
.offset(x: x(phrase.start))
|
||||
.contentShape(Rectangle())
|
||||
.onTapGesture { model.goTo(phraseID: phrase.id) }
|
||||
.contextMenu { phraseMenu(phrase) }
|
||||
.help(phrase.reason.isEmpty ? phrase.text : "\(phrase.text)\n— \(phrase.reason)")
|
||||
}
|
||||
|
||||
private func trimHandle(_ phrase: ReviewPhrase, edge: TrimEdge) -> some View {
|
||||
let offset = edge == .start
|
||||
? width(from: phrase.start, to: phrase.trimStart)
|
||||
: width(from: phrase.start, to: phrase.trimEnd) - handleWidth
|
||||
return RoundedRectangle(cornerRadius: 2)
|
||||
.fill(Color.accentColor)
|
||||
.frame(width: handleWidth, height: phraseHeight)
|
||||
.offset(x: offset)
|
||||
.gesture(
|
||||
DragGesture(minimumDistance: 1)
|
||||
.onChanged { value in
|
||||
let time = phrase.start + Double((value.location.x) / pps)
|
||||
model.trim(phrase.id, edge: edge, to: time)
|
||||
}
|
||||
)
|
||||
.help(edge == .start ? "Arraste para cortar o começo (pula de palavra em palavra)"
|
||||
: "Arraste para cortar o fim (pula de palavra em palavra)")
|
||||
}
|
||||
|
||||
@ViewBuilder
|
||||
private func phraseMenu(_ phrase: ReviewPhrase) -> some View {
|
||||
Button("Tocar esta frase") {
|
||||
model.goTo(phraseID: phrase.id)
|
||||
model.playSelectedPhrase()
|
||||
}
|
||||
Button(phrase.active ? "Remover do corte" : "Trazer de volta") {
|
||||
model.toggleActive(phrase.id)
|
||||
}
|
||||
Button("Adicionar zoom nesta frase") { model.addZoomForPhrase(phrase.id) }
|
||||
Divider()
|
||||
ForEach(0..<4, id: \.self) { level in
|
||||
Button("Ênfase: \(EmphasisPalette.label(level))") {
|
||||
model.setEmphasis(level, for: phrase.id)
|
||||
}
|
||||
}
|
||||
Divider()
|
||||
Button(phrase.isBackstage ? "Marcar como roteiro" : "Marcar como bastidor") {
|
||||
model.setTrack(phrase.isBackstage ? ReviewPhrase.trackScript : ReviewPhrase.trackBackstage,
|
||||
for: phrase.id)
|
||||
}
|
||||
if phrase.isTrimmed {
|
||||
Divider()
|
||||
Button("Desfazer corte da frase") { model.resetTrim(phrase.id) }
|
||||
}
|
||||
}
|
||||
|
||||
/// Per-word energy/emphasis, straight from the voice timeline — the closest
|
||||
/// thing to a waveform without opening the audio again.
|
||||
private var energyTrack: some View {
|
||||
Canvas { context, size in
|
||||
for phrase in model.phrases {
|
||||
for word in phrase.words {
|
||||
let start = x(word.start)
|
||||
let barWidth = max(1, width(from: word.start, to: word.end) - 1)
|
||||
let height = size.height * CGFloat(max(0.04, word.energy))
|
||||
let rect = CGRect(x: start, y: size.height - height,
|
||||
width: barWidth, height: height)
|
||||
let color = word.emphasis >= 0.65 ? Color.pink
|
||||
: word.emphasis >= 0.45 ? Color.orange
|
||||
: Color.secondary
|
||||
context.fill(Path(rect),
|
||||
with: .color(color.opacity(phrase.active ? 0.6 : 0.2)))
|
||||
}
|
||||
}
|
||||
}
|
||||
.frame(width: contentWidth, height: energyHeight)
|
||||
.background(RoundedRectangle(cornerRadius: 4).fill(Color.secondary.opacity(0.06)))
|
||||
}
|
||||
|
||||
private var speakerTrack: some View {
|
||||
stripTrack { phrase in
|
||||
EmphasisPalette.speakerColor(phrase.speaker, among: model.speakers)
|
||||
}
|
||||
}
|
||||
|
||||
private var scriptTrack: some View {
|
||||
stripTrack { phrase in phrase.isBackstage ? Color.gray : Color.mint }
|
||||
}
|
||||
|
||||
private func stripTrack(_ color: @escaping (ReviewPhrase) -> Color) -> some View {
|
||||
Canvas { context, size in
|
||||
for phrase in model.phrases {
|
||||
let rect = CGRect(x: x(phrase.start), y: 0,
|
||||
width: max(1, width(from: phrase.start, to: phrase.end)),
|
||||
height: size.height)
|
||||
context.fill(Path(roundedRect: rect, cornerRadius: 2),
|
||||
with: .color(color(phrase).opacity(phrase.active ? 0.7 : 0.2)))
|
||||
}
|
||||
}
|
||||
.frame(width: contentWidth, height: stripHeight)
|
||||
}
|
||||
|
||||
private var playhead: some View {
|
||||
Rectangle()
|
||||
.fill(Color.red)
|
||||
.frame(width: 1.5)
|
||||
.offset(x: x(model.currentTime))
|
||||
.allowsHitTesting(false)
|
||||
}
|
||||
|
||||
/// One gesture, two meanings, decided by whether the mouse moved: a click
|
||||
/// parks the playhead, a drag marks in/out. Splitting them across separate
|
||||
/// controls would mean choosing a tool before every action, which is
|
||||
/// exactly the ceremony this screen is meant to avoid.
|
||||
private var scrubGesture: some Gesture {
|
||||
DragGesture(minimumDistance: 0)
|
||||
.onChanged { value in
|
||||
let from = Double(value.startLocation.x / pps)
|
||||
let to = Double(value.location.x / pps)
|
||||
if abs(value.translation.width) > 3 {
|
||||
model.setRange(from: from, to: to)
|
||||
model.seek(to: min(from, to))
|
||||
} else {
|
||||
model.clearRange()
|
||||
model.seek(to: to)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The marked in/out, drawn over every track so the span reads against the
|
||||
/// phrases and the energy at once.
|
||||
private var rangeOverlay: some View {
|
||||
Group {
|
||||
if let span = model.rangeSpan {
|
||||
Rectangle()
|
||||
.fill(Color.accentColor.opacity(0.18))
|
||||
.overlay(Rectangle().stroke(Color.accentColor.opacity(0.6), lineWidth: 1))
|
||||
.frame(width: max(1, width(from: span.start, to: span.end)))
|
||||
.offset(x: x(span.start))
|
||||
.allowsHitTesting(false)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@ViewBuilder
|
||||
private var timelineMenu: some View {
|
||||
if model.hasRange, let span = model.rangeSpan {
|
||||
Button("Adicionar zoom no trecho (\(secondsLabel(span.end - span.start)))") {
|
||||
model.addZoomForRange()
|
||||
}
|
||||
Button("Tocar o trecho") { model.playRange(from: span.start, to: span.end) }
|
||||
Button("Limpar seleção") { model.clearRange() }
|
||||
} else {
|
||||
Text("Arraste na timeline para marcar um trecho")
|
||||
}
|
||||
if let zoom = model.zoom(at: model.currentTime) {
|
||||
Divider()
|
||||
Button("Remover o zoom daqui") { model.removeZoom(zoom.id) }
|
||||
}
|
||||
}
|
||||
|
||||
private func secondsLabel(_ seconds: Double) -> String {
|
||||
String(format: "%.1fs", seconds)
|
||||
}
|
||||
|
||||
/// Punch-ins, on their own lane above the script: they are a second layer
|
||||
/// over the same time, not a property of a phrase.
|
||||
private var zoomTrack: some View {
|
||||
ZStack(alignment: .topLeading) {
|
||||
RoundedRectangle(cornerRadius: 3)
|
||||
.fill(Color.secondary.opacity(0.06))
|
||||
.frame(width: contentWidth, height: stripHeight + 6)
|
||||
ForEach(model.zooms) { zoom in
|
||||
RoundedRectangle(cornerRadius: 3)
|
||||
.fill(Color.yellow.opacity(0.55))
|
||||
.overlay(
|
||||
Image(systemName: "plus.magnifyingglass")
|
||||
.font(.system(size: 8)).foregroundStyle(.black.opacity(0.6))
|
||||
)
|
||||
.frame(width: max(6, width(from: zoom.start, to: zoom.end)),
|
||||
height: stripHeight + 6)
|
||||
.offset(x: x(zoom.start))
|
||||
.help("Zoom marcado — \(secondsLabel(zoom.end - zoom.start)). A escala vem de Análise de Voz.")
|
||||
.contextMenu {
|
||||
Button("Remover este zoom") { model.removeZoom(zoom.id) }
|
||||
}
|
||||
}
|
||||
}
|
||||
.frame(width: contentWidth, height: stripHeight + 6, alignment: .topLeading)
|
||||
}
|
||||
|
||||
/// Delivery emotion per phrase — the fourth signal to read against the text.
|
||||
private var emotionTrack: some View {
|
||||
stripTrack { phrase in
|
||||
switch phrase.emotion {
|
||||
case "excited": return .orange
|
||||
case "tense": return .red
|
||||
case "calm": return .blue
|
||||
case "reflective": return .purple
|
||||
default: return .secondary
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Escala
|
||||
|
||||
private func x(_ time: Double) -> CGFloat { CGFloat(time) * pps }
|
||||
|
||||
private func width(from: Double, to: Double) -> CGFloat {
|
||||
max(0, CGFloat(to - from) * pps)
|
||||
}
|
||||
|
||||
/// Ruler spacing that keeps labels ~80pt apart at any zoom.
|
||||
private func tickStep() -> Double {
|
||||
let candidates: [Double] = [1, 2, 5, 10, 15, 30, 60, 120, 300, 600]
|
||||
let wanted = 80 / Double(pps)
|
||||
return candidates.first { $0 >= wanted } ?? 600
|
||||
}
|
||||
|
||||
private func timecode(_ seconds: Double) -> String {
|
||||
let total = Int(seconds.rounded(.down))
|
||||
return String(format: "%02d:%02d", total / 60, total % 60)
|
||||
}
|
||||
}
|
||||
@@ -271,7 +271,7 @@ struct TranscriptionView: View {
|
||||
Toggle("Marcar o que foi dito na timeline", isOn: $batchMarkers)
|
||||
|
||||
Divider()
|
||||
Toggle("Exportar legendas SRT", isOn: $batchSubtitles)
|
||||
Toggle("Gerar legenda comum (texto editável no FCP)", isOn: $batchSubtitles)
|
||||
|
||||
Divider()
|
||||
batchOptionRow(
|
||||
@@ -793,7 +793,7 @@ struct TranscriptionView: View {
|
||||
if batchFillers { operations.append("remove_filler_words") }
|
||||
if batchPhrases { operations.append("edit_by_transcript") }
|
||||
if batchMarkers { operations.append("transcript_markers") }
|
||||
if batchSubtitles { operations.append("export_srt") }
|
||||
if batchSubtitles { operations.append("generate_plain_subtitles") }
|
||||
// Runs last, on the timing already cut by any earlier steps (see the
|
||||
// "Abrir no Final Cut Pro" fallback chain and exportSubtitles()'s own
|
||||
// preference for `processedPath` — same reasoning).
|
||||
@@ -834,7 +834,7 @@ struct TranscriptionView: View {
|
||||
}
|
||||
let nextPath = result?["path"] as? String ?? currentPath
|
||||
if operation == "remove_silences" { processedPath = nextPath }
|
||||
if operation == "export_srt" { subtitlePaths = result?["paths"] as? [String] ?? [] }
|
||||
if operation == "generate_plain_subtitles" { subtitlePaths = [nextPath] }
|
||||
if operation == "generate_dynamic_subtitles" { dynamicSubtitlesPath = nextPath }
|
||||
processBatchStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder)
|
||||
}
|
||||
|
||||
@@ -23,6 +23,7 @@ struct VoiceAnalysisView: View {
|
||||
} else {
|
||||
energySection
|
||||
emphasisSection
|
||||
zoomSection
|
||||
weightsSection
|
||||
emotionSection
|
||||
resetSection
|
||||
@@ -90,6 +91,44 @@ struct VoiceAnalysisView: View {
|
||||
}
|
||||
}
|
||||
|
||||
private var zoomSection: some View {
|
||||
Section {
|
||||
sliderRow(
|
||||
title: "Zoom na ênfase",
|
||||
value: $config.zoomScale,
|
||||
range: 1.0...3.0,
|
||||
readout: "\(Int(config.zoomScale * 100))%",
|
||||
help: "Fator aplicado nos punch-ins de ênfase. 130% equivale a escala 1,30 no Final Cut."
|
||||
)
|
||||
Picker("Movimento", selection: $config.zoomMode) {
|
||||
Text("Zoom in e out").tag("in_out")
|
||||
Text("Só zoom in").tag("in")
|
||||
Text("Só zoom out").tag("out")
|
||||
}
|
||||
.onChange(of: config.zoomMode) { _, _ in save() }
|
||||
sliderRow(
|
||||
title: "Velocidade do zoom in",
|
||||
value: $config.zoomEaseIn,
|
||||
range: 0.05...2.0,
|
||||
readout: String(format: "%.2fs", config.zoomEaseIn),
|
||||
help: "Duração da entrada do zoom. Menor é mais rápido."
|
||||
)
|
||||
sliderRow(
|
||||
title: "Velocidade do zoom out",
|
||||
value: $config.zoomEaseOut,
|
||||
range: 0.01...2.0,
|
||||
readout: String(format: "%.2fs", config.zoomEaseOut),
|
||||
help: "Duração da saída do zoom. Menor é mais seco."
|
||||
)
|
||||
} header: {
|
||||
Text("Zoom de Ênfase")
|
||||
} footer: {
|
||||
Text("Esses valores viram o padrão para ações de zoom que não trouxerem scale/ease/ease_out no JSON da edição por voz.")
|
||||
.font(.caption)
|
||||
.foregroundStyle(.secondary)
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Emoção
|
||||
|
||||
private var emotionSection: some View {
|
||||
@@ -129,13 +168,14 @@ struct VoiceAnalysisView: View {
|
||||
title: String,
|
||||
value: Binding<Double>,
|
||||
range: ClosedRange<Double> = 0...1,
|
||||
readout: String? = nil,
|
||||
help: String? = nil
|
||||
) -> some View {
|
||||
VStack(alignment: .leading, spacing: 2) {
|
||||
HStack {
|
||||
Text(title)
|
||||
Spacer()
|
||||
Text(String(format: "%.2f", value.wrappedValue))
|
||||
Text(readout ?? String(format: "%.2f", value.wrappedValue))
|
||||
.monospacedDigit()
|
||||
.foregroundStyle(.secondary)
|
||||
}
|
||||
@@ -188,6 +228,10 @@ struct VoiceAnalysisConfig {
|
||||
var weightDuration: Double
|
||||
var emotionEnabled: Bool
|
||||
var emotionSensitivity: Double
|
||||
var zoomScale: Double
|
||||
var zoomMode: String
|
||||
var zoomEaseIn: Double
|
||||
var zoomEaseOut: Double
|
||||
|
||||
static let defaults = VoiceAnalysisConfig(
|
||||
energyThreshold: 0.5,
|
||||
@@ -198,7 +242,11 @@ struct VoiceAnalysisConfig {
|
||||
weightPause: 0.15,
|
||||
weightDuration: 0.10,
|
||||
emotionEnabled: false,
|
||||
emotionSensitivity: 0.5
|
||||
emotionSensitivity: 0.5,
|
||||
zoomScale: 1.30,
|
||||
zoomMode: "in_out",
|
||||
zoomEaseIn: 0.25,
|
||||
zoomEaseOut: 0.04
|
||||
)
|
||||
|
||||
init(
|
||||
@@ -210,7 +258,11 @@ struct VoiceAnalysisConfig {
|
||||
weightPause: Double,
|
||||
weightDuration: Double,
|
||||
emotionEnabled: Bool,
|
||||
emotionSensitivity: Double
|
||||
emotionSensitivity: Double,
|
||||
zoomScale: Double,
|
||||
zoomMode: String,
|
||||
zoomEaseIn: Double,
|
||||
zoomEaseOut: Double
|
||||
) {
|
||||
self.energyThreshold = energyThreshold
|
||||
self.emphasisThreshold = emphasisThreshold
|
||||
@@ -221,6 +273,10 @@ struct VoiceAnalysisConfig {
|
||||
self.weightDuration = weightDuration
|
||||
self.emotionEnabled = emotionEnabled
|
||||
self.emotionSensitivity = emotionSensitivity
|
||||
self.zoomScale = zoomScale
|
||||
self.zoomMode = zoomMode
|
||||
self.zoomEaseIn = zoomEaseIn
|
||||
self.zoomEaseOut = zoomEaseOut
|
||||
}
|
||||
|
||||
/// Lê a resposta do bridge, caindo no padrão para qualquer campo ausente.
|
||||
@@ -236,7 +292,11 @@ struct VoiceAnalysisConfig {
|
||||
weightPause: weights["pause_before"] as? Double ?? defaults.weightPause,
|
||||
weightDuration: weights["duration"] as? Double ?? defaults.weightDuration,
|
||||
emotionEnabled: json["emotion_enabled"] as? Bool ?? defaults.emotionEnabled,
|
||||
emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity
|
||||
emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity,
|
||||
zoomScale: json["zoom_scale"] as? Double ?? defaults.zoomScale,
|
||||
zoomMode: json["zoom_mode"] as? String ?? defaults.zoomMode,
|
||||
zoomEaseIn: json["zoom_ease_in"] as? Double ?? defaults.zoomEaseIn,
|
||||
zoomEaseOut: json["zoom_ease_out"] as? Double ?? defaults.zoomEaseOut
|
||||
)
|
||||
}
|
||||
|
||||
@@ -253,6 +313,10 @@ struct VoiceAnalysisConfig {
|
||||
],
|
||||
"emotion_enabled": emotionEnabled,
|
||||
"emotion_sensitivity": emotionSensitivity,
|
||||
"zoom_scale": zoomScale,
|
||||
"zoom_mode": zoomMode,
|
||||
"zoom_ease_in": zoomEaseIn,
|
||||
"zoom_ease_out": zoomEaseOut,
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,808 @@
|
||||
import SwiftUI
|
||||
import AppKit
|
||||
|
||||
/// Guia passo a passo do fluxo completo: projeto → transcrição → análise de
|
||||
/// voz → copiar para o chat e trazer as decisões → revisar as ênfases →
|
||||
/// processamento final. Existe para que o usuário não precise entender a ordem
|
||||
/// certa de botões espalhados em várias abas — cada etapa só libera a próxima
|
||||
/// quando o passo anterior terminou, e a "ponte" com o chat (que hoje exigia
|
||||
/// sair do app e escolher um arquivo na mão) vira copiar/colar assistido
|
||||
/// dentro da própria tela.
|
||||
enum WizardStep: Int, CaseIterable, Identifiable {
|
||||
case projeto, transcricao, analise, exportarChat, revisar, finalizar, concluido
|
||||
var id: Int { rawValue }
|
||||
|
||||
var titulo: String {
|
||||
switch self {
|
||||
case .projeto: return "Projeto"
|
||||
case .transcricao: return "Transcrever"
|
||||
case .analise: return "Analisar voz"
|
||||
case .exportarChat: return "Decisões da IA"
|
||||
case .revisar: return "Revisar ênfases"
|
||||
case .finalizar: return "Processar"
|
||||
case .concluido: return "Concluído"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct WizardView: View {
|
||||
@State private var step: WizardStep = .projeto
|
||||
|
||||
// Passo 1 — projeto
|
||||
@State private var outputFolder: String?
|
||||
@State private var projectPath: String?
|
||||
@State private var catalog: Catalog?
|
||||
|
||||
// Passo 2 — transcrição
|
||||
@State private var isTranscribing = false
|
||||
@State private var transcribeProgress: Double = 0
|
||||
@State private var transcribeStage = ""
|
||||
@State private var transcribeResults: [TranscriptResult] = []
|
||||
|
||||
// Passo 3 — análise de voz
|
||||
@State private var isAnalyzing = false
|
||||
@State private var voiceTimelinePath: String?
|
||||
@State private var voiceAnalysisMessage = ""
|
||||
@State private var acousticsAvailable: Bool?
|
||||
@State private var showVoiceTimelineReuseAlert = false
|
||||
@State private var existingVoiceTimelinePath: String?
|
||||
|
||||
// Passo 4 — enviar ao chat e trazer as decisões de volta
|
||||
@State private var copiedFeedback = ""
|
||||
@State private var decisionsText = ""
|
||||
@State private var isApplyingDecisions = false
|
||||
@State private var appliedPath: String?
|
||||
@State private var skippedVoiceEdit = false
|
||||
|
||||
// Passo 5 — revisar ênfases
|
||||
@StateObject private var reviewModel = PhraseReviewModel()
|
||||
@State private var reviewLoadedFor: String?
|
||||
@State private var phraseReviewPath: String?
|
||||
|
||||
// Passo 6 — processamento final
|
||||
@State private var finalSilences = true
|
||||
@State private var finalFillers = false
|
||||
@State private var finalSubtitles = true
|
||||
@State private var finalDynamicSubtitles = false
|
||||
@State private var isFinalizing = false
|
||||
@State private var finalStatus = ""
|
||||
@State private var finalPath: String?
|
||||
|
||||
@State private var errorMessage: String?
|
||||
|
||||
var body: some View {
|
||||
VStack(spacing: 0) {
|
||||
stepperHeader
|
||||
.padding(.horizontal, 24)
|
||||
.padding(.top, 20)
|
||||
.padding(.bottom, 16)
|
||||
|
||||
Divider()
|
||||
|
||||
// A revisão é uma sala de edição, não um formulário: ela precisa da
|
||||
// largura toda e rola por conta própria (timeline horizontal, lista
|
||||
// vertical). As demais etapas continuam na coluna estreita, que é o
|
||||
// que mantém um passo a passo legível.
|
||||
if step == .revisar {
|
||||
revisarStep
|
||||
} else {
|
||||
ScrollView {
|
||||
VStack(alignment: .leading, spacing: 18) {
|
||||
if let errorMessage, !errorMessage.isEmpty {
|
||||
Label(errorMessage, systemImage: "exclamationmark.triangle.fill")
|
||||
.foregroundStyle(.red)
|
||||
.padding(.top, 4)
|
||||
}
|
||||
content
|
||||
}
|
||||
.padding(24)
|
||||
.frame(maxWidth: 640, alignment: .leading)
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
|
||||
Divider()
|
||||
navFooter
|
||||
.padding(.horizontal, 24)
|
||||
.padding(.vertical, 16)
|
||||
}
|
||||
.task {
|
||||
loadProjectConfig()
|
||||
await loadCatalog()
|
||||
}
|
||||
.alert("Análise de voz já existe", isPresented: $showVoiceTimelineReuseAlert) {
|
||||
Button("Usar existente") {
|
||||
if let existingVoiceTimelinePath {
|
||||
voiceTimelinePath = existingVoiceTimelinePath
|
||||
voiceAnalysisMessage = "Reaproveitando análise existente: \(existingVoiceTimelinePath)"
|
||||
}
|
||||
}
|
||||
Button("Reprocessar") {
|
||||
analyzeVoice(forceReprocess: true)
|
||||
}
|
||||
Button("Cancelar", role: .cancel) {}
|
||||
} message: {
|
||||
Text("Já existe um arquivo voice_timeline para este projeto. Quer manter o processamento anterior para ganhar tempo?")
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Cabeçalho com os passos
|
||||
|
||||
private var stepperHeader: some View {
|
||||
HStack(spacing: 6) {
|
||||
ForEach(WizardStep.allCases) { s in
|
||||
HStack(spacing: 6) {
|
||||
ZStack {
|
||||
Circle()
|
||||
.fill(colorFor(s))
|
||||
.frame(width: 24, height: 24)
|
||||
if s.rawValue < step.rawValue {
|
||||
Image(systemName: "checkmark")
|
||||
.font(.caption2.weight(.bold))
|
||||
.foregroundStyle(.white)
|
||||
} else {
|
||||
Text("\(s.rawValue + 1)")
|
||||
.font(.caption2.weight(.bold))
|
||||
.foregroundStyle(s == step ? .white : .secondary)
|
||||
}
|
||||
}
|
||||
Text(s.titulo)
|
||||
.font(.caption)
|
||||
.foregroundStyle(s == step ? .primary : .secondary)
|
||||
.fontWeight(s == step ? .semibold : .regular)
|
||||
}
|
||||
if s != WizardStep.allCases.last {
|
||||
Rectangle()
|
||||
.fill(s.rawValue < step.rawValue ? Color.accentColor : Color.secondary.opacity(0.25))
|
||||
.frame(height: 2)
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func colorFor(_ s: WizardStep) -> Color {
|
||||
if s.rawValue < step.rawValue { return .accentColor }
|
||||
if s == step { return .accentColor }
|
||||
return Color.secondary.opacity(0.25)
|
||||
}
|
||||
|
||||
// MARK: - Conteúdo por etapa
|
||||
|
||||
@ViewBuilder
|
||||
private var content: some View {
|
||||
switch step {
|
||||
case .projeto: projetoStep
|
||||
case .transcricao: transcricaoStep
|
||||
case .analise: analiseStep
|
||||
case .exportarChat: exportarChatStep
|
||||
case .revisar: revisarStep
|
||||
case .finalizar: finalizarStep
|
||||
case .concluido: concluidoStep
|
||||
}
|
||||
}
|
||||
|
||||
private var projetoStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Text("1. Escolha o projeto").font(.title3.weight(.semibold))
|
||||
Text("A pasta é onde tudo o que for gerado nesse fluxo fica salvo. O arquivo é o .fcpxml exportado do Final Cut Pro.")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
|
||||
fieldRow(icon: "folder", label: outputFolder ?? "Nenhuma pasta selecionada", isSet: outputFolder != nil) {
|
||||
pickOutputFolder()
|
||||
}
|
||||
fieldRow(icon: "doc.text", label: projectPath.map { URL(fileURLWithPath: $0).lastPathComponent } ?? "Nenhum arquivo selecionado", isSet: projectPath != nil) {
|
||||
pickProjectFile()
|
||||
}
|
||||
|
||||
if looksLikeGeneratedFile(projectPath) {
|
||||
Label("Esse arquivo parece já ter sido processado por este fluxo (o nome tem um sufixo como \"_voice_edit\" ou \"_silence_removed\"). Rodar o wizard de novo em cima dele reaplica os cortes por cima de cortes já feitos. Selecione o .fcpxml original do Final Cut, a menos que a intenção seja mesmo reprocessar.",
|
||||
systemImage: "exclamationmark.triangle.fill")
|
||||
.font(.caption).foregroundStyle(.orange)
|
||||
}
|
||||
|
||||
if (catalog?.installedCount ?? 0) == 0 {
|
||||
Label("Nenhum modelo de transcrição instalado. Baixe um na aba \"Modelos\" antes de continuar.",
|
||||
systemImage: "exclamationmark.triangle.fill")
|
||||
.font(.caption).foregroundStyle(.orange)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var transcricaoStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Text("2. Transcreva o áudio").font(.title3.weight(.semibold))
|
||||
Text("Roda localmente com o modelo escolhido na aba Modelos. Vira a base de tudo que vem depois — o corte por voz, as legendas, os marcadores.")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
|
||||
Button {
|
||||
startTranscription()
|
||||
} label: {
|
||||
if isTranscribing {
|
||||
HStack { ProgressView().controlSize(.small); Text(transcribeStage.isEmpty ? "Transcrevendo…" : transcribeStage) }
|
||||
.frame(maxWidth: .infinity)
|
||||
} else {
|
||||
Label(transcribeResults.isEmpty ? "Transcrever" : "Transcrever novamente", systemImage: "waveform")
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
.buttonStyle(.borderedProminent)
|
||||
.controlSize(.large)
|
||||
.disabled(isTranscribing || projectPath == nil || outputFolder == nil)
|
||||
|
||||
if isTranscribing {
|
||||
VStack(alignment: .leading, spacing: 6) {
|
||||
ProgressView(value: transcribeProgress)
|
||||
Text("\(Int(transcribeProgress * 100))%").font(.caption).foregroundStyle(.secondary).monospacedDigit()
|
||||
}
|
||||
}
|
||||
|
||||
if !transcribeResults.isEmpty {
|
||||
ForEach(transcribeResults, id: \.media) { r in
|
||||
VStack(alignment: .leading, spacing: 4) {
|
||||
HStack {
|
||||
Image(systemName: "checkmark.circle.fill").foregroundStyle(.green)
|
||||
Text(r.media).font(.body.weight(.medium))
|
||||
Spacer()
|
||||
Text("\(r.language) · \(r.words) palavras").font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
Text(r.preview).font(.caption).foregroundStyle(.secondary).lineLimit(2)
|
||||
}
|
||||
.padding(12)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var analiseStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Text("3. Analise a voz").font(.title3.weight(.semibold))
|
||||
Text("Gera o JSON com transcrição, locutor e intensidade (pitch/energia/ritmo) por palavra — é esse arquivo que o chat lê para decidir o que cortar. Não corta nada sozinho.")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
|
||||
Button {
|
||||
analyzeVoice()
|
||||
} label: {
|
||||
if isAnalyzing {
|
||||
HStack { ProgressView().controlSize(.small); Text("Analisando…") }.frame(maxWidth: .infinity)
|
||||
} else {
|
||||
Label(voiceTimelinePath == nil ? "Analisar voz" : "Analisar novamente", systemImage: "waveform.badge.magnifyingglass")
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
.buttonStyle(.borderedProminent)
|
||||
.controlSize(.large)
|
||||
.disabled(isAnalyzing || projectPath == nil || outputFolder == nil)
|
||||
|
||||
if let voiceTimelinePath {
|
||||
VStack(alignment: .leading, spacing: 6) {
|
||||
Label("Análise pronta", systemImage: "checkmark.circle.fill").foregroundStyle(.green)
|
||||
Text(voiceTimelinePath).font(.caption).foregroundStyle(.secondary).lineLimit(1).truncationMode(.middle)
|
||||
}
|
||||
.padding(12)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
|
||||
|
||||
if acousticsAvailable == false {
|
||||
VStack(alignment: .leading, spacing: 4) {
|
||||
Label("Sem análise acústica real", systemImage: "exclamationmark.triangle.fill")
|
||||
.font(.caption.weight(.semibold)).foregroundStyle(.orange)
|
||||
Text("Falta o componente \"librosa\" — os cortes ainda são decididos pelo texto, mas o chat não vai propor zoom com confiança. Instale em Avançado → Modelos → \"Análise Acústica\", e refaça esta etapa depois.")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
.padding(12)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.orange.opacity(0.08)))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var exportarChatStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Text("4. Envie para o chat decidir os cortes").font(.title3.weight(.semibold))
|
||||
Text("Esta é a única etapa manual que sobra: o julgamento de qual tomada usar, onde dar zoom e o que escrever na tela é feito pela IA numa conversa, não por um botão. Copie abaixo, cole numa sessão do Claude e peça pra rodar a skill \"editar-por-voz\".")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
|
||||
if let voiceTimelinePath {
|
||||
Button {
|
||||
copyForChat(path: voiceTimelinePath)
|
||||
} label: {
|
||||
Label("Copiar para colar no chat", systemImage: "doc.on.clipboard")
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
.buttonStyle(.borderedProminent)
|
||||
.controlSize(.large)
|
||||
|
||||
if !copiedFeedback.isEmpty {
|
||||
Label(copiedFeedback, systemImage: "checkmark.circle.fill")
|
||||
.font(.caption).foregroundStyle(.green)
|
||||
}
|
||||
|
||||
VStack(alignment: .leading, spacing: 8) {
|
||||
Text("O que é copiado").font(.caption.weight(.semibold)).foregroundStyle(.secondary)
|
||||
Text("Um pedido pronto + o conteúdo de \(URL(fileURLWithPath: voiceTimelinePath).lastPathComponent), já formatado. É só colar (⌘V) numa conversa com o Claude.")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
.padding(12)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
|
||||
|
||||
Divider().padding(.vertical, 4)
|
||||
|
||||
Text("Cole aqui o que o chat devolveu").font(.callout.weight(.semibold))
|
||||
Text("Na próxima etapa essas decisões aparecem já marcadas na timeline, frase por frase, para você lapidar.")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
|
||||
HStack {
|
||||
Button {
|
||||
if let s = NSPasteboard.general.string(forType: .string) {
|
||||
decisionsText = s
|
||||
}
|
||||
} label: {
|
||||
Label("Colar da área de transferência", systemImage: "list.clipboard")
|
||||
}
|
||||
Spacer()
|
||||
if !decisionsText.isEmpty {
|
||||
Label(jsonIsValid ? "JSON válido" : "JSON inválido",
|
||||
systemImage: jsonIsValid ? "checkmark.circle.fill" : "xmark.circle.fill")
|
||||
.font(.caption)
|
||||
.foregroundStyle(jsonIsValid ? .green : .red)
|
||||
}
|
||||
}
|
||||
|
||||
TextEditor(text: $decisionsText)
|
||||
.font(.system(.caption, design: .monospaced))
|
||||
.frame(minHeight: 140)
|
||||
.padding(8)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
|
||||
.overlay(RoundedRectangle(cornerRadius: 8).stroke(Color.secondary.opacity(0.2)))
|
||||
|
||||
Button {
|
||||
applyDecisions()
|
||||
} label: {
|
||||
if isApplyingDecisions {
|
||||
HStack { ProgressView().controlSize(.small); Text("Aplicando…") }
|
||||
.frame(maxWidth: .infinity)
|
||||
} else {
|
||||
Label("Aplicar decisões", systemImage: "checkmark.seal")
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
.buttonStyle(.borderedProminent)
|
||||
.controlSize(.large)
|
||||
.disabled(isApplyingDecisions || !jsonIsValid)
|
||||
|
||||
if let appliedPath {
|
||||
Label("Decisões aplicadas — \(URL(fileURLWithPath: appliedPath).lastPathComponent)",
|
||||
systemImage: "checkmark.circle.fill")
|
||||
.font(.caption).foregroundStyle(.green)
|
||||
}
|
||||
|
||||
Divider()
|
||||
Button("Pular esta etapa (revisar as ênfases direto, sem passar pela IA)") {
|
||||
skippedVoiceEdit = true
|
||||
appliedPath = nil
|
||||
decisionsText = ""
|
||||
}
|
||||
.buttonStyle(.plain)
|
||||
.font(.caption)
|
||||
.foregroundStyle(.secondary)
|
||||
} else {
|
||||
Label("Volte ao passo anterior e rode a análise de voz primeiro.", systemImage: "exclamationmark.triangle.fill")
|
||||
.font(.caption).foregroundStyle(.orange)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Etapa 5 — a sala de edição. Diferente das outras, não é um formulário
|
||||
/// dentro da coluna do assistente: ocupa a janela toda e se carrega sozinha
|
||||
/// na primeira vez que aparece para aquela análise de voz.
|
||||
private var revisarStep: some View {
|
||||
Group {
|
||||
if voiceTimelinePath != nil {
|
||||
PhraseReviewView(model: reviewModel)
|
||||
} else {
|
||||
VStack(spacing: 8) {
|
||||
Label("Volte ao passo 3 e rode a análise de voz primeiro.",
|
||||
systemImage: "exclamationmark.triangle.fill")
|
||||
.foregroundStyle(.orange)
|
||||
}
|
||||
.frame(maxWidth: .infinity, maxHeight: .infinity)
|
||||
}
|
||||
}
|
||||
.onAppear { loadReviewIfNeeded() }
|
||||
}
|
||||
|
||||
private var finalizarStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Text("6. Finalize o corte").font(.title3.weight(.semibold))
|
||||
Text("Últimos passos automáticos, sem decisão envolvida — rodam com os parâmetros já configurados na aba \"Análise de Voz\" / \"Legendas Dinâmicas\".")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
|
||||
Toggle("Remover silêncios do áudio", isOn: $finalSilences)
|
||||
Toggle("Remover palavras de preenchimento", isOn: $finalFillers)
|
||||
Toggle("Gerar legenda comum (texto editável no FCP)", isOn: $finalSubtitles)
|
||||
Toggle("Gerar legendas dinâmicas (estilo configurado na aba própria)", isOn: $finalDynamicSubtitles)
|
||||
|
||||
Button {
|
||||
finalizeProcessing()
|
||||
} label: {
|
||||
if isFinalizing {
|
||||
HStack { ProgressView().controlSize(.small); Text(finalStatus.isEmpty ? "Processando…" : finalStatus) }
|
||||
.frame(maxWidth: .infinity)
|
||||
} else {
|
||||
Label("Processar", systemImage: "play.fill").frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
.buttonStyle(.borderedProminent)
|
||||
.controlSize(.large)
|
||||
.disabled(isFinalizing || (!finalSilences && !finalFillers && !finalSubtitles && !finalDynamicSubtitles))
|
||||
|
||||
if !finalStatus.isEmpty && !isFinalizing {
|
||||
Text(finalStatus).font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var concluidoStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Label("Concluído", systemImage: "checkmark.seal.fill")
|
||||
.font(.title3.weight(.semibold))
|
||||
.foregroundStyle(.green)
|
||||
if let finalPath {
|
||||
Text(finalPath).font(.caption).foregroundStyle(.secondary).lineLimit(1).truncationMode(.middle)
|
||||
HStack {
|
||||
Button("Abrir no Final Cut Pro") { NSWorkspace.shared.open(URL(fileURLWithPath: finalPath)) }
|
||||
.buttonStyle(.borderedProminent)
|
||||
Button("Mostrar no Finder") {
|
||||
NSWorkspace.shared.activateFileViewerSelecting([URL(fileURLWithPath: finalPath)])
|
||||
}
|
||||
}
|
||||
}
|
||||
Divider().padding(.vertical, 8)
|
||||
Button("Começar outro projeto") { resetWizard() }
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Navegação
|
||||
|
||||
private var navFooter: some View {
|
||||
HStack {
|
||||
if step != .projeto && step != .concluido {
|
||||
Button("Voltar") { goBack() }
|
||||
}
|
||||
Spacer()
|
||||
if step != .concluido {
|
||||
Button(step == .finalizar ? "Concluir" : "Continuar") { goNext() }
|
||||
.buttonStyle(.borderedProminent)
|
||||
.disabled(!canAdvance)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var canAdvance: Bool {
|
||||
switch step {
|
||||
case .projeto: return outputFolder != nil && projectPath != nil
|
||||
case .transcricao: return !transcribeResults.isEmpty
|
||||
case .analise: return voiceTimelinePath != nil
|
||||
case .exportarChat: return appliedPath != nil || skippedVoiceEdit
|
||||
// Revisar é opcional: a sugestão da IA já é utilizável como veio, então
|
||||
// o botão nunca trava aqui — o passo existe para lapidar, não para
|
||||
// exigir mais uma confirmação.
|
||||
case .revisar: return true
|
||||
case .finalizar: return finalPath != nil && !isFinalizing
|
||||
case .concluido: return false
|
||||
}
|
||||
}
|
||||
|
||||
private func goNext() {
|
||||
guard let next = WizardStep(rawValue: step.rawValue + 1) else { return }
|
||||
// Sair da revisão grava o que foi decidido (e as ações derivadas dela)
|
||||
// ao lado da análise de voz. Nada é renderizado aqui: a etapa 6 é que
|
||||
// lê esse arquivo para dar zoom e legenda dinâmica só nas ênfases.
|
||||
if step == .revisar {
|
||||
reviewModel.save { path in
|
||||
phraseReviewPath = path
|
||||
}
|
||||
}
|
||||
step = next
|
||||
}
|
||||
|
||||
private func goBack() {
|
||||
guard let prev = WizardStep(rawValue: step.rawValue - 1) else { return }
|
||||
step = prev
|
||||
}
|
||||
|
||||
private func resetWizard() {
|
||||
step = .projeto
|
||||
transcribeResults = []
|
||||
voiceTimelinePath = nil
|
||||
voiceAnalysisMessage = ""
|
||||
decisionsText = ""
|
||||
appliedPath = nil
|
||||
skippedVoiceEdit = false
|
||||
reviewLoadedFor = nil
|
||||
phraseReviewPath = nil
|
||||
finalStatus = ""
|
||||
finalPath = nil
|
||||
errorMessage = nil
|
||||
}
|
||||
|
||||
// MARK: - Componentes auxiliares
|
||||
|
||||
@ViewBuilder
|
||||
private func fieldRow(icon: String, label: String, isSet: Bool, action: @escaping () -> Void) -> some View {
|
||||
HStack {
|
||||
Image(systemName: icon).foregroundStyle(isSet ? .primary : .secondary)
|
||||
Text(label).lineLimit(1).truncationMode(.middle).foregroundStyle(isSet ? .primary : .secondary)
|
||||
Spacer()
|
||||
Button("Escolher…", action: action)
|
||||
}
|
||||
.padding(12)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
|
||||
}
|
||||
|
||||
/// Todo output do fluxo carrega um destes sufixos no nome (ver
|
||||
/// `_derived_output` / suffixes usados por `apply_voice_actions`,
|
||||
/// `remove_silences`, `generate_dynamic_subtitles` em
|
||||
/// `admin/models_api.py`). Selecionar um deles como "o projeto" no passo
|
||||
/// 1 é o erro que gerou arquivos como `_voice_edit_voice_edit_...`: os
|
||||
/// cortes de voz assumem timestamps da mídia ORIGINAL, então reaplicá-los
|
||||
/// sobre um arquivo já cortado desloca tudo silenciosamente.
|
||||
private static let generatedSuffixes = [
|
||||
"_voice_edit", "_silence_removed", "_dynamic_subtitles",
|
||||
"_transcript_edit", "_fillers_removed", "_markers",
|
||||
]
|
||||
|
||||
private func looksLikeGeneratedFile(_ path: String?) -> Bool {
|
||||
guard let path else { return false }
|
||||
let stem = URL(fileURLWithPath: path).deletingPathExtension().lastPathComponent
|
||||
return Self.generatedSuffixes.contains { stem.contains($0) }
|
||||
}
|
||||
|
||||
private var jsonIsValid: Bool {
|
||||
guard let data = decisionsText.data(using: .utf8), !decisionsText.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { return false }
|
||||
return (try? JSONSerialization.jsonObject(with: data)) != nil
|
||||
}
|
||||
|
||||
// MARK: - Ações — Python bridge
|
||||
|
||||
private func loadProjectConfig() {
|
||||
PythonBridge.call(command: "project_config") { result, _ in
|
||||
DispatchQueue.main.async {
|
||||
guard let result, result["ok"] as? Bool == true else { return }
|
||||
if let folder = result["folder"] as? String, !folder.isEmpty { outputFolder = folder }
|
||||
if let file = result["file"] as? String, !file.isEmpty { projectPath = file }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func loadCatalog() async {
|
||||
PythonBridge.call(command: "catalog") { result, _ in
|
||||
DispatchQueue.main.async {
|
||||
if let result { catalog = Catalog(json: result) }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func pickOutputFolder() {
|
||||
let panel = NSOpenPanel()
|
||||
panel.canChooseFiles = false
|
||||
panel.canChooseDirectories = true
|
||||
panel.allowsMultipleSelection = false
|
||||
panel.prompt = "Usar esta pasta"
|
||||
panel.message = "Escolha a pasta onde os resultados serão salvos."
|
||||
if panel.runModal() == .OK, let url = panel.url {
|
||||
outputFolder = url.path
|
||||
PythonBridge.call(command: "set_project_config", arguments: ["folder": url.path]) { _, _ in }
|
||||
}
|
||||
}
|
||||
|
||||
private func pickProjectFile() {
|
||||
let panel = NSOpenPanel()
|
||||
panel.canChooseFiles = true
|
||||
panel.canChooseDirectories = false
|
||||
panel.allowsMultipleSelection = false
|
||||
panel.prompt = "Selecionar"
|
||||
panel.message = "Selecione o arquivo (.fcpxml) ou o bundle (.fcpxmld) exportado pelo Final Cut Pro."
|
||||
if panel.runModal() == .OK, let url = panel.url {
|
||||
let ext = url.pathExtension.lowercased()
|
||||
if ext == "fcpxml" || ext == "fcpxmld" || ext == "xml" {
|
||||
projectPath = url.path
|
||||
PythonBridge.call(command: "set_project_config", arguments: ["file": url.path]) { _, _ in }
|
||||
} else {
|
||||
errorMessage = "Selecione um arquivo .fcpxml, .fcpxmld ou .xml do Final Cut Pro."
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func startTranscription() {
|
||||
guard let projectPath, let outputFolder else { return }
|
||||
isTranscribing = true
|
||||
errorMessage = nil
|
||||
transcribeResults = []
|
||||
transcribeProgress = 0
|
||||
PythonBridge.run(command: "transcribe", arguments: ["path": projectPath, "output_dir": outputFolder]) { obj in
|
||||
DispatchQueue.main.async {
|
||||
let type = obj["type"] as? String
|
||||
if type == "progress" {
|
||||
transcribeProgress = (obj["fraction"] as? NSNumber)?.doubleValue ?? 0
|
||||
transcribeStage = obj["stage"] as? String ?? ""
|
||||
} else if type == "error" {
|
||||
errorMessage = obj["message"] as? String ?? "Erro na transcrição."
|
||||
} else if type == "result", let arr = obj["transcripts"] as? [[String: Any]] {
|
||||
transcribeResults = arr.map(TranscriptResult.init)
|
||||
}
|
||||
}
|
||||
} completion: { code, err in
|
||||
DispatchQueue.main.async {
|
||||
isTranscribing = false
|
||||
transcribeProgress = 1
|
||||
if code != 0 && transcribeResults.isEmpty {
|
||||
errorMessage = err ?? "A transcrição falhou."
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func analyzeVoice(forceReprocess: Bool = false) {
|
||||
guard let projectPath, let outputFolder else { return }
|
||||
isAnalyzing = true
|
||||
errorMessage = nil
|
||||
PythonBridge.call(command: "analyze_voice", arguments: [
|
||||
"path": projectPath,
|
||||
"output_dir": outputFolder,
|
||||
"force_reprocess": forceReprocess,
|
||||
]) { result, err in
|
||||
DispatchQueue.main.async {
|
||||
isAnalyzing = false
|
||||
guard result?["ok"] as? Bool == true else {
|
||||
errorMessage = result?["error"] as? String ?? err ?? "Falha ao analisar a voz."
|
||||
return
|
||||
}
|
||||
if result?["reused"] as? Bool == true, !forceReprocess {
|
||||
let timelines = result?["timelines"] as? [String] ?? []
|
||||
existingVoiceTimelinePath = timelines.first ?? extractPath(from: result?["message"] as? String ?? "", marker: "**Timeline JSON**:")
|
||||
showVoiceTimelineReuseAlert = true
|
||||
return
|
||||
}
|
||||
let message = result?["message"] as? String ?? ""
|
||||
voiceAnalysisMessage = message
|
||||
if let path = extractPath(from: message, marker: "**Timeline JSON**:") {
|
||||
voiceTimelinePath = path
|
||||
} else {
|
||||
voiceTimelinePath = nil
|
||||
// ok:true não garante que a análise gerou timeline — se
|
||||
// não houver fala detectável no áudio, o Python volta com
|
||||
// sucesso mas sem "Timeline JSON" na mensagem. Sem isso
|
||||
// aqui, a etapa parecia não fazer nada.
|
||||
errorMessage = "A análise terminou mas não encontrou fala reconhecível no áudio. Mensagem do motor: " + (message.isEmpty ? "(vazia)" : message)
|
||||
}
|
||||
checkAcoustics()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A ênfase de voz (energia/tom) depende do `librosa`, dependência
|
||||
/// opcional. Sem ela, a análise ainda transcreve e corta pelo texto,
|
||||
/// mas nunca deveria propor zoom — por isso avisamos aqui, no ponto
|
||||
/// onde o usuário sentiria falta, em vez de só na aba Modelos.
|
||||
private func checkAcoustics() {
|
||||
PythonBridge.call(command: "acoustics_capability") { result, _ in
|
||||
DispatchQueue.main.async {
|
||||
guard let result, result["ok"] as? Bool == true else { return }
|
||||
acousticsAvailable = result["available"] as? Bool
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Localiza uma linha markdown do tipo "- **Marker**: valor" (usado nas
|
||||
/// mensagens do bridge Python) e devolve o valor. Aceita o marcador de
|
||||
/// lista "- " opcional antes dos asteriscos.
|
||||
private func extractPath(from message: String, marker: String) -> String? {
|
||||
for line in message.split(separator: "\n") {
|
||||
var trimmed = Substring(line.trimmingCharacters(in: .whitespaces))
|
||||
if trimmed.hasPrefix("- ") { trimmed = trimmed.dropFirst(2) }
|
||||
if trimmed.hasPrefix(marker) {
|
||||
return trimmed.dropFirst(marker.count).trimmingCharacters(in: .whitespaces)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
private func copyForChat(path: String) {
|
||||
guard let content = try? String(contentsOfFile: path, encoding: .utf8) else {
|
||||
errorMessage = "Não foi possível ler \(path)."
|
||||
return
|
||||
}
|
||||
let prompt = """
|
||||
Use a skill "editar-por-voz" para decidir os cortes deste projeto a partir da timeline de voz abaixo. Devolva só o JSON de decisões (cortes, zooms, textos, marcadores) pronto para eu colar de volta no app.
|
||||
|
||||
```json
|
||||
\(content)
|
||||
```
|
||||
"""
|
||||
let pasteboard = NSPasteboard.general
|
||||
pasteboard.clearContents()
|
||||
pasteboard.setString(prompt, forType: .string)
|
||||
copiedFeedback = "Copiado — cole (⌘V) numa conversa com o Claude."
|
||||
}
|
||||
|
||||
/// Monta a revisão uma vez por análise de voz. Voltar e avançar de novo não
|
||||
/// recarrega: isso jogaria fora as edições manuais em silêncio, que é
|
||||
/// exatamente o que esta tela existe para preservar.
|
||||
private func loadReviewIfNeeded() {
|
||||
guard let voiceTimelinePath, reviewLoadedFor != voiceTimelinePath else { return }
|
||||
reviewLoadedFor = voiceTimelinePath
|
||||
// A pasta do projeto e a do .fcpxml entram como onde procurar a mídia:
|
||||
// a análise de voz guarda só o nome do arquivo, não o caminho.
|
||||
reviewModel.load(
|
||||
voiceTimelinePath: voiceTimelinePath,
|
||||
decisionsJSON: decisionsText,
|
||||
outputFolder: outputFolder,
|
||||
mediaFolder: projectPath.map { URL(fileURLWithPath: $0).deletingLastPathComponent().path }
|
||||
)
|
||||
if let projectPath { reviewModel.loadProjectFormat(projectPath: projectPath) }
|
||||
}
|
||||
|
||||
private func applyDecisions() {
|
||||
guard let projectPath, let outputFolder,
|
||||
let data = decisionsText.data(using: .utf8),
|
||||
let parsed = try? JSONSerialization.jsonObject(with: data) else { return }
|
||||
isApplyingDecisions = true
|
||||
errorMessage = nil
|
||||
PythonBridge.call(command: "apply_voice_actions", arguments: [
|
||||
"path": projectPath,
|
||||
"output_dir": outputFolder,
|
||||
"actions": parsed,
|
||||
]) { result, err in
|
||||
DispatchQueue.main.async {
|
||||
isApplyingDecisions = false
|
||||
guard result?["ok"] as? Bool == true else {
|
||||
errorMessage = result?["error"] as? String ?? err ?? "Falha ao aplicar as decisões."
|
||||
return
|
||||
}
|
||||
appliedPath = result?["path"] as? String ?? projectPath
|
||||
skippedVoiceEdit = false
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func finalizeProcessing() {
|
||||
guard let outputFolder else { return }
|
||||
let startPath = appliedPath ?? projectPath
|
||||
guard let startPath else { return }
|
||||
var operations: [String] = []
|
||||
if finalSilences { operations.append("remove_silences") }
|
||||
if finalFillers { operations.append("remove_filler_words") }
|
||||
if finalSubtitles { operations.append("generate_plain_subtitles") }
|
||||
if finalDynamicSubtitles { operations.append("generate_dynamic_subtitles") }
|
||||
guard !operations.isEmpty else { return }
|
||||
isFinalizing = true
|
||||
errorMessage = nil
|
||||
finalStatus = "Iniciando…"
|
||||
finalizeStep(operations, index: 0, currentPath: startPath, outputFolder: outputFolder)
|
||||
}
|
||||
|
||||
private func finalizeStep(_ operations: [String], index: Int, currentPath: String, outputFolder: String) {
|
||||
guard index < operations.count else {
|
||||
isFinalizing = false
|
||||
finalStatus = "Processamento concluído."
|
||||
finalPath = currentPath
|
||||
return
|
||||
}
|
||||
let operation = operations[index]
|
||||
finalStatus = "Processando: \(operation)…"
|
||||
PythonBridge.call(command: operation, arguments: ["path": currentPath, "output_dir": outputFolder]) { result, err in
|
||||
DispatchQueue.main.async {
|
||||
guard result?["ok"] as? Bool == true else {
|
||||
isFinalizing = false
|
||||
errorMessage = result?["error"] as? String ?? err ?? "Falha em \(operation)."
|
||||
finalStatus = "Processamento interrompido."
|
||||
return
|
||||
}
|
||||
let nextPath = result?["path"] as? String ?? currentPath
|
||||
finalizeStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -382,6 +382,10 @@ DEFAULT_VOICE_ANALYSIS_CONFIG: dict = {
|
||||
"emphasis_floor": 0.25,
|
||||
"emotion_enabled": False,
|
||||
"emotion_sensitivity": 0.5,
|
||||
"zoom_scale": 1.30,
|
||||
"zoom_mode": "in_out",
|
||||
"zoom_ease_in": 0.25,
|
||||
"zoom_ease_out": 0.04,
|
||||
}
|
||||
|
||||
|
||||
@@ -402,12 +406,28 @@ def load_voice_analysis_config() -> dict:
|
||||
stored = _load_config().get("voice_analysis")
|
||||
if not isinstance(stored, dict):
|
||||
return cfg
|
||||
for key in ("energy_threshold", "peak_percentile", "emphasis_floor", "emotion_sensitivity"):
|
||||
for key in (
|
||||
"energy_threshold", "peak_percentile", "emphasis_floor",
|
||||
"emotion_sensitivity", "zoom_scale", "zoom_ease_in", "zoom_ease_out",
|
||||
):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = max(0.0, min(1.0, float(stored[key])))
|
||||
value = float(stored[key])
|
||||
if key == "zoom_scale":
|
||||
cfg[key] = max(1.0, min(3.0, value))
|
||||
elif key.startswith("zoom_ease"):
|
||||
cfg[key] = max(0.01, min(5.0, value))
|
||||
else:
|
||||
cfg[key] = max(0.0, min(1.0, value))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if "emphasis_threshold" in stored and "emphasis_floor" not in stored:
|
||||
try:
|
||||
cfg["emphasis_floor"] = max(0.0, min(1.0, float(stored["emphasis_threshold"])))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if stored.get("zoom_mode") in ("in_out", "in", "out"):
|
||||
cfg["zoom_mode"] = stored["zoom_mode"]
|
||||
if "emotion_enabled" in stored:
|
||||
cfg["emotion_enabled"] = bool(stored["emotion_enabled"])
|
||||
weights = stored.get("emphasis_weights")
|
||||
@@ -428,6 +448,10 @@ def save_voice_analysis_config(
|
||||
emphasis_floor: float | None = None,
|
||||
emotion_enabled: bool | None = None,
|
||||
emotion_sensitivity: float | None = None,
|
||||
zoom_scale: float | None = None,
|
||||
zoom_mode: str | None = None,
|
||||
zoom_ease_in: float | None = None,
|
||||
zoom_ease_out: float | None = None,
|
||||
) -> dict:
|
||||
"""Persist voice-analysis thresholds/weights. Only given fields change.
|
||||
|
||||
@@ -446,6 +470,14 @@ def save_voice_analysis_config(
|
||||
cfg["emotion_enabled"] = bool(emotion_enabled)
|
||||
if emotion_sensitivity is not None:
|
||||
cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity)))
|
||||
if zoom_scale is not None:
|
||||
cfg["zoom_scale"] = max(1.0, min(3.0, float(zoom_scale)))
|
||||
if zoom_mode in ("in_out", "in", "out"):
|
||||
cfg["zoom_mode"] = zoom_mode
|
||||
if zoom_ease_in is not None:
|
||||
cfg["zoom_ease_in"] = max(0.01, min(5.0, float(zoom_ease_in)))
|
||||
if zoom_ease_out is not None:
|
||||
cfg["zoom_ease_out"] = max(0.01, min(5.0, float(zoom_ease_out)))
|
||||
if emphasis_weights is not None:
|
||||
for key, value in emphasis_weights.items():
|
||||
if key in cfg["emphasis_weights"] and value is not None:
|
||||
@@ -536,6 +568,73 @@ def save_dynamic_subtitle_config(**fields) -> dict:
|
||||
return cfg
|
||||
|
||||
|
||||
DEFAULT_PLAIN_SUBTITLE_CONFIG: dict = {
|
||||
"font": "Helvetica Neue",
|
||||
"font_size": 82,
|
||||
"font_color": "1 1 1 1",
|
||||
"max_words": 7,
|
||||
"position_y": -820.0,
|
||||
"uppercase": False,
|
||||
"keep_punctuation": True,
|
||||
"text_scale": 2.0,
|
||||
}
|
||||
|
||||
|
||||
def load_plain_subtitle_config() -> dict:
|
||||
"""Persisted style for simple editable FCPXML title subtitles."""
|
||||
cfg = dict(DEFAULT_PLAIN_SUBTITLE_CONFIG)
|
||||
stored = _load_config().get("plain_subtitles")
|
||||
if not isinstance(stored, dict):
|
||||
return cfg
|
||||
for key in ("position_y", "text_scale"):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = float(stored[key])
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
for key in ("font_size", "max_words"):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = int(stored[key])
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
for key in ("font", "font_color"):
|
||||
if key in stored and isinstance(stored[key], str) and stored[key]:
|
||||
cfg[key] = stored[key]
|
||||
for key in ("uppercase", "keep_punctuation"):
|
||||
if key in stored:
|
||||
cfg[key] = bool(stored[key])
|
||||
cfg["max_words"] = max(1, int(cfg["max_words"]))
|
||||
return cfg
|
||||
|
||||
|
||||
def save_plain_subtitle_config(**fields) -> dict:
|
||||
"""Persist simple subtitle style fields. Only given fields change."""
|
||||
cfg = load_plain_subtitle_config()
|
||||
for key, value in fields.items():
|
||||
if key not in DEFAULT_PLAIN_SUBTITLE_CONFIG or value is None:
|
||||
continue
|
||||
if isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], bool):
|
||||
cfg[key] = bool(value)
|
||||
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], float):
|
||||
try:
|
||||
cfg[key] = float(value)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], int):
|
||||
try:
|
||||
cfg[key] = int(value)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
else:
|
||||
cfg[key] = str(value)
|
||||
cfg["max_words"] = max(1, int(cfg["max_words"]))
|
||||
data = _load_config()
|
||||
data["plain_subtitles"] = cfg
|
||||
_write_config(data)
|
||||
return cfg
|
||||
|
||||
|
||||
# Mirrors the silence thresholds the detection/removal handlers use when no
|
||||
# argument is passed (server_tools/qc.py). Persisted so the app's slider and
|
||||
# any later run agree without threading three fields through every call.
|
||||
|
||||
@@ -0,0 +1,547 @@
|
||||
"""Phrase review — the human pass between the AI's decisions and the render.
|
||||
|
||||
A voice timeline says *how* every line was spoken; a list of voice actions says
|
||||
what the model decided to do about it. Neither is reviewable on its own: the
|
||||
timeline has no editorial intent, and the action list is a set of timecodes with
|
||||
no text attached. This module joins them into the one view an editor can
|
||||
actually judge — the script, phrase by phrase, each carrying the decision that
|
||||
was made about it.
|
||||
|
||||
The phrase is the unit on purpose. Emphasis, in this pipeline, is not a property
|
||||
of a word but of a line: an emphasized phrase gets a punch-in and a dynamic
|
||||
caption, everything else gets a plain caption. Keeping the same granularity in
|
||||
the review, the JSON, and the render means a toggle in the UI maps to exactly
|
||||
one editorial outcome, with nothing to reconcile in between.
|
||||
|
||||
Trimming stays inside the phrase for the same reason. A line is rarely wrong as
|
||||
a whole — it has a false start, or a trailing "né" — so each phrase carries a
|
||||
``trim_start``/``trim_end`` pair that rides on word boundaries. Editing a cut
|
||||
therefore means picking a word, never hunting for a frame, and a partial cut
|
||||
from the model arrives as a trim instead of being rounded away.
|
||||
|
||||
Round-tripping is the other half of the contract. :func:`build_phrase_review`
|
||||
derives the review from actions, :func:`phrase_review_to_actions` derives
|
||||
actions back from the edited review, and everything the editor touched wins over
|
||||
what was inferred — so re-opening the screen shows what was left there, not a
|
||||
re-derivation that quietly discards the edits.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Sequence, Tuple
|
||||
|
||||
from .voice_actions import VoiceAction, merge_cut_ranges, parse_actions
|
||||
|
||||
PHRASE_REVIEW_VERSION = "1.0"
|
||||
|
||||
# Emphasis is stored 0-3 rather than as a float so the UI, the JSON and the
|
||||
# render agree on the same discrete decision. The thresholds map the continuous
|
||||
# `peak_emphasis` of the voice timeline onto those levels when the model gave no
|
||||
# explicit direction for a phrase.
|
||||
EMPHASIS_LEVELS = (0, 1, 2, 3)
|
||||
EMPHASIS_THRESHOLDS = (0.25, 0.45, 0.65)
|
||||
|
||||
# Zoom scale applied per emphasis level when the review is turned back into
|
||||
# actions. Level 0 never produces a zoom. The values stay inside
|
||||
# voice_actions.MIN_ZOOM_SCALE..MAX_ZOOM_SCALE.
|
||||
ZOOM_SCALE_BY_LEVEL = {1: 1.15, 2: 1.3, 3: 1.5}
|
||||
|
||||
# A phrase only survives if most of it does. Speech boundaries from a transcript
|
||||
# are approximate, so a cut clipping a fraction of a second off the tail is a
|
||||
# trim, not a removal — treating that as "phrase deleted" would grey out lines
|
||||
# that are still fully audible.
|
||||
CUT_COVERAGE_TO_DEACTIVATE = 0.6
|
||||
|
||||
# A punch-in shorter than this has no time to ramp in and back out — the writer
|
||||
# rejects the window anyway (see the zoom ease-in/ease-out shape), so refusing
|
||||
# it here turns a silent drop at render time into nothing being placed at all.
|
||||
MIN_ZOOM_DURATION = 0.4
|
||||
|
||||
TRACK_SCRIPT = "roteiro"
|
||||
TRACK_BACKSTAGE = "bastidor"
|
||||
TRACKS = (TRACK_SCRIPT, TRACK_BACKSTAGE)
|
||||
|
||||
|
||||
def resolve_source(
|
||||
source: str, voice_timeline_path: str, extra_dirs: Sequence[str] = ()
|
||||
) -> str:
|
||||
"""The playable path for a timeline's ``source``, or "" when it's gone.
|
||||
|
||||
The voice timeline stores only the media's *file name* — it is written to be
|
||||
read by a model, where a machine-specific absolute path is noise. That makes
|
||||
it useless for opening a preview, so the file is looked up where it can
|
||||
actually be: beside its own timeline JSON first (that is where
|
||||
``analyze_voice`` writes it), then in whatever project folders the caller
|
||||
knows about.
|
||||
"""
|
||||
if not source:
|
||||
return ""
|
||||
candidate = Path(source)
|
||||
if candidate.is_absolute() and candidate.is_file():
|
||||
return str(candidate)
|
||||
|
||||
directories = [Path(voice_timeline_path).parent] if voice_timeline_path else []
|
||||
directories += [Path(d) for d in extra_dirs if d]
|
||||
for directory in directories:
|
||||
found = directory / candidate.name
|
||||
if found.is_file():
|
||||
return str(found)
|
||||
return ""
|
||||
|
||||
|
||||
def _overlap(a_start: float, a_end: float, b_start: float, b_end: float) -> float:
|
||||
"""Seconds shared by two spans (0.0 when they don't touch)."""
|
||||
return max(0.0, min(a_end, b_end) - max(a_start, b_start))
|
||||
|
||||
|
||||
def _cut_coverage(
|
||||
start: float, end: float, cuts: Sequence[Tuple[float, float]]
|
||||
) -> float:
|
||||
"""Fraction of ``start``-``end`` that falls inside ``cuts`` (0-1)."""
|
||||
span = end - start
|
||||
if span <= 0:
|
||||
return 0.0
|
||||
removed = sum(_overlap(start, end, c_start, c_end) for c_start, c_end in cuts)
|
||||
return min(1.0, removed / span)
|
||||
|
||||
|
||||
def snap_to_words(
|
||||
time: float, words: Sequence[dict], fallback: float, edge: str
|
||||
) -> float:
|
||||
"""Move ``time`` onto the nearest word boundary of this phrase.
|
||||
|
||||
Trims are expressed by pointing at a word, so a trim handle that landed
|
||||
mid-word would cut a syllable in half. ``edge`` is ``"in"`` (snap to word
|
||||
starts) or ``"out"`` (snap to word ends); with no word timings available the
|
||||
time is left as-is.
|
||||
"""
|
||||
boundaries = [
|
||||
float(word.get("start" if edge == "in" else "end", 0.0)) for word in words
|
||||
]
|
||||
boundaries = [b for b in boundaries if b > 0]
|
||||
if not boundaries:
|
||||
return fallback
|
||||
return min(boundaries, key=lambda b: abs(b - time))
|
||||
|
||||
|
||||
def _trim_from_cuts(
|
||||
start: float,
|
||||
end: float,
|
||||
words: Sequence[dict],
|
||||
cuts: Sequence[Tuple[float, float]],
|
||||
) -> Tuple[float, float]:
|
||||
"""Read a partial cut over this phrase as a head/tail trim.
|
||||
|
||||
Only cuts that touch an edge become trims: a cut carved out of the middle of
|
||||
a line has no representation here (the phrase is the unit), so it is left
|
||||
for the whole-phrase coverage rule to decide.
|
||||
"""
|
||||
trim_start, trim_end = start, end
|
||||
for cut_start, cut_end in cuts:
|
||||
if _overlap(start, end, cut_start, cut_end) <= 0:
|
||||
continue
|
||||
if cut_start <= trim_start < cut_end < end:
|
||||
trim_start = snap_to_words(cut_end, words, cut_end, "in")
|
||||
if start < cut_start < trim_end <= cut_end:
|
||||
trim_end = snap_to_words(cut_start, words, cut_start, "out")
|
||||
if trim_end <= trim_start:
|
||||
return start, end
|
||||
return trim_start, trim_end
|
||||
|
||||
|
||||
def _level_from_peak(peak: float) -> int:
|
||||
"""Map a 0-1 ``peak_emphasis`` onto a 0-3 level."""
|
||||
for level, threshold in enumerate(EMPHASIS_THRESHOLDS):
|
||||
if peak < threshold:
|
||||
return level
|
||||
return 3
|
||||
|
||||
|
||||
def _level_from_scale(scale: Optional[float]) -> int:
|
||||
"""Map a zoom's scale factor back onto a 0-3 level.
|
||||
|
||||
The model is free to send any scale inside the allowed range, so this picks
|
||||
the nearest level rather than requiring one of our own three values.
|
||||
"""
|
||||
if scale is None:
|
||||
return 2
|
||||
best = 1
|
||||
smallest = None
|
||||
for level, level_scale in ZOOM_SCALE_BY_LEVEL.items():
|
||||
distance = abs(level_scale - float(scale))
|
||||
if smallest is None or distance < smallest:
|
||||
smallest, best = distance, level
|
||||
return best
|
||||
|
||||
|
||||
def _emphasis_from_actions(
|
||||
start: float,
|
||||
end: float,
|
||||
actions: Sequence[VoiceAction],
|
||||
) -> Tuple[Optional[int], str]:
|
||||
"""The level the model asked for on this phrase, and why.
|
||||
|
||||
A ``zoom`` or ``text`` action anywhere inside the phrase is read as "this
|
||||
line is the emphasis" — the model places them on the word that carries the
|
||||
point, not on the whole line, so requiring a full-span match would find
|
||||
nothing. Returns ``(None, "")`` when no action touches the phrase.
|
||||
"""
|
||||
level: Optional[int] = None
|
||||
reason = ""
|
||||
for action in actions:
|
||||
if action.kind not in ("zoom", "text"):
|
||||
continue
|
||||
if _overlap(start, end, action.start, action.end) <= 0:
|
||||
continue
|
||||
if action.kind == "zoom":
|
||||
candidate = _level_from_scale(action.params.get("scale"))
|
||||
else:
|
||||
candidate = 2
|
||||
if level is None or candidate > level:
|
||||
level = candidate
|
||||
reason = action.reason
|
||||
return level, reason
|
||||
|
||||
|
||||
def _cut_reason(
|
||||
start: float, end: float, actions: Sequence[VoiceAction]
|
||||
) -> str:
|
||||
"""The reason given for the cut that removes this phrase."""
|
||||
for action in actions:
|
||||
if action.kind != "cut":
|
||||
continue
|
||||
if _overlap(start, end, action.start, action.end) > 0 and action.reason:
|
||||
return action.reason
|
||||
return ""
|
||||
|
||||
|
||||
def build_phrase_review(
|
||||
timeline: dict,
|
||||
actions: Any = None,
|
||||
voice_timeline_path: str = "",
|
||||
extra_dirs: Sequence[str] = (),
|
||||
) -> dict:
|
||||
"""Join a voice timeline with the AI's actions into a reviewable script.
|
||||
|
||||
``actions`` accepts whatever :func:`~.voice_actions.parse_actions` accepts —
|
||||
a bare list, ``{"actions": [...]}``, or ``None`` when there is no AI pass and
|
||||
the review starts from the acoustics alone. Malformed rows are skipped and
|
||||
reported in ``errors`` rather than raising, matching the rest of the
|
||||
decision pipeline.
|
||||
"""
|
||||
parsed, errors = parse_actions(actions) if actions else ([], [])
|
||||
cuts = merge_cut_ranges(parsed)
|
||||
|
||||
phrases: List[dict] = []
|
||||
for index, segment in enumerate(timeline.get("segments", [])):
|
||||
start = float(segment.get("start", 0.0))
|
||||
end = float(segment.get("end", 0.0))
|
||||
peak = float(segment.get("peak_emphasis", 0.0))
|
||||
take_boundary = bool(segment.get("take_boundary", False))
|
||||
|
||||
words = list(segment.get("words", []))
|
||||
coverage = _cut_coverage(start, end, cuts)
|
||||
active = coverage < CUT_COVERAGE_TO_DEACTIVATE
|
||||
trim_start, trim_end = (
|
||||
_trim_from_cuts(start, end, words, cuts) if active else (start, end)
|
||||
)
|
||||
|
||||
asked_level, asked_reason = _emphasis_from_actions(start, end, parsed)
|
||||
if asked_level is not None:
|
||||
emphasis, reason = asked_level, asked_reason
|
||||
else:
|
||||
emphasis = _level_from_peak(peak)
|
||||
reason = f"ênfase {peak:.2f}" if emphasis else ""
|
||||
if not active:
|
||||
# A removed line carries the reason it was removed; the emphasis it
|
||||
# would have had is kept so re-activating it restores the decision.
|
||||
reason = _cut_reason(start, end, parsed) or reason
|
||||
|
||||
phrases.append(
|
||||
{
|
||||
"index": index,
|
||||
"start": round(start, 3),
|
||||
"end": round(end, 3),
|
||||
"trim_start": round(trim_start, 3),
|
||||
"trim_end": round(trim_end, 3),
|
||||
"text": str(segment.get("text", "")).strip(),
|
||||
"speaker": str(segment.get("speaker", "")),
|
||||
"active": active,
|
||||
"emphasis": emphasis,
|
||||
"track": TRACK_BACKSTAGE if (not active and take_boundary) else TRACK_SCRIPT,
|
||||
"peak_emphasis": round(peak, 3),
|
||||
# Delivery emotion is a heuristic over the acoustics (see
|
||||
# voice_timeline._emotion_for_word) and only means anything when
|
||||
# the analysis actually ran — `emotion_available` below is what
|
||||
# separates "spoken flat" from "never measured".
|
||||
"emotion": str(segment.get("emotion", "neutral")),
|
||||
"emotion_confidence": round(
|
||||
float(segment.get("emotion_confidence", 0.0)), 3
|
||||
),
|
||||
"take_boundary": take_boundary,
|
||||
"gap_before": round(float(segment.get("gap_before", 0.0)), 3),
|
||||
"reason": reason,
|
||||
"words": [
|
||||
{
|
||||
"text": str(word.get("text", "")),
|
||||
"start": round(float(word.get("start", 0.0)), 3),
|
||||
"end": round(float(word.get("end", 0.0)), 3),
|
||||
"energy": round(float(word.get("energy", 0.0)), 3),
|
||||
"emphasis": round(float(word.get("emphasis", 0.0)), 3),
|
||||
}
|
||||
for word in words
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
source = timeline.get("source", "")
|
||||
layers = timeline.get("layers", {}) if isinstance(timeline.get("layers"), dict) else {}
|
||||
return {
|
||||
"version": PHRASE_REVIEW_VERSION,
|
||||
"source": source,
|
||||
"source_path": resolve_source(source, voice_timeline_path, extra_dirs),
|
||||
"duration": round(phrases[-1]["end"], 3) if phrases else 0.0,
|
||||
"speakers": timeline.get("speakers", []),
|
||||
"emotion_available": bool(layers.get("emotion", False)),
|
||||
"phrases": phrases,
|
||||
# Punch-ins the editor places by hand on an arbitrary range, alongside
|
||||
# the whole-phrase zoom that an emphasis level produces. Both end up as
|
||||
# zoom actions; this one exists because the moment worth punching into
|
||||
# is not always a whole sentence.
|
||||
"zooms": [],
|
||||
"errors": errors,
|
||||
}
|
||||
|
||||
|
||||
def _coerce_zoom(raw: Any) -> Optional[Dict[str, float]]:
|
||||
"""Normalize one manually placed zoom range."""
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
try:
|
||||
start = float(raw.get("start"))
|
||||
end = float(raw.get("end"))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if end - start < MIN_ZOOM_DURATION:
|
||||
return None
|
||||
return {"start": start, "end": end}
|
||||
|
||||
|
||||
def _coerce_phrase(raw: Any, index: int) -> Optional[Dict[str, Any]]:
|
||||
"""Normalize one edited phrase row coming back from the UI."""
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
try:
|
||||
start = float(raw.get("start"))
|
||||
end = float(raw.get("end"))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if end <= start:
|
||||
return None
|
||||
try:
|
||||
emphasis = int(raw.get("emphasis", 0))
|
||||
except (TypeError, ValueError):
|
||||
emphasis = 0
|
||||
try:
|
||||
trim_start = float(raw.get("trim_start", start))
|
||||
trim_end = float(raw.get("trim_end", end))
|
||||
except (TypeError, ValueError):
|
||||
trim_start, trim_end = start, end
|
||||
# A trim that escaped the phrase, or inverted, is treated as no trim at all:
|
||||
# the UI is the only thing that writes these, and silently discarding a bad
|
||||
# pair keeps a rounding slip from deleting material the editor kept.
|
||||
if not (start <= trim_start < trim_end <= end):
|
||||
trim_start, trim_end = start, end
|
||||
track = str(raw.get("track", TRACK_SCRIPT))
|
||||
return {
|
||||
"index": int(raw.get("index", index)),
|
||||
"start": start,
|
||||
"end": end,
|
||||
"trim_start": trim_start,
|
||||
"trim_end": trim_end,
|
||||
"text": str(raw.get("text", "")).strip(),
|
||||
"speaker": str(raw.get("speaker", "")),
|
||||
"active": bool(raw.get("active", True)),
|
||||
"emphasis": min(3, max(0, emphasis)),
|
||||
"track": track if track in TRACKS else TRACK_SCRIPT,
|
||||
"reason": str(raw.get("reason", "")),
|
||||
}
|
||||
|
||||
|
||||
def phrase_review_to_actions(review: dict) -> dict:
|
||||
"""Turn an edited review back into the action list the applier consumes.
|
||||
|
||||
Every deactivated phrase becomes a ``cut``, a trimmed one becomes a cut over
|
||||
the head and/or tail it lost, and every emphasized one becomes a ``zoom``
|
||||
scaled by its level. The emphasis flags ride along in ``emphasis_spans`` so
|
||||
the caption step can give those lines the dynamic treatment and everything
|
||||
else the plain one, without re-deriving the decision from the acoustics.
|
||||
"""
|
||||
phrases = [
|
||||
coerced
|
||||
for index, raw in enumerate(review.get("phrases", []))
|
||||
if (coerced := _coerce_phrase(raw, index)) is not None
|
||||
]
|
||||
|
||||
actions: List[dict] = []
|
||||
emphasis_spans: List[dict] = []
|
||||
for phrase in phrases:
|
||||
if not phrase["active"]:
|
||||
actions.append(
|
||||
VoiceAction(
|
||||
kind="cut",
|
||||
start=phrase["start"],
|
||||
end=phrase["end"],
|
||||
reason=phrase["reason"] or "desativada na revisão",
|
||||
speaker=phrase["speaker"],
|
||||
).as_dict()
|
||||
)
|
||||
continue
|
||||
|
||||
# Head and tail the editor trimmed off — each becomes its own cut, so a
|
||||
# false start disappears without taking the line with it.
|
||||
for trim_start, trim_end, where in (
|
||||
(phrase["start"], phrase["trim_start"], "início"),
|
||||
(phrase["trim_end"], phrase["end"], "fim"),
|
||||
):
|
||||
if trim_end - trim_start <= 0:
|
||||
continue
|
||||
actions.append(
|
||||
VoiceAction(
|
||||
kind="cut",
|
||||
start=trim_start,
|
||||
end=trim_end,
|
||||
reason=f"trecho do {where} da frase removido na revisão",
|
||||
speaker=phrase["speaker"],
|
||||
).as_dict()
|
||||
)
|
||||
|
||||
if phrase["emphasis"] >= 1:
|
||||
actions.append(
|
||||
VoiceAction(
|
||||
kind="zoom",
|
||||
start=phrase["trim_start"],
|
||||
end=phrase["trim_end"],
|
||||
params={"scale": ZOOM_SCALE_BY_LEVEL[phrase["emphasis"]]},
|
||||
reason=phrase["reason"] or f"ênfase nível {phrase['emphasis']}",
|
||||
speaker=phrase["speaker"],
|
||||
).as_dict()
|
||||
)
|
||||
emphasis_spans.append(
|
||||
{
|
||||
"start": phrase["trim_start"],
|
||||
"end": phrase["trim_end"],
|
||||
"level": phrase["emphasis"],
|
||||
"text": phrase["text"],
|
||||
}
|
||||
)
|
||||
|
||||
# Hand-placed punch-ins carry no scale on purpose: an omitted scale lets the
|
||||
# applier use the shape configured in "Análise de Voz" (zoom_scale, ease in
|
||||
# and out), so changing that setting restyles every manual zoom instead of
|
||||
# leaving a scale frozen into each one at the moment it was drawn.
|
||||
for raw in review.get("zooms", []):
|
||||
zoom = _coerce_zoom(raw)
|
||||
if zoom is None:
|
||||
continue
|
||||
actions.append(
|
||||
VoiceAction(
|
||||
kind="zoom",
|
||||
start=zoom["start"],
|
||||
end=zoom["end"],
|
||||
reason="zoom marcado na revisão",
|
||||
).as_dict()
|
||||
)
|
||||
|
||||
return {
|
||||
"source": review.get("source", ""),
|
||||
"actions": actions,
|
||||
"emphasis_spans": emphasis_spans,
|
||||
}
|
||||
|
||||
|
||||
def merge_saved_decisions(review: dict, saved: Optional[dict]) -> dict:
|
||||
"""Lay a previously saved review's decisions over a freshly built one.
|
||||
|
||||
Only the editorial fields travel — active, emphasis, track, text, trims.
|
||||
Everything else (words, emotion, energy) is re-derived from the current
|
||||
analysis, so re-running the voice pass with better settings improves the
|
||||
screen instead of being masked by a stale copy of itself, and the saved file
|
||||
never has to carry a duplicate of data it does not own.
|
||||
|
||||
Phrases are matched by index *and* start time: if the analysis changed
|
||||
enough to move a line, the old decision for that slot is dropped rather than
|
||||
applied to a different sentence.
|
||||
"""
|
||||
if not saved:
|
||||
return review
|
||||
|
||||
review["zooms"] = [
|
||||
zoom for raw in saved.get("zooms", []) if (zoom := _coerce_zoom(raw)) is not None
|
||||
]
|
||||
|
||||
by_index = {}
|
||||
for raw in saved.get("phrases", []):
|
||||
if isinstance(raw, dict) and "index" in raw:
|
||||
by_index[raw["index"]] = raw
|
||||
|
||||
for phrase in review["phrases"]:
|
||||
previous = by_index.get(phrase["index"])
|
||||
if previous is None:
|
||||
continue
|
||||
if abs(float(previous.get("start", -1)) - phrase["start"]) > 0.25:
|
||||
continue
|
||||
phrase["active"] = bool(previous.get("active", phrase["active"]))
|
||||
phrase["emphasis"] = min(3, max(0, int(previous.get("emphasis", phrase["emphasis"]))))
|
||||
track = str(previous.get("track", phrase["track"]))
|
||||
phrase["track"] = track if track in TRACKS else phrase["track"]
|
||||
if previous.get("text"):
|
||||
phrase["text"] = str(previous["text"])
|
||||
trim_start = float(previous.get("trim_start", phrase["trim_start"]))
|
||||
trim_end = float(previous.get("trim_end", phrase["trim_end"]))
|
||||
if phrase["start"] <= trim_start < trim_end <= phrase["end"]:
|
||||
phrase["trim_start"], phrase["trim_end"] = trim_start, trim_end
|
||||
|
||||
return review
|
||||
|
||||
|
||||
def review_paths(voice_timeline_path: str) -> Tuple[Path, Path]:
|
||||
"""Where the review and its derived actions live, next to the timeline.
|
||||
|
||||
Both files sit beside the ``_voice_timeline.json`` they came from and are
|
||||
named after it, so a project folder stays readable and re-running the wizard
|
||||
on the same take overwrites its own files instead of accumulating copies.
|
||||
"""
|
||||
base = Path(voice_timeline_path)
|
||||
stem = base.stem
|
||||
if stem.endswith("_voice_timeline"):
|
||||
stem = stem[: -len("_voice_timeline")]
|
||||
return (
|
||||
base.with_name(f"{stem}_phrase_review.json"),
|
||||
base.with_name(f"{stem}_phrase_actions.json"),
|
||||
)
|
||||
|
||||
|
||||
def save_phrase_review(voice_timeline_path: str, review: dict) -> Tuple[Path, Path]:
|
||||
"""Write the edited review and the actions derived from it. Returns both paths."""
|
||||
review_path, actions_path = review_paths(voice_timeline_path)
|
||||
review_path.write_text(
|
||||
json.dumps(review, ensure_ascii=False, indent=2), encoding="utf-8"
|
||||
)
|
||||
actions_path.write_text(
|
||||
json.dumps(phrase_review_to_actions(review), ensure_ascii=False, indent=2),
|
||||
encoding="utf-8",
|
||||
)
|
||||
return review_path, actions_path
|
||||
|
||||
|
||||
def load_phrase_review(voice_timeline_path: str) -> Optional[dict]:
|
||||
"""The review saved earlier for this timeline, or ``None`` if there is none."""
|
||||
review_path, _ = review_paths(voice_timeline_path)
|
||||
if not review_path.is_file():
|
||||
return None
|
||||
try:
|
||||
data = json.loads(review_path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return None
|
||||
return data if isinstance(data, dict) else None
|
||||
@@ -28,8 +28,11 @@ ALLOWED_MODELS = (
|
||||
)
|
||||
|
||||
# Conservative by default: interjections that are near-universally filler.
|
||||
# Portuguese "um"/"uma" are usually articles/numerals inside real phrases
|
||||
# ("de um jeito") rather than discardable hesitations, so only cut them when
|
||||
# the caller explicitly opts in through the fillers argument.
|
||||
# "like" / "so" / "actually" are speech, not noise, unless the user opts in.
|
||||
DEFAULT_FILLERS = ("um", "uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm")
|
||||
DEFAULT_FILLERS = ("uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm")
|
||||
|
||||
_NORM_RE = re.compile(r"[^\w']+")
|
||||
|
||||
|
||||
@@ -82,8 +82,9 @@ def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]:
|
||||
params = dict(params) if isinstance(params, dict) else {}
|
||||
|
||||
if kind == "zoom":
|
||||
if "scale" in params and params.get("scale") is not None:
|
||||
try:
|
||||
scale = float(params.get("scale", 1.3))
|
||||
scale = float(params["scale"])
|
||||
except (TypeError, ValueError):
|
||||
return None, f"{where}: zoom scale must be a number"
|
||||
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
|
||||
@@ -97,6 +98,20 @@ def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]:
|
||||
if not content:
|
||||
return None, f"{where}: text action needs params.content"
|
||||
params["content"] = content[:MAX_TEXT_LENGTH]
|
||||
# Style is optional — omitted fields fall back to the "Legendas
|
||||
# Dinâmicas" emphasis style at apply time (see _apply_placed_action),
|
||||
# so a callout matches the captions' look without the caller having
|
||||
# to know or repeat that configuration. Anything given here wins.
|
||||
for key in ("font", "font_color", "face"):
|
||||
if key in params and not isinstance(params[key], str):
|
||||
del params[key]
|
||||
if "font_size" in params:
|
||||
try:
|
||||
params["font_size"] = int(params["font_size"])
|
||||
except (TypeError, ValueError):
|
||||
del params["font_size"]
|
||||
if "bold" in params:
|
||||
params["bold"] = bool(params["bold"])
|
||||
|
||||
return (
|
||||
VoiceAction(
|
||||
|
||||
@@ -56,12 +56,20 @@ VALUE_SCALES = {
|
||||
"rate_delta": "0-1, how much the local speaking rate departs from the average",
|
||||
"pause_before": "seconds of silence immediately before the word",
|
||||
"emphasis": "0-1 combined index; high values are punch-in/highlight candidates",
|
||||
"emotion": "heuristic label from delivery: neutral, excited, tense, calm, reflective",
|
||||
"emotion_confidence": "0-1 confidence in the heuristic emotion label",
|
||||
"arousal": "0-1 vocal activation from energy/rate/pitch movement",
|
||||
"valence": "0-1 rough positive tone; lower values suggest tension/weight",
|
||||
},
|
||||
"segment": {
|
||||
"gap_before": "seconds of silence before this line",
|
||||
"take_boundary": "true when the gap is long enough that the take likely restarted here",
|
||||
"avg_energy": "0-1 mean loudness across the line",
|
||||
"peak_emphasis": "0-1 highest emphasis of any word in the line",
|
||||
"emotion": "dominant delivery emotion across the line",
|
||||
"emotion_confidence": "0-1 confidence in the dominant segment emotion",
|
||||
"arousal": "0-1 mean vocal activation across the line",
|
||||
"valence": "0-1 mean rough positive tone across the line",
|
||||
},
|
||||
}
|
||||
|
||||
@@ -92,11 +100,74 @@ def _round_word(word: dict) -> dict:
|
||||
"rate_delta": round(word.get("rate_delta", 0.0), 3),
|
||||
"pause_before": round(word.get("pause_before", 0.0), 3),
|
||||
"emphasis": round(word.get("emphasis", 0.0), 3),
|
||||
"emotion": word.get("emotion", "neutral"),
|
||||
"emotion_confidence": round(word.get("emotion_confidence", 0.0), 3),
|
||||
"arousal": round(word.get("arousal", 0.0), 3),
|
||||
"valence": round(word.get("valence", 0.5), 3),
|
||||
"energy_raw": word.get("energy"),
|
||||
"pitch_hz": word.get("pitch_hz"),
|
||||
}
|
||||
|
||||
|
||||
def _emotion_for_word(word: dict, enabled: bool, sensitivity: float) -> dict:
|
||||
"""Classify delivery emotion from normalized acoustic features.
|
||||
|
||||
This is deliberately a local heuristic rather than a claimed clinical
|
||||
emotion model. It gives the editor a useful signal about delivery shape
|
||||
while degrading predictably when acoustic extraction is unavailable.
|
||||
"""
|
||||
if not enabled:
|
||||
return {
|
||||
"emotion": "neutral",
|
||||
"emotion_confidence": 0.0,
|
||||
"arousal": 0.0,
|
||||
"valence": 0.5,
|
||||
}
|
||||
|
||||
energy = float(word.get("energy_norm", 0.0))
|
||||
pitch = float(word.get("pitch_delta", 0.0))
|
||||
rate = float(word.get("rate_delta", 0.0))
|
||||
pause = min(float(word.get("pause_before", 0.0)) / 2.0, 1.0)
|
||||
emphasis = float(word.get("emphasis", 0.0))
|
||||
|
||||
arousal = max(0.0, min(1.0, energy * 0.45 + pitch * 0.25 + rate * 0.20 + emphasis * 0.10))
|
||||
valence = max(0.0, min(1.0, 0.55 + energy * 0.15 - pause * 0.20 - rate * 0.10))
|
||||
|
||||
if arousal >= 0.68 and valence >= 0.50:
|
||||
label = "excited"
|
||||
confidence = arousal
|
||||
elif arousal >= 0.58 and valence < 0.50:
|
||||
label = "tense"
|
||||
confidence = max(arousal, 1.0 - valence)
|
||||
elif arousal <= 0.28 and pause >= 0.25:
|
||||
label = "reflective"
|
||||
confidence = max(1.0 - arousal, pause)
|
||||
elif arousal <= 0.35:
|
||||
label = "calm"
|
||||
confidence = 1.0 - arousal
|
||||
else:
|
||||
label = "neutral"
|
||||
confidence = 1.0 - abs(arousal - 0.5) * 2.0
|
||||
|
||||
confidence = max(0.0, min(1.0, confidence))
|
||||
if confidence < sensitivity:
|
||||
label = "neutral"
|
||||
return {
|
||||
"emotion": label,
|
||||
"emotion_confidence": confidence,
|
||||
"arousal": arousal,
|
||||
"valence": valence,
|
||||
}
|
||||
|
||||
|
||||
def annotate_emotions(words: Sequence[dict], enabled: bool, sensitivity: float) -> List[dict]:
|
||||
"""Attach heuristic emotion labels to enriched word rows."""
|
||||
return [
|
||||
{**w, **_emotion_for_word(w, enabled, sensitivity)}
|
||||
for w in words
|
||||
]
|
||||
|
||||
|
||||
def enrich_words(
|
||||
words: Sequence[dict],
|
||||
pitch_track: Optional[Sequence] = None,
|
||||
@@ -166,6 +237,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]
|
||||
in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end]
|
||||
energies = [w["energy_norm"] for w in in_seg]
|
||||
emphases = [w["emphasis"] for w in in_seg]
|
||||
arousals = [w.get("arousal", 0.0) for w in in_seg]
|
||||
valences = [w.get("valence", 0.5) for w in in_seg]
|
||||
emotions = [w.get("emotion", "neutral") for w in in_seg]
|
||||
dominant = max(set(emotions), key=emotions.count) if emotions else "neutral"
|
||||
emotion_confidences = [
|
||||
w.get("emotion_confidence", 0.0) for w in in_seg if w.get("emotion") == dominant
|
||||
]
|
||||
gap = max(0.0, start - previous_end)
|
||||
rows.append(
|
||||
{
|
||||
@@ -181,6 +259,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]
|
||||
"take_boundary": gap >= TAKE_BOUNDARY_GAP,
|
||||
"avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0,
|
||||
"peak_emphasis": round(max(emphases), 3) if emphases else 0.0,
|
||||
"emotion": dominant,
|
||||
"emotion_confidence": (
|
||||
round(sum(emotion_confidences) / len(emotion_confidences), 3)
|
||||
if emotion_confidences else 0.0
|
||||
),
|
||||
"arousal": round(sum(arousals) / len(arousals), 3) if arousals else 0.0,
|
||||
"valence": round(sum(valences) / len(valences), 3) if valences else 0.5,
|
||||
"words": [_round_word(w) for w in in_seg],
|
||||
}
|
||||
)
|
||||
@@ -425,6 +510,8 @@ def build_voice_timeline(
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
peak_percentile: float = 0.02,
|
||||
emphasis_floor: float = 0.25,
|
||||
emotion_enabled: bool = False,
|
||||
emotion_sensitivity: float = 0.5,
|
||||
progress_cb: Optional[Callable[[float, str], None]] = None,
|
||||
) -> dict:
|
||||
"""Build the consolidated voice timeline for one media file.
|
||||
@@ -445,6 +532,7 @@ def build_voice_timeline(
|
||||
|
||||
report(0.5, "Calculando ênfase...")
|
||||
words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights)
|
||||
words = annotate_emotions(words, emotion_enabled, emotion_sensitivity)
|
||||
|
||||
report(0.7, "Identificando participantes...")
|
||||
tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None
|
||||
@@ -465,6 +553,7 @@ def build_voice_timeline(
|
||||
"transcript": bool(transcript.get("words")),
|
||||
"acoustics": pitch_track is not None or energy_track is not None,
|
||||
"speakers": tracks is not None,
|
||||
"emotion": bool(emotion_enabled),
|
||||
},
|
||||
"scales": VALUE_SCALES,
|
||||
"summary": _summary(
|
||||
|
||||
+63
-2
@@ -2218,13 +2218,23 @@ class FCPXMLModifier:
|
||||
seg_start: 'TimeValue',
|
||||
seg_duration: 'TimeValue',
|
||||
) -> None:
|
||||
"""Remove markers/keywords from *clip* that fall outside the segment range.
|
||||
"""Remove markers/keywords/titles from *clip* that fall outside the segment range.
|
||||
|
||||
After ``split_clip`` deepcopy's the original clip into each segment, every
|
||||
segment inherits all child elements. Markers whose ``start`` falls outside
|
||||
``[seg_start, seg_start + seg_duration)`` are phantom duplicates and must be
|
||||
removed. Keywords that partially overlap get their ``start``/``duration``
|
||||
clamped to the segment boundaries.
|
||||
|
||||
A lane-nested ``<title>`` (a "text" voice action's on-screen callout,
|
||||
or a caption from an earlier `generate_dynamic_subtitles` pass) is
|
||||
the same kind of phantom duplicate, just keyed on ``offset`` instead
|
||||
of ``start`` — its offset lives in the same source-media coordinate
|
||||
space as a marker's ``start`` (see ``add_text_title``/``add_marker``,
|
||||
both anchored at ``parent.start``). Left unfiltered, every further
|
||||
cut (silence removal, filler removal) duplicates it into every
|
||||
resulting piece, so the same word shows up several times across the
|
||||
edited timeline instead of once where it was placed.
|
||||
"""
|
||||
seg_end = seg_start + seg_duration
|
||||
to_remove = []
|
||||
@@ -2234,6 +2244,10 @@ class FCPXMLModifier:
|
||||
child_start = TimeValue.from_timecode(child.get('start', '0s'))
|
||||
if child_start < seg_start or child_start >= seg_end:
|
||||
to_remove.append(child)
|
||||
elif tag == 'title':
|
||||
title_offset = TimeValue.from_timecode(child.get('offset', '0s'))
|
||||
if title_offset < seg_start or title_offset >= seg_end:
|
||||
to_remove.append(child)
|
||||
elif tag == 'keyword':
|
||||
kw_start = TimeValue.from_timecode(child.get('start', '0s'))
|
||||
kw_dur = TimeValue.from_timecode(child.get('duration', '0s'))
|
||||
@@ -2304,6 +2318,7 @@ class FCPXMLModifier:
|
||||
self._filter_children_for_segment(
|
||||
new_clip, current_start, segment_duration
|
||||
)
|
||||
self._reassign_text_style_ids(new_clip)
|
||||
|
||||
spine.insert(clip_index + len(new_clips), new_clip)
|
||||
new_clips.append(new_clip)
|
||||
@@ -2404,6 +2419,7 @@ class FCPXMLModifier:
|
||||
new_clip.set('start', seg_start.to_fcpxml())
|
||||
new_clip.set('duration', seg_duration.to_fcpxml())
|
||||
self._filter_children_for_segment(new_clip, seg_start, seg_duration)
|
||||
self._reassign_text_style_ids(new_clip)
|
||||
spine.insert(clip_index + len(new_clips), new_clip)
|
||||
new_clips.append(new_clip)
|
||||
current_offset = current_offset + seg_duration
|
||||
@@ -2946,6 +2962,7 @@ class FCPXMLModifier:
|
||||
('-469658744/1000000000s', '0'),
|
||||
('12328542033/1000000000s', '1'),
|
||||
)
|
||||
_TEXT_SIZE_KEY = '9999/10003/13260/3296672360/5/3296672362/3'
|
||||
|
||||
def _ensure_text_title_effect(self, resources: ET.Element) -> str:
|
||||
"""Return the resource id of the "Text" (Basic Text) effect, creating it if absent."""
|
||||
@@ -2995,6 +3012,34 @@ class FCPXMLModifier:
|
||||
self._text_style_ids.add(candidate)
|
||||
return candidate
|
||||
|
||||
def _reassign_text_style_ids(self, clip: ET.Element) -> None:
|
||||
"""Give every ``<text-style-def>`` inside a just-deepcopy'd *clip* a
|
||||
fresh document-unique id, repointing any ``<text-style ref="...">``
|
||||
in the same subtree that pointed at the old one.
|
||||
|
||||
``split_clip``/``cut_clip_ranges`` deepcopy the clip once per
|
||||
resulting segment, so a clip carrying a ``<title>`` (from a "text"
|
||||
voice action) keeps the exact same ``text-style-def id`` in every
|
||||
copy. A single cut is harmless — but the batch chain re-cuts the
|
||||
same clip at each step (silence removal, filler removal, dynamic
|
||||
subtitles), and every pass multiplies the duplicate, so the DTD
|
||||
validator eventually rejects the file with "ID ... already
|
||||
defined". Regenerating here, at the only place copies are made,
|
||||
fixes it for every caller instead of each one having to remember to.
|
||||
"""
|
||||
for style_def in clip.findall('.//text-style-def'):
|
||||
old_id = style_def.get('id')
|
||||
if not old_id:
|
||||
continue
|
||||
slug = old_id[3:] if old_id.startswith('ts_') else old_id
|
||||
slug = re.sub(r'_\d+$', '', slug) # drop a prior _<N> counter
|
||||
new_id = self._unique_text_style_id(slug)
|
||||
if new_id == old_id:
|
||||
continue
|
||||
style_def.set('id', new_id)
|
||||
for ref_el in clip.findall(f".//text-style[@ref='{old_id}']"):
|
||||
ref_el.set('ref', new_id)
|
||||
|
||||
def _make_text_title_clip(
|
||||
self,
|
||||
effect_id: str,
|
||||
@@ -3012,6 +3057,8 @@ class FCPXMLModifier:
|
||||
face: Optional[str] = None,
|
||||
kerning: Optional[float] = None,
|
||||
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
|
||||
animated: bool = True,
|
||||
size_param: Optional[float] = None,
|
||||
) -> ET.Element:
|
||||
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
|
||||
|
||||
@@ -3042,9 +3089,12 @@ class FCPXMLModifier:
|
||||
param.set('key', key)
|
||||
param.set('value', value)
|
||||
|
||||
animation_params = {'Opacity', 'Speed', 'Apply Speed'}
|
||||
for param_name, param_key, param_value in self._TEXT_TITLE_PARAMS:
|
||||
if not animated and param_name in animation_params:
|
||||
continue
|
||||
_add_param(param_name, param_key, param_value)
|
||||
if param_name == 'Speed':
|
||||
if animated and param_name == 'Speed':
|
||||
# "Custom Speed" lands between "Speed" and "Apply Speed" and
|
||||
# carries a <keyframeAnimation> child instead of a value.
|
||||
cs = ET.SubElement(elem, 'param')
|
||||
@@ -3056,6 +3106,9 @@ class FCPXMLModifier:
|
||||
kf.set('time', kf_time)
|
||||
kf.set('value', kf_value)
|
||||
|
||||
if size_param is not None:
|
||||
_add_param('Size', self._TEXT_SIZE_KEY, f"{float(size_param):g}")
|
||||
|
||||
text_el = ET.SubElement(elem, 'text')
|
||||
ts_id = self._unique_text_style_id(name)
|
||||
run = ET.SubElement(text_el, 'text-style')
|
||||
@@ -3109,6 +3162,10 @@ class FCPXMLModifier:
|
||||
font_size: int = 196,
|
||||
font_color: str = '1 1 1 1',
|
||||
bold: bool = True,
|
||||
face: Optional[str] = None,
|
||||
animated: bool = True,
|
||||
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
|
||||
size_param: Optional[float] = None,
|
||||
) -> ET.Element:
|
||||
"""Add a single static "Text" (Basic Text) title over *parent_clip*.
|
||||
|
||||
@@ -3142,6 +3199,10 @@ class FCPXMLModifier:
|
||||
font_size=font_size,
|
||||
font_color=font_color,
|
||||
bold=bold,
|
||||
face=face,
|
||||
animated=animated,
|
||||
font_scale=font_scale,
|
||||
size_param=size_param,
|
||||
)
|
||||
_dtd_insert(parent, title)
|
||||
return title
|
||||
|
||||
@@ -154,6 +154,7 @@ from server_tools.roles import (
|
||||
)
|
||||
from server_tools.subtitles import (
|
||||
handle_generate_dynamic_subtitles,
|
||||
handle_generate_plain_subtitles,
|
||||
handle_validate_subtitle_layout,
|
||||
)
|
||||
from server_tools.timeline import (
|
||||
@@ -320,6 +321,7 @@ __all__ = [
|
||||
"handle_save_voice_analysis_config",
|
||||
"handle_validate_subtitle_layout",
|
||||
"handle_generate_dynamic_subtitles",
|
||||
"handle_generate_plain_subtitles",
|
||||
"handle_push_to_fcp",
|
||||
"handle_list_fcp_libraries",
|
||||
]
|
||||
|
||||
@@ -15,6 +15,7 @@ from typing import Any, Sequence
|
||||
from mcp.types import TextContent
|
||||
|
||||
from fcpxml.media_intel import media_src_to_path
|
||||
from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config
|
||||
from fcpxml.models import (
|
||||
DuplicateGroup,
|
||||
FlashFrame,
|
||||
@@ -25,6 +26,7 @@ from fcpxml.models import (
|
||||
)
|
||||
from fcpxml.parser import FCPXMLParser
|
||||
from fcpxml.rough_cut import RoughCutGenerator
|
||||
from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text
|
||||
from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
|
||||
@@ -639,28 +641,79 @@ def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str:
|
||||
rel_end = action.end - clip_start
|
||||
|
||||
if action.kind == "zoom":
|
||||
config = load_voice_analysis_config()
|
||||
# Only forward an explicit ease — otherwise add_zoom's own default
|
||||
# (a fast ramp in, instant snap back out) is what should apply.
|
||||
zoom_args = {}
|
||||
if action.params.get("ease") is not None:
|
||||
zoom_args["ease"] = float(action.params["ease"])
|
||||
if action.params.get("ease_out") is not None:
|
||||
zoom_args["ease_out"] = float(action.params["ease_out"])
|
||||
zoom_args = {
|
||||
"ease": float(action.params.get("ease", config["zoom_ease_in"])),
|
||||
"ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])),
|
||||
}
|
||||
mode = str(action.params.get("mode", config["zoom_mode"]))
|
||||
if mode == "in":
|
||||
zoom_args["hold_at_end"] = True
|
||||
zoom_args["start_at_peak"] = False
|
||||
elif mode == "out":
|
||||
zoom_args["hold_at_end"] = False
|
||||
zoom_args["start_at_peak"] = True
|
||||
elif mode == "in_out":
|
||||
zoom_args["hold_at_end"] = False
|
||||
zoom_args["start_at_peak"] = False
|
||||
modifier.add_zoom(
|
||||
clip_id=clip_el,
|
||||
start=rel_start,
|
||||
end=rel_end,
|
||||
scale=float(action.params.get("scale", 1.3)),
|
||||
scale=float(action.params.get("scale", config["zoom_scale"])),
|
||||
**zoom_args,
|
||||
)
|
||||
return f"zoom {action.params.get('scale', 1.3):.2f}x"
|
||||
return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x"
|
||||
|
||||
if action.kind == "text":
|
||||
# Default to the "Legendas Dinâmicas" emphasis style (the font used
|
||||
# to highlight a word in the captions) rather than a hardcoded
|
||||
# Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of
|
||||
# the video's on-screen text instead of looking like a stray default
|
||||
# title. Any of these the action itself specifies still wins.
|
||||
subtitle_cfg = load_dynamic_subtitle_config()
|
||||
font = action.params.get("font", subtitle_cfg["emphasis_font"])
|
||||
face = action.params.get("face", subtitle_cfg["emphasis_face"])
|
||||
font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0)
|
||||
requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"]))
|
||||
requested_kerning = float(action.params.get("kerning", 0.0) or 0.0)
|
||||
|
||||
# Voice-action callouts are not part of the dynamic subtitle block.
|
||||
# When omitted, put them above the subtitle band and shrink wide
|
||||
# phrases to the title-safe width. The previous default (Position 0 0,
|
||||
# full emphasis size) made long callouts like "PRÓTESES DE SILICONE"
|
||||
# collide with captions and run off both sides of a vertical frame.
|
||||
emitted_size = requested_size * font_scale
|
||||
emitted_kerning = requested_kerning * font_scale
|
||||
safe_width = modifier.frame_width() * 0.90
|
||||
width = measure_text(
|
||||
action.params["content"],
|
||||
emitted_size,
|
||||
bold=bool(action.params.get("bold", False)),
|
||||
kerning=emitted_kerning,
|
||||
font=font,
|
||||
face=face,
|
||||
)
|
||||
font_size = requested_size
|
||||
if width > safe_width and width > 0:
|
||||
font_size = max(32, int(requested_size * safe_width / width))
|
||||
position = action.params.get("position")
|
||||
if not position:
|
||||
position = f"0 {modifier.frame_height() * 0.23:g}"
|
||||
|
||||
modifier.add_text_title(
|
||||
clip_el,
|
||||
action.params["content"],
|
||||
offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
|
||||
duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(),
|
||||
position=position,
|
||||
font=font,
|
||||
font_size=font_size,
|
||||
font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]),
|
||||
face=face,
|
||||
bold=action.params.get("bold", False),
|
||||
)
|
||||
return f"text \"{action.params['content'][:24]}\""
|
||||
|
||||
|
||||
+176
-10
@@ -6,13 +6,14 @@ Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalo
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Sequence
|
||||
|
||||
from mcp.types import TextContent, Tool
|
||||
|
||||
from fcpxml.media_intel import media_src_to_path
|
||||
from fcpxml.model_manager import load_dynamic_subtitle_config
|
||||
from fcpxml.model_manager import load_dynamic_subtitle_config, load_plain_subtitle_config
|
||||
from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
from server_tools._shared import (
|
||||
@@ -69,9 +70,76 @@ TOOLS = [
|
||||
"required": ["filepath"]
|
||||
}
|
||||
),
|
||||
Tool(
|
||||
name="generate_plain_subtitles",
|
||||
description="Generate simple editable FCPXML text-title subtitles, synchronized to transcript words but without visual build-in/build-out effects. Words are grouped into short blocks, placed at a configurable vertical position, and written as static Text titles rather than SRT captions.",
|
||||
inputSchema={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
||||
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
|
||||
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
||||
"language": {"type": "string", "description": "ISO language code hint (e.g. 'pt'); auto-detected if omitted"},
|
||||
"font": {"type": "string", "description": "Text font family. Falls back to saved plain-subtitle config."},
|
||||
"font_size": {"type": "integer", "description": "Font size in canvas points. Falls back to saved plain-subtitle config."},
|
||||
"font_color": {"type": "string", "description": "RGBA (0-1, space-separated). Falls back to saved plain-subtitle config."},
|
||||
"max_words": {"type": "integer", "description": "Maximum words per subtitle block. Falls back to saved plain-subtitle config."},
|
||||
"position_y": {"type": "number", "description": "Vertical title position in canvas points; negative sits lower in frame."},
|
||||
"uppercase": {"type": "boolean", "description": "Render text in uppercase."},
|
||||
"keep_punctuation": {"type": "boolean", "description": "Keep punctuation such as comma and period."},
|
||||
"text_scale": {"type": "number", "description": "Template font-size scale. Falls back to saved plain-subtitle config."},
|
||||
"output_path": {"type": "string", "description": "Output path (default: adds _plain_subtitles suffix)"},
|
||||
},
|
||||
"required": ["filepath"]
|
||||
}
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
_PUNCT_RE = re.compile(r"[^\w\sÀ-ÖØ-öø-ÿ]", re.UNICODE)
|
||||
|
||||
|
||||
def _words_overlapping_clip(words: Sequence[dict], start: float, end: float) -> list[dict]:
|
||||
"""Return transcript words that overlap a source window, rebased to it."""
|
||||
clip_words: list[dict] = []
|
||||
for w in words:
|
||||
word_start = float(w.get("start", 0.0))
|
||||
word_end = float(w.get("end", word_start))
|
||||
if word_end <= start or word_start >= end:
|
||||
continue
|
||||
clip_words.append(
|
||||
{
|
||||
"word": w.get("word", ""),
|
||||
"start": max(0.0, word_start - start),
|
||||
"end": max(0.0, min(word_end, end) - start),
|
||||
}
|
||||
)
|
||||
return clip_words
|
||||
|
||||
|
||||
def _plain_word_text(word: str, *, uppercase: bool, keep_punctuation: bool) -> str:
|
||||
text = str(word or "").strip()
|
||||
if not keep_punctuation:
|
||||
text = _PUNCT_RE.sub("", text)
|
||||
text = re.sub(r"\s+", " ", text).strip()
|
||||
return text.upper() if uppercase else text
|
||||
|
||||
|
||||
def _plain_subtitle_blocks(words: Sequence[dict], max_words: int) -> list[list[dict]]:
|
||||
blocks: list[list[dict]] = []
|
||||
pending: list[dict] = []
|
||||
for word in words:
|
||||
if not str(word.get("word", "")).strip():
|
||||
continue
|
||||
pending.append(word)
|
||||
if len(pending) >= max(1, max_words):
|
||||
blocks.append(pending)
|
||||
pending = []
|
||||
if pending:
|
||||
blocks.append(pending)
|
||||
return blocks
|
||||
|
||||
|
||||
async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Validate title/subtitle layout for spatial collisions and safe-area
|
||||
containment (collision.validate_titles over every <title> in the file)."""
|
||||
@@ -210,15 +278,7 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
|
||||
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
||||
window_end = clip_source_start + clip_duration
|
||||
|
||||
clip_words = [
|
||||
{
|
||||
"word": w.get("word", ""),
|
||||
"start": float(w.get("start", 0.0)) - clip_source_start,
|
||||
"end": float(w.get("end", 0.0)) - clip_source_start,
|
||||
}
|
||||
for w in data.get("words", [])
|
||||
if clip_source_start <= float(w.get("start", 0.0)) < window_end
|
||||
]
|
||||
clip_words = _words_overlapping_clip(data.get("words", []), clip_source_start, window_end)
|
||||
if not clip_words:
|
||||
skipped.append((name, "no words in clip's source range"))
|
||||
continue
|
||||
@@ -277,7 +337,113 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Generate static, editable title subtitles from word-level transcripts."""
|
||||
model = arguments.get("model", "base")
|
||||
language = arguments.get("language")
|
||||
output_dir = arguments.get("output_dir")
|
||||
clip_filter = arguments.get("clip_name")
|
||||
|
||||
saved = load_plain_subtitle_config()
|
||||
font = arguments.get("font") or saved["font"]
|
||||
font_size = int(arguments.get("font_size", saved["font_size"]))
|
||||
font_color = arguments.get("font_color") or saved["font_color"]
|
||||
max_words = max(1, int(arguments.get("max_words", saved["max_words"])))
|
||||
position_y = float(arguments.get("position_y", saved["position_y"]))
|
||||
uppercase = bool(arguments.get("uppercase", saved["uppercase"]))
|
||||
keep_punctuation = bool(arguments.get("keep_punctuation", saved["keep_punctuation"]))
|
||||
|
||||
filepath, output_path, modifier = _setup_modifier(arguments, "_plain_subtitles")
|
||||
|
||||
added: list[tuple[str, int, int]] = []
|
||||
skipped: list[tuple[str, str]] = []
|
||||
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
||||
for el in spine_clips:
|
||||
name = el.get("name", "")
|
||||
if clip_filter and name != clip_filter:
|
||||
continue
|
||||
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
||||
media_path = media_src_to_path(src)
|
||||
if not media_path or not Path(media_path).is_file():
|
||||
skipped.append((name, "media file missing"))
|
||||
continue
|
||||
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
||||
if data is None:
|
||||
skipped.append((name, reason))
|
||||
continue
|
||||
|
||||
clip_source_start = modifier.source_file_start(el).to_seconds()
|
||||
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
||||
clip_words = _words_overlapping_clip(
|
||||
data.get("words", []), clip_source_start, clip_source_start + clip_duration
|
||||
)
|
||||
if not clip_words:
|
||||
skipped.append((name, "no words in clip's source range"))
|
||||
continue
|
||||
|
||||
blocks = _plain_subtitle_blocks(clip_words, max_words)
|
||||
created = 0
|
||||
for block in blocks:
|
||||
parts = [
|
||||
_plain_word_text(w.get("word", ""), uppercase=uppercase, keep_punctuation=keep_punctuation)
|
||||
for w in block
|
||||
]
|
||||
text = " ".join(p for p in parts if p).strip()
|
||||
if not text:
|
||||
continue
|
||||
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
|
||||
end = max(float(w.get("end", start)) for w in block)
|
||||
duration = max(end - start, modifier.frame_duration_fraction())
|
||||
modifier.add_text_title(
|
||||
el,
|
||||
text,
|
||||
offset=f"{start:.6f}s",
|
||||
duration=f"{duration:.6f}s",
|
||||
lane=20,
|
||||
position=f"0 {position_y:g}",
|
||||
font=font,
|
||||
font_size=font_size,
|
||||
font_color=font_color,
|
||||
bold=True,
|
||||
face=None,
|
||||
font_scale=1.0,
|
||||
size_param=font_size,
|
||||
)
|
||||
created += 1
|
||||
if created:
|
||||
added.append((name, created, len(clip_words)))
|
||||
|
||||
if not added:
|
||||
text = "# Plain Subtitles\n\nNo subtitles generated — file unchanged (nothing saved)."
|
||||
if skipped:
|
||||
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
||||
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
||||
)
|
||||
return _text_result(text)
|
||||
|
||||
modifier.save(output_path)
|
||||
total_titles = sum(lines for _, lines, _ in added)
|
||||
total_words = sum(words for _, _, words in added)
|
||||
result = "# Plain Subtitles Generated\n\n## Summary\n"
|
||||
result += (
|
||||
f"- **Clips Captioned**: {len(added)}\n"
|
||||
f"- **Title Clips**: {total_titles}\n"
|
||||
f"- **Total Words**: {total_words}\n\n"
|
||||
)
|
||||
result += _markdown_table(
|
||||
["Clip", "Title Clips", "Words"],
|
||||
[[n, str(lines), str(words)] for n, lines, words in added],
|
||||
)
|
||||
if skipped:
|
||||
result += "\n## Skipped Clips\n" + _markdown_table(
|
||||
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
||||
)
|
||||
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
HANDLERS = {
|
||||
"validate_subtitle_layout": handle_validate_subtitle_layout,
|
||||
"generate_dynamic_subtitles": handle_generate_dynamic_subtitles,
|
||||
"generate_plain_subtitles": handle_generate_plain_subtitles,
|
||||
}
|
||||
|
||||
@@ -67,12 +67,12 @@ TOOLS = [
|
||||
),
|
||||
Tool(
|
||||
name="remove_filler_words",
|
||||
description="Cut filler words (um, uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.",
|
||||
description="Cut filler interjections (uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'um', 'uma', 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.",
|
||||
inputSchema={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
||||
"fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: um, uh, uhh, umm, erm, ehm, mmm, hmm, mhm)"},
|
||||
"fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: uh, uhh, umm, erm, ehm, mmm, hmm, mhm; pass um/uma explicitly if desired)"},
|
||||
"clip_name": {"type": "string", "description": "Only clean the clip with this name"},
|
||||
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
||||
"padding": {"type": "number", "default": 0.02, "description": "Seconds to widen each cut on both sides (0-2, default 0.02)"},
|
||||
|
||||
@@ -348,8 +348,9 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
|
||||
language = arguments.get("language")
|
||||
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
|
||||
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
|
||||
output_dir = arguments.get("output_dir")
|
||||
|
||||
transcript, reason = _load_or_transcribe(media_path, model, language)
|
||||
transcript, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
||||
if transcript is None:
|
||||
return _text_result(
|
||||
f"# Voice Timeline\n\nCould not obtain a transcript "
|
||||
@@ -365,9 +366,10 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
|
||||
weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
|
||||
peak_percentile=config["peak_percentile"],
|
||||
emphasis_floor=config["emphasis_floor"],
|
||||
emotion_enabled=config["emotion_enabled"],
|
||||
emotion_sensitivity=config["emotion_sensitivity"],
|
||||
)
|
||||
|
||||
output_dir = arguments.get("output_dir")
|
||||
json_path = Path(_validate_output_path(
|
||||
str(voice_timeline_path(media_path, output_dir)),
|
||||
anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent),
|
||||
@@ -398,6 +400,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
|
||||
"yes" if layers["acoustics"] else "FAILED — every acoustic value is 0",
|
||||
],
|
||||
["Speakers", "yes" if layers["speakers"] else "not run — single default speaker"],
|
||||
["Emotion", "yes" if layers.get("emotion") else "not run"],
|
||||
],
|
||||
) + "\n"
|
||||
|
||||
|
||||
@@ -29,6 +29,7 @@ from fcpxml.text_layout import (
|
||||
ink_extent,
|
||||
)
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
from server_tools.subtitles import _words_overlapping_clip
|
||||
|
||||
SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml"
|
||||
def font_points(style) -> float:
|
||||
@@ -51,6 +52,21 @@ WORDS = [
|
||||
]
|
||||
|
||||
|
||||
def test_words_overlapping_clip_keeps_word_that_starts_just_before_in_point():
|
||||
words = [
|
||||
{"word": "Aquela", "start": 2.03, "end": 2.69},
|
||||
{"word": "mama", "start": 2.69, "end": 2.89},
|
||||
{"word": "fora", "start": 10.0, "end": 10.2},
|
||||
]
|
||||
|
||||
clip_words = _words_overlapping_clip(words, 2.0437166666666666, 3.0)
|
||||
|
||||
assert clip_words == [
|
||||
{"word": "Aquela", "start": 0.0, "end": pytest.approx(0.6462833333333332)},
|
||||
{"word": "mama", "start": pytest.approx(0.6462833333333332), "end": pytest.approx(0.8462833333333334)},
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def temp_fcpxml():
|
||||
with tempfile.NamedTemporaryFile(suffix=".fcpxml", delete=False) as f:
|
||||
|
||||
@@ -0,0 +1,479 @@
|
||||
"""Tests for the phrase review model (voice timeline + AI actions → editable script)."""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from fcpxml.phrase_review import (
|
||||
TRACK_BACKSTAGE,
|
||||
TRACK_SCRIPT,
|
||||
ZOOM_SCALE_BY_LEVEL,
|
||||
build_phrase_review,
|
||||
load_phrase_review,
|
||||
merge_saved_decisions,
|
||||
phrase_review_to_actions,
|
||||
resolve_source,
|
||||
review_paths,
|
||||
save_phrase_review,
|
||||
snap_to_words,
|
||||
)
|
||||
|
||||
|
||||
def _words(spans, emphasis=0.0):
|
||||
return [
|
||||
{
|
||||
"text": f"w{i}",
|
||||
"start": start,
|
||||
"end": end,
|
||||
"energy": 0.5,
|
||||
"emphasis": emphasis,
|
||||
}
|
||||
for i, (start, end) in enumerate(spans)
|
||||
]
|
||||
|
||||
|
||||
def _timeline(segments):
|
||||
return {"source": "/tmp/take.mov", "speakers": ["SPEAKER_00"], "segments": segments}
|
||||
|
||||
|
||||
def _segment(start, end, text="linha", peak=0.1, take_boundary=False, words=None):
|
||||
return {
|
||||
"start": start,
|
||||
"end": end,
|
||||
"text": text,
|
||||
"speaker": "SPEAKER_00",
|
||||
"peak_emphasis": peak,
|
||||
"take_boundary": take_boundary,
|
||||
"gap_before": 0.0,
|
||||
"words": words if words is not None else _words([(start, end)]),
|
||||
}
|
||||
|
||||
|
||||
class TestBuildFromAcoustics:
|
||||
def test_emphasis_levels_follow_peak_thresholds(self):
|
||||
review = build_phrase_review(
|
||||
_timeline(
|
||||
[
|
||||
_segment(0, 1, peak=0.10),
|
||||
_segment(1, 2, peak=0.30),
|
||||
_segment(2, 3, peak=0.50),
|
||||
_segment(3, 4, peak=0.90),
|
||||
]
|
||||
)
|
||||
)
|
||||
assert [p["emphasis"] for p in review["phrases"]] == [0, 1, 2, 3]
|
||||
|
||||
def test_every_phrase_starts_active_without_actions(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 1), _segment(1, 2)]))
|
||||
assert all(p["active"] for p in review["phrases"])
|
||||
assert all(p["track"] == TRACK_SCRIPT for p in review["phrases"])
|
||||
|
||||
def test_carries_text_speaker_and_words(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2, text=" olá ")]))
|
||||
phrase = review["phrases"][0]
|
||||
assert phrase["text"] == "olá"
|
||||
assert phrase["speaker"] == "SPEAKER_00"
|
||||
assert phrase["words"][0]["text"] == "w0"
|
||||
assert review["duration"] == 2.0
|
||||
|
||||
|
||||
class TestCutsDeactivate:
|
||||
def test_fully_cut_phrase_is_inactive(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 2), _segment(2, 4)]),
|
||||
{"actions": [{"kind": "cut", "start": 0, "end": 2, "reason": "gaguejou"}]},
|
||||
)
|
||||
assert review["phrases"][0]["active"] is False
|
||||
assert review["phrases"][0]["reason"] == "gaguejou"
|
||||
assert review["phrases"][1]["active"] is True
|
||||
|
||||
def test_small_overlap_keeps_the_phrase(self):
|
||||
# 0.2s off a 2s line is a trim, not a removal.
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(1, 3, words=_words([(1, 1.2), (1.2, 3)]))]),
|
||||
{"actions": [{"kind": "cut", "start": 0.5, "end": 1.2}]},
|
||||
)
|
||||
assert review["phrases"][0]["active"] is True
|
||||
|
||||
def test_majority_overlap_deactivates(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 2)]),
|
||||
{"actions": [{"kind": "cut", "start": 0, "end": 1.5}]},
|
||||
)
|
||||
assert review["phrases"][0]["active"] is False
|
||||
|
||||
def test_inactive_after_take_boundary_is_backstage(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(10, 12, take_boundary=True)]),
|
||||
{"actions": [{"kind": "cut", "start": 10, "end": 12}]},
|
||||
)
|
||||
assert review["phrases"][0]["track"] == TRACK_BACKSTAGE
|
||||
|
||||
|
||||
class TestTrimFromPartialCuts:
|
||||
def test_head_cut_becomes_a_trim_snapped_to_a_word(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(1, 4, words=_words([(1, 1.4), (1.4, 4)]))]),
|
||||
{"actions": [{"kind": "cut", "start": 0.8, "end": 1.35}]},
|
||||
)
|
||||
phrase = review["phrases"][0]
|
||||
assert phrase["active"] is True
|
||||
assert phrase["trim_start"] == 1.4 # snapped to the second word's start
|
||||
assert phrase["trim_end"] == 4.0
|
||||
|
||||
def test_tail_cut_becomes_a_trim(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 3, words=_words([(0, 2.5), (2.5, 3)]))]),
|
||||
{"actions": [{"kind": "cut", "start": 2.6, "end": 3.5}]},
|
||||
)
|
||||
phrase = review["phrases"][0]
|
||||
assert phrase["trim_start"] == 0.0
|
||||
assert phrase["trim_end"] == 2.5
|
||||
|
||||
def test_untouched_phrase_trims_to_its_own_bounds(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
phrase = review["phrases"][0]
|
||||
assert (phrase["trim_start"], phrase["trim_end"]) == (0.0, 2.0)
|
||||
|
||||
|
||||
class TestAIDirectionWins:
|
||||
def test_zoom_action_sets_the_level_over_the_heuristic(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 2, peak=0.05)]),
|
||||
{
|
||||
"actions": [
|
||||
{
|
||||
"kind": "zoom",
|
||||
"start": 0.5,
|
||||
"end": 0.9,
|
||||
"params": {"scale": 1.5},
|
||||
"reason": "virada da história",
|
||||
}
|
||||
]
|
||||
},
|
||||
)
|
||||
phrase = review["phrases"][0]
|
||||
assert phrase["emphasis"] == 3
|
||||
assert phrase["reason"] == "virada da história"
|
||||
|
||||
def test_text_action_marks_emphasis(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 2, peak=0.0)]),
|
||||
{
|
||||
"actions": [
|
||||
{
|
||||
"kind": "text",
|
||||
"start": 0.5,
|
||||
"end": 1.0,
|
||||
"params": {"content": "3x mais rápido"},
|
||||
}
|
||||
]
|
||||
},
|
||||
)
|
||||
assert review["phrases"][0]["emphasis"] == 2
|
||||
|
||||
def test_highest_level_wins_when_several_actions_overlap(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 4)]),
|
||||
{
|
||||
"actions": [
|
||||
{"kind": "zoom", "start": 0.2, "end": 0.5, "params": {"scale": 1.15}},
|
||||
{"kind": "zoom", "start": 2.0, "end": 2.4, "params": {"scale": 1.5}},
|
||||
]
|
||||
},
|
||||
)
|
||||
assert review["phrases"][0]["emphasis"] == 3
|
||||
|
||||
def test_malformed_rows_are_reported_not_fatal(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 2)]),
|
||||
{"actions": [{"kind": "voar", "start": 0, "end": 1}]},
|
||||
)
|
||||
assert len(review["errors"]) == 1
|
||||
assert review["phrases"][0]["active"] is True
|
||||
|
||||
|
||||
class TestBackToActions:
|
||||
def test_inactive_phrase_becomes_a_cut(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2), _segment(2, 4)]))
|
||||
review["phrases"][0]["active"] = False
|
||||
result = phrase_review_to_actions(review)
|
||||
cuts = [a for a in result["actions"] if a["kind"] == "cut"]
|
||||
assert len(cuts) == 1
|
||||
assert (cuts[0]["start"], cuts[0]["end"]) == (0.0, 2.0)
|
||||
|
||||
def test_emphasis_becomes_a_zoom_and_a_span(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
review["phrases"][0]["emphasis"] = 2
|
||||
result = phrase_review_to_actions(review)
|
||||
zooms = [a for a in result["actions"] if a["kind"] == "zoom"]
|
||||
assert zooms[0]["params"]["scale"] == ZOOM_SCALE_BY_LEVEL[2]
|
||||
assert result["emphasis_spans"] == [
|
||||
{"start": 0.0, "end": 2.0, "level": 2, "text": "linha"}
|
||||
]
|
||||
|
||||
def test_level_zero_produces_nothing(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
review["phrases"][0]["emphasis"] = 0
|
||||
result = phrase_review_to_actions(review)
|
||||
assert result["actions"] == []
|
||||
assert result["emphasis_spans"] == []
|
||||
|
||||
def test_trim_becomes_head_and_tail_cuts(self):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 4, words=_words([(0, 1), (1, 3), (3, 4)]))])
|
||||
)
|
||||
review["phrases"][0]["trim_start"] = 1.0
|
||||
review["phrases"][0]["trim_end"] = 3.0
|
||||
result = phrase_review_to_actions(review)
|
||||
spans = [(a["start"], a["end"]) for a in result["actions"] if a["kind"] == "cut"]
|
||||
assert spans == [(0.0, 1.0), (3.0, 4.0)]
|
||||
|
||||
def test_inactive_phrase_is_cut_whole_ignoring_its_trim(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 4)]))
|
||||
review["phrases"][0].update({"active": False, "trim_start": 1.0, "trim_end": 3.0})
|
||||
result = phrase_review_to_actions(review)
|
||||
assert [(a["start"], a["end"]) for a in result["actions"]] == [(0.0, 4.0)]
|
||||
|
||||
def test_zoom_follows_the_trimmed_span(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 4)]))
|
||||
review["phrases"][0].update({"emphasis": 1, "trim_start": 1.0, "trim_end": 3.0})
|
||||
result = phrase_review_to_actions(review)
|
||||
zoom = next(a for a in result["actions"] if a["kind"] == "zoom")
|
||||
assert (zoom["start"], zoom["end"]) == (1.0, 3.0)
|
||||
|
||||
def test_impossible_trim_is_ignored(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 4)]))
|
||||
review["phrases"][0].update({"trim_start": 3.0, "trim_end": 1.0})
|
||||
result = phrase_review_to_actions(review)
|
||||
assert result["actions"] == []
|
||||
|
||||
def test_emphasis_out_of_range_is_clamped(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
review["phrases"][0]["emphasis"] = 99
|
||||
result = phrase_review_to_actions(review)
|
||||
assert result["actions"][0]["params"]["scale"] == ZOOM_SCALE_BY_LEVEL[3]
|
||||
|
||||
def test_rows_that_make_no_sense_are_skipped(self):
|
||||
result = phrase_review_to_actions(
|
||||
{"phrases": ["nope", {"start": 5, "end": 1}, {"start": 0, "end": 1}]}
|
||||
)
|
||||
assert result["actions"] == []
|
||||
|
||||
|
||||
class TestManualZooms:
|
||||
def test_manual_zoom_becomes_an_action_without_a_scale(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 10)]))
|
||||
review["zooms"] = [{"start": 2.0, "end": 4.0}]
|
||||
result = phrase_review_to_actions(review)
|
||||
zoom = next(a for a in result["actions"] if a["kind"] == "zoom")
|
||||
assert (zoom["start"], zoom["end"]) == (2.0, 4.0)
|
||||
# Sem scale: o aplicador usa o zoom_scale configurado pelo usuário.
|
||||
assert "scale" not in zoom["params"]
|
||||
|
||||
def test_zoom_shorter_than_the_ramp_is_refused(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 10)]))
|
||||
review["zooms"] = [{"start": 2.0, "end": 2.1}]
|
||||
assert phrase_review_to_actions(review)["actions"] == []
|
||||
|
||||
def test_manual_zoom_coexists_with_phrase_emphasis(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 10)]))
|
||||
review["phrases"][0]["emphasis"] = 2
|
||||
review["zooms"] = [{"start": 2.0, "end": 4.0}]
|
||||
zooms = [a for a in phrase_review_to_actions(review)["actions"] if a["kind"] == "zoom"]
|
||||
assert len(zooms) == 2
|
||||
|
||||
def test_malformed_zoom_rows_are_skipped(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 10)]))
|
||||
review["zooms"] = ["nope", {"start": 5}, {"start": 4, "end": 1}]
|
||||
assert phrase_review_to_actions(review)["actions"] == []
|
||||
|
||||
def test_saved_zooms_are_restored(self):
|
||||
review = merge_saved_decisions(
|
||||
build_phrase_review(_timeline([_segment(0, 10)])),
|
||||
{"phrases": [], "zooms": [{"start": 1.0, "end": 3.0}]},
|
||||
)
|
||||
assert review["zooms"] == [{"start": 1.0, "end": 3.0}]
|
||||
|
||||
def test_new_review_starts_with_no_manual_zooms(self):
|
||||
assert build_phrase_review(_timeline([_segment(0, 2)]))["zooms"] == []
|
||||
|
||||
|
||||
class TestRoundTrip:
|
||||
def test_review_survives_actions_and_back(self):
|
||||
timeline = _timeline(
|
||||
[_segment(0, 2, peak=0.9), _segment(2, 4), _segment(4, 6, peak=0.5)]
|
||||
)
|
||||
first = build_phrase_review(timeline)
|
||||
first["phrases"][1]["active"] = False
|
||||
actions = phrase_review_to_actions(first)
|
||||
|
||||
second = build_phrase_review(timeline, actions)
|
||||
assert [p["active"] for p in second["phrases"]] == [True, False, True]
|
||||
assert [p["emphasis"] for p in second["phrases"]] == [3, 0, 2]
|
||||
|
||||
|
||||
class TestSnapToWords:
|
||||
def test_snaps_to_the_nearest_start(self):
|
||||
words = _words([(1.0, 1.5), (1.5, 2.0)])
|
||||
assert snap_to_words(1.6, words, 1.6, "in") == 1.5
|
||||
|
||||
def test_snaps_to_the_nearest_end(self):
|
||||
words = _words([(1.0, 1.5), (1.5, 2.0)])
|
||||
assert snap_to_words(1.9, words, 1.9, "out") == 2.0
|
||||
|
||||
def test_falls_back_without_word_timings(self):
|
||||
assert snap_to_words(1.2, [], 3.4, "in") == 3.4
|
||||
|
||||
|
||||
class TestResolveSource:
|
||||
def test_finds_the_media_beside_its_timeline(self, tmp_path):
|
||||
media = tmp_path / "take.mov"
|
||||
media.write_bytes(b"0")
|
||||
timeline = tmp_path / "take_voice_timeline.json"
|
||||
assert resolve_source("take.mov", str(timeline)) == str(media)
|
||||
|
||||
def test_falls_back_to_the_project_folder(self, tmp_path):
|
||||
media_dir = tmp_path / "midia"
|
||||
media_dir.mkdir()
|
||||
media = media_dir / "take.mov"
|
||||
media.write_bytes(b"0")
|
||||
timeline = tmp_path / "json" / "take_voice_timeline.json"
|
||||
assert resolve_source("take.mov", str(timeline), [str(media_dir)]) == str(media)
|
||||
|
||||
def test_absolute_path_is_used_as_is(self, tmp_path):
|
||||
media = tmp_path / "take.mov"
|
||||
media.write_bytes(b"0")
|
||||
assert resolve_source(str(media), "") == str(media)
|
||||
|
||||
def test_missing_media_resolves_to_empty(self, tmp_path):
|
||||
assert resolve_source("take.mov", str(tmp_path / "x_voice_timeline.json")) == ""
|
||||
|
||||
def test_stale_absolute_path_still_finds_the_file_by_name(self, tmp_path):
|
||||
# The fixture's source is an absolute path that no longer exists (the
|
||||
# everyday case: the project moved). Falling back to the file name next
|
||||
# to the timeline is what keeps the preview working after a move.
|
||||
media = tmp_path / "take.mov"
|
||||
media.write_bytes(b"0")
|
||||
assert resolve_source("/tmp/gone/take.mov", str(tmp_path / "t.json")) == str(media)
|
||||
|
||||
def test_review_carries_the_resolved_path(self, tmp_path):
|
||||
media = tmp_path / "take.mov"
|
||||
media.write_bytes(b"0")
|
||||
review = build_phrase_review(
|
||||
{**_timeline([_segment(0, 1)]), "source": "take.mov"},
|
||||
voice_timeline_path=str(tmp_path / "take_voice_timeline.json"),
|
||||
)
|
||||
assert review["source_path"] == str(media)
|
||||
|
||||
def test_review_without_media_reports_no_path(self, tmp_path):
|
||||
review = build_phrase_review(
|
||||
_timeline([_segment(0, 1)]),
|
||||
voice_timeline_path=str(tmp_path / "take_voice_timeline.json"),
|
||||
)
|
||||
assert review["source_path"] == ""
|
||||
|
||||
|
||||
class TestEmotion:
|
||||
def test_segment_emotion_reaches_the_phrase(self):
|
||||
segment = _segment(0, 2)
|
||||
segment["emotion"] = "excited"
|
||||
segment["emotion_confidence"] = 0.72
|
||||
review = build_phrase_review(_timeline([segment]))
|
||||
assert review["phrases"][0]["emotion"] == "excited"
|
||||
assert review["phrases"][0]["emotion_confidence"] == 0.72
|
||||
|
||||
def test_defaults_to_neutral_when_absent(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
assert review["phrases"][0]["emotion"] == "neutral"
|
||||
assert review["phrases"][0]["emotion_confidence"] == 0.0
|
||||
|
||||
def test_availability_comes_from_the_analysis_layers(self):
|
||||
assert build_phrase_review(_timeline([_segment(0, 1)]))["emotion_available"] is False
|
||||
timeline = {**_timeline([_segment(0, 1)]), "layers": {"emotion": True}}
|
||||
assert build_phrase_review(timeline)["emotion_available"] is True
|
||||
|
||||
|
||||
class TestMergeSavedDecisions:
|
||||
def test_saved_decisions_win_over_the_derivation(self):
|
||||
timeline = _timeline([_segment(0, 2, peak=0.9), _segment(2, 4)])
|
||||
saved = {
|
||||
"phrases": [
|
||||
{"index": 0, "start": 0.0, "emphasis": 0, "active": False,
|
||||
"track": TRACK_BACKSTAGE, "text": "corrigido"},
|
||||
]
|
||||
}
|
||||
review = merge_saved_decisions(build_phrase_review(timeline), saved)
|
||||
first = review["phrases"][0]
|
||||
assert (first["emphasis"], first["active"]) == (0, False)
|
||||
assert first["track"] == TRACK_BACKSTAGE
|
||||
assert first["text"] == "corrigido"
|
||||
assert review["phrases"][1]["active"] is True
|
||||
|
||||
def test_fresh_analysis_fields_are_not_overwritten(self):
|
||||
segment = _segment(0, 2, peak=0.9)
|
||||
segment["emotion"] = "tense"
|
||||
review = merge_saved_decisions(
|
||||
build_phrase_review(_timeline([segment])),
|
||||
{"phrases": [{"index": 0, "start": 0.0, "emphasis": 1}]},
|
||||
)
|
||||
assert review["phrases"][0]["emotion"] == "tense"
|
||||
assert review["phrases"][0]["peak_emphasis"] == 0.9
|
||||
|
||||
def test_decision_is_dropped_when_the_line_moved(self):
|
||||
review = merge_saved_decisions(
|
||||
build_phrase_review(_timeline([_segment(10, 12, peak=0.9)])),
|
||||
{"phrases": [{"index": 0, "start": 0.0, "active": False}]},
|
||||
)
|
||||
assert review["phrases"][0]["active"] is True
|
||||
|
||||
def test_saved_trim_is_restored(self):
|
||||
review = merge_saved_decisions(
|
||||
build_phrase_review(_timeline([_segment(0, 4)])),
|
||||
{"phrases": [{"index": 0, "start": 0.0, "trim_start": 1.0, "trim_end": 3.0}]},
|
||||
)
|
||||
assert (review["phrases"][0]["trim_start"], review["phrases"][0]["trim_end"]) == (1.0, 3.0)
|
||||
|
||||
def test_impossible_saved_trim_is_ignored(self):
|
||||
review = merge_saved_decisions(
|
||||
build_phrase_review(_timeline([_segment(0, 4)])),
|
||||
{"phrases": [{"index": 0, "start": 0.0, "trim_start": 9.0, "trim_end": 12.0}]},
|
||||
)
|
||||
assert (review["phrases"][0]["trim_start"], review["phrases"][0]["trim_end"]) == (0.0, 4.0)
|
||||
|
||||
def test_no_saved_review_is_a_no_op(self):
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
assert merge_saved_decisions(review, None) is review
|
||||
|
||||
|
||||
class TestPersistence:
|
||||
def test_paths_are_named_after_the_timeline(self, tmp_path):
|
||||
timeline_path = tmp_path / "take_voice_timeline.json"
|
||||
review_path, actions_path = review_paths(str(timeline_path))
|
||||
assert review_path.name == "take_phrase_review.json"
|
||||
assert actions_path.name == "take_phrase_actions.json"
|
||||
|
||||
def test_save_writes_both_files_and_load_reads_it_back(self, tmp_path):
|
||||
timeline_path = tmp_path / "take_voice_timeline.json"
|
||||
review = build_phrase_review(_timeline([_segment(0, 2)]))
|
||||
review["phrases"][0]["emphasis"] = 3
|
||||
|
||||
review_path, actions_path = save_phrase_review(str(timeline_path), review)
|
||||
assert review_path.is_file() and actions_path.is_file()
|
||||
|
||||
written = json.loads(actions_path.read_text(encoding="utf-8"))
|
||||
assert written["actions"][0]["kind"] == "zoom"
|
||||
|
||||
assert load_phrase_review(str(timeline_path))["phrases"][0]["emphasis"] == 3
|
||||
|
||||
def test_load_returns_none_when_absent_or_broken(self, tmp_path):
|
||||
timeline_path = tmp_path / "take_voice_timeline.json"
|
||||
assert load_phrase_review(str(timeline_path)) is None
|
||||
|
||||
review_path, _ = review_paths(str(timeline_path))
|
||||
review_path.write_text("{ not json", encoding="utf-8")
|
||||
assert load_phrase_review(str(timeline_path)) is None
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
pytest.main([__file__, "-v"])
|
||||
@@ -40,7 +40,7 @@ WORDS = [
|
||||
w("the", 2.5, 2.6),
|
||||
w("show", 2.65, 3.0),
|
||||
w("is", 3.05, 3.15),
|
||||
w("um", 3.2, 3.5),
|
||||
w("uh", 3.2, 3.5),
|
||||
w("great.", 3.6, 4.0),
|
||||
]
|
||||
|
||||
@@ -60,7 +60,7 @@ class TestNormalizeWord:
|
||||
class TestFindPhraseSpans:
|
||||
def test_single_word_multiple_hits(self):
|
||||
spans = find_phrase_spans(WORDS, "um")
|
||||
assert spans == [(0.4, 0.6), (3.2, 3.5)]
|
||||
assert spans == [(0.4, 0.6)]
|
||||
|
||||
def test_multi_word_phrase(self):
|
||||
spans = find_phrase_spans(WORDS, "welcome to the show")
|
||||
@@ -84,9 +84,13 @@ class TestFindPhraseSpans:
|
||||
class TestFindFillerSpans:
|
||||
def test_default_fillers(self):
|
||||
spans = find_filler_spans(WORDS)
|
||||
assert (0.4, 0.6) in spans
|
||||
assert (0.4, 0.6) not in spans
|
||||
assert (3.2, 3.5) in spans
|
||||
|
||||
def test_um_can_still_be_explicit(self):
|
||||
spans = find_filler_spans(WORDS, fillers=("um",))
|
||||
assert spans == [(0.4, 0.6)]
|
||||
|
||||
def test_multi_word_filler(self):
|
||||
spans = find_filler_spans(WORDS, fillers=("you know",))
|
||||
assert spans == [(2.0, 2.4)]
|
||||
@@ -96,7 +100,10 @@ class TestFindFillerSpans:
|
||||
assert spans == sorted(spans)
|
||||
|
||||
def test_defaults_are_conservative(self):
|
||||
# "like" and "so" are speech, not noise — must not be default-cut.
|
||||
# "um", "uma", "like" and "so" are speech, not noise — must not be
|
||||
# default-cut.
|
||||
assert "um" not in DEFAULT_FILLERS
|
||||
assert "uma" not in DEFAULT_FILLERS
|
||||
assert "like" not in DEFAULT_FILLERS
|
||||
assert "so" not in DEFAULT_FILLERS
|
||||
|
||||
@@ -269,11 +276,12 @@ class TestRemoveFillerWordsHandler:
|
||||
assert "_defillered" in result[0].text
|
||||
|
||||
segments = _spine_segments(str(tmp_path / "project_defillered.fcpxml"))
|
||||
# um (0-0.5) trims the clip head, uh (3-3.5) splits -> 2 segments, ~7s total
|
||||
# Only uh (3-3.5) is removed by default; Portuguese "um" is preserved
|
||||
# because it is often grammatical speech ("de um jeito").
|
||||
assert len(segments) == 2
|
||||
total = sum(c.duration.seconds for c in segments)
|
||||
assert total == pytest.approx(7.0, abs=0.1)
|
||||
assert segments[0].source_start.seconds == pytest.approx(0.5, abs=0.05)
|
||||
assert total == pytest.approx(7.5, abs=0.1)
|
||||
assert segments[0].source_start.seconds == pytest.approx(0.0, abs=0.05)
|
||||
assert segments[1].source_start.seconds == pytest.approx(3.5, abs=0.05)
|
||||
|
||||
async def test_no_fillers_found_saves_nothing(self, tmp_path):
|
||||
|
||||
@@ -76,9 +76,14 @@ class TestParseActions:
|
||||
|
||||
|
||||
class TestZoomValidation:
|
||||
def test_default_scale_when_absent(self):
|
||||
def test_absent_scale_is_left_absent(self):
|
||||
# The parser no longer stamps a default: an omitted scale must reach the
|
||||
# applier untouched so it can fall back to the user's configured
|
||||
# `zoom_scale` (see server_tools/_shared.py). Filling one in here would
|
||||
# silently override that setting for every action the model sends
|
||||
# without an explicit scale.
|
||||
actions, _ = parse_actions([{"kind": "zoom", "start": 1.0, "end": 2.0}])
|
||||
assert actions[0].params["scale"] == 1.3
|
||||
assert "scale" not in actions[0].params
|
||||
|
||||
def test_rejects_scale_below_one(self):
|
||||
_, errors = parse_actions([
|
||||
|
||||
@@ -94,6 +94,32 @@ class TestApplyVoiceActionsHandler:
|
||||
texts = [t.text for t in titles[0].iter() if t.text]
|
||||
assert any("SEGURANÇA" in t for t in texts)
|
||||
|
||||
async def test_text_callout_defaults_fit_the_frame(self, project):
|
||||
from fcpxml.writer import FCPXMLModifier
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
await handle_apply_voice_actions({
|
||||
"filepath": str(project),
|
||||
"actions": [{
|
||||
"kind": "text", "start": 3.0, "end": 4.0,
|
||||
"params": {"content": "PRÓTESES DE SILICONE"},
|
||||
}],
|
||||
})
|
||||
|
||||
modifier = FCPXMLModifier(str(_out(project)))
|
||||
report = modifier.validate_subtitle_layout()
|
||||
assert report["summary"]["outside_frame"] == 0
|
||||
|
||||
title = modifier.root.find(".//title")
|
||||
style = title.find("text-style-def/text-style")
|
||||
position = next(
|
||||
p.get("value")
|
||||
for p in title.findall("param")
|
||||
if p.get("name") == "Position"
|
||||
)
|
||||
assert float(style.get("fontSize")) < 530
|
||||
assert position != "0 0"
|
||||
|
||||
async def test_applies_marker(self, project):
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ import pytest
|
||||
|
||||
from fcpxml.voice_timeline import (
|
||||
VOICE_TIMELINE_VERSION,
|
||||
annotate_emotions,
|
||||
build_voice_timeline,
|
||||
enrich_words,
|
||||
load_voice_timeline,
|
||||
@@ -60,6 +61,14 @@ class TestEnrichWords:
|
||||
assert all(w["energy_norm"] == 0.0 for w in enriched)
|
||||
assert all(w["pitch_delta"] == 0.0 for w in enriched)
|
||||
|
||||
def test_emotion_labels_are_added_when_enabled(self):
|
||||
enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY)
|
||||
emotional = annotate_emotions(enriched, enabled=True, sensitivity=0.1)
|
||||
loudest = max(emotional, key=lambda w: w["arousal"])
|
||||
assert loudest["word"] == "seguranca"
|
||||
assert loudest["emotion"] in {"excited", "tense", "neutral"}
|
||||
assert 0.0 <= loudest["emotion_confidence"] <= 1.0
|
||||
|
||||
|
||||
class TestBuildVoiceTimeline:
|
||||
@pytest.fixture
|
||||
@@ -74,6 +83,7 @@ class TestBuildVoiceTimeline:
|
||||
for key in ("version", "source", "language", "scales", "summary", "speakers", "segments"):
|
||||
assert key in timeline
|
||||
assert timeline["version"] == VOICE_TIMELINE_VERSION
|
||||
assert "emotion" in timeline["layers"]
|
||||
|
||||
def test_scales_document_every_word_metric(self, timeline):
|
||||
word = timeline["segments"][0]["words"][0]
|
||||
@@ -106,6 +116,8 @@ class TestBuildVoiceTimeline:
|
||||
quiet_segment = timeline["segments"][0]
|
||||
assert loud_segment["avg_energy"] > quiet_segment["avg_energy"]
|
||||
assert loud_segment["peak_emphasis"] >= max(w["emphasis"] for w in loud_segment["words"])
|
||||
assert "emotion" in loud_segment
|
||||
assert "arousal" in loud_segment
|
||||
|
||||
def test_peak_moments_are_sorted_by_emphasis(self, timeline):
|
||||
peaks = timeline["summary"]["peak_moments"]
|
||||
|
||||
@@ -94,6 +94,10 @@ class TestBuildVoiceTimelineHandler:
|
||||
},
|
||||
"emotion_enabled": False,
|
||||
"emotion_sensitivity": 0.5,
|
||||
"zoom_scale": 1.3,
|
||||
"zoom_mode": "in_out",
|
||||
"zoom_ease_in": 0.25,
|
||||
"zoom_ease_out": 0.04,
|
||||
}
|
||||
|
||||
monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(0.01))
|
||||
|
||||
Reference in New Issue
Block a user