From 1bebee4359a80bf8e81efe901e9ec2c7e4772191 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jo=C3=A3o=20Henrique?= Date: Wed, 19 Aug 2026 21:29:27 -0400 Subject: [PATCH] =?UTF-8?q?feat:=20etapa=205=20do=20assistente=20=E2=80=94?= =?UTF-8?q?=20revis=C3=A3o=20de=20=C3=AAnfases=20com=20timeline?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 --- admin/models_api.py | 281 ++++++- code/Engine/docs/05_EXPERIENCIAS.md | 47 ++ code/MacApp/Sources/App.swift | 30 +- code/MacApp/Sources/CaptionsView.swift | 127 ++- code/MacApp/Sources/ModelDownloadView.swift | 100 ++- code/MacApp/Sources/Models.swift | 139 ++++ code/MacApp/Sources/PhraseReviewModel.swift | 429 ++++++++++ code/MacApp/Sources/PhraseReviewView.swift | 413 ++++++++++ code/MacApp/Sources/PythonBridge.swift | 71 +- code/MacApp/Sources/TimelineTracksView.swift | 506 ++++++++++++ code/MacApp/Sources/TranscriptionView.swift | 6 +- code/MacApp/Sources/VoiceAnalysisView.swift | 72 +- code/MacApp/Sources/WizardView.swift | 808 +++++++++++++++++++ code/fcpxml/model_manager.py | 103 ++- code/fcpxml/phrase_review.py | 547 +++++++++++++ code/fcpxml/transcribe.py | 5 +- code/fcpxml/voice_actions.py | 33 +- code/fcpxml/voice_timeline.py | 89 ++ code/fcpxml/writer.py | 65 +- code/server.py | 2 + code/server_tools/_shared.py | 67 +- code/server_tools/subtitles.py | 186 ++++- code/server_tools/transcript.py | 4 +- code/server_tools/voice.py | 7 +- code/tests/test_dynamic_subtitles.py | 16 + code/tests/test_phrase_review.py | 479 +++++++++++ code/tests/test_transcribe.py | 22 +- code/tests/test_voice_actions.py | 9 +- code/tests/test_voice_actions_tool.py | 26 + code/tests/test_voice_timeline.py | 12 + code/tests/test_voice_timeline_tool.py | 4 + 31 files changed, 4622 insertions(+), 83 deletions(-) create mode 100644 code/MacApp/Sources/PhraseReviewModel.swift create mode 100644 code/MacApp/Sources/PhraseReviewView.swift create mode 100644 code/MacApp/Sources/TimelineTracksView.swift create mode 100644 code/MacApp/Sources/WizardView.swift create mode 100644 code/fcpxml/phrase_review.py create mode 100644 code/tests/test_phrase_review.py diff --git a/admin/models_api.py b/admin/models_api.py index 3052586..5601a8f 100644 --- a/admin/models_api.py +++ b/admin/models_api.py @@ -51,6 +51,23 @@ Commands: `refine_voice_timeline` never has to reopen the audio later. -> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."} + build_phrase_review {"voice_timeline": "..._voice_timeline.json", + "actions": {...}|[...]|null, "fresh": false} + The reviewable script for the wizard's emphasis step: every phrase with + the AI's decision already applied (active/emphasis/trim). A review saved + earlier for the same timeline is returned as-is unless `fresh` is true. + -> {"ok": true, "reused": bool, "source", "duration", "speakers", + "phrases": [{index, start, end, trim_start, trim_end, text, speaker, + active, emphasis (0-3), track, peak_emphasis, + take_boundary, gap_before, reason, words}], + "errors": [...]} + + save_phrase_review {"voice_timeline": "...", "phrases": [...], "source": "...", + "duration": 0.0, "speakers": [...]} + Writes _phrase_review.json plus the _phrase_actions.json derived from it. + -> {"ok": true, "review_path", "actions_path", "emphasis_count", + "removed_count"} + dynamic_subtitle_config {} -> {"ok": true, "band_height", "block_center_y", "line_gap", "font", "font_size", "emphasis_font", "emphasis_face", "emphasis_size", @@ -119,6 +136,10 @@ Commands: -> {"ok": true, "diarization": bool, "diarization_message": "...", "num_speakers": "..."} + acoustics_capability + Whether librosa (pitch/energy for voice analysis) is installed. + -> {"ok": true, "available": bool, "message": "..."} + voice_analysis -> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85, "emphasis_weights": {...}, "emotion_enabled": false, @@ -165,6 +186,7 @@ from fcpxml.model_manager import ( # noqa: E402 load_dynamic_subtitle_config, load_hf_token, load_num_speakers, + load_plain_subtitle_config, load_project_config, load_selected_model, load_silence_config, @@ -175,6 +197,7 @@ from fcpxml.model_manager import ( # noqa: E402 save_hf_token, save_models_dir, save_num_speakers, + save_plain_subtitle_config, save_project_config, save_selected_model, save_silence_config, @@ -200,6 +223,28 @@ def _derived_output(path: str, suffix: str, args: dict) -> str: from server import generate_output_path return generate_output_path(path, suffix) + +def _is_no_change_message(message: str) -> bool: + """Whether a tool completed cleanly without needing to save a new file.""" + text = message.lower() + return any( + token in text + for token in ( + "no cuts to make", + "no silence", + "file unchanged", + "nothing saved", + ) + ) + + +def _emit_no_change_or_error(path: str, message: str) -> int: + if _is_no_change_message(message): + _emit({"ok": True, "path": path, "unchanged": True, "message": message}) + return 0 + _emit({"ok": False, "error": message}) + return 1 + # Download cancellation events, keyed by model name. _CANCEL: dict[str, threading.Event] = {} _LOCK = threading.Lock() @@ -243,6 +288,42 @@ def _save_json_atomic(path: Path, data: Any) -> None: json.load(fh) +def _project_media_paths(path: str) -> list[str]: + proj = parse_fcpxml(path) + tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None) + media_paths: list[str] = [] + if tl is not None: + for clip in getattr(tl, "clips", []): + mp = media_src_to_path(clip.media_path or "") + if mp and Path(mp).is_file() and mp not in media_paths: + media_paths.append(mp) + return media_paths + + +def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path: + p = Path(media_path) + if output_dir: + directory = Path(output_dir).expanduser() + directory.mkdir(parents=True, exist_ok=True) + return directory / f"{p.stem}_voice_timeline.json" + return p.with_name(p.stem + "_voice_timeline.json") + + +def _load_cached_voice_timeline(json_path: Path, media_path: str) -> dict | None: + try: + with open(json_path, encoding="utf-8") as fh: + data = json.load(fh) + except (OSError, json.JSONDecodeError, UnicodeDecodeError): + return None + if not isinstance(data, dict): + return None + if data.get("source") != Path(media_path).name: + return None + if not isinstance(data.get("segments"), list): + return None + return data + + # ── commands ──────────────────────────────────────────────────────────────── @@ -350,8 +431,7 @@ def cmd_remove_silences(args: dict) -> int: contents = asyncio.run(handle_remove_media_silence({**args, "filepath": path, "output_path": output})) message = "\n".join(getattr(content, "text", str(content)) for content in contents) if not Path(output).exists(): - _emit({"ok": False, "error": message}) - return 1 + return _emit_no_change_or_error(path, message) _emit({"ok": True, "path": output, "message": message}) return 0 except Exception as exc: @@ -398,8 +478,7 @@ def cmd_remove_filler_words(args: dict) -> int: contents = asyncio.run(handle_remove_filler_words({**args, "filepath": path, "output_path": output})) message = "\n".join(getattr(content, "text", str(content)) for content in contents) if not Path(output).exists(): - _emit({"ok": False, "error": message}) - return 1 + return _emit_no_change_or_error(path, message) _emit({"ok": True, "path": output, "message": message}) return 0 except Exception as exc: @@ -454,6 +533,29 @@ def cmd_generate_dynamic_subtitles(args: dict) -> int: return 1 +def cmd_generate_plain_subtitles(args: dict) -> int: + """Generate simple static editable subtitle title clips.""" + path = str(args.get("path", "")) + if not path or not Path(path).exists(): + _emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) + return 1 + try: + from server import handle_generate_plain_subtitles + + output = _derived_output(path, "_plain_subtitles", args) + contents = asyncio.run( + handle_generate_plain_subtitles({**args, "filepath": path, "output_path": output}) + ) + message = "\n".join(getattr(content, "text", str(content)) for content in contents) + if not Path(output).exists(): + return _emit_no_change_or_error(path, message) + _emit({"ok": True, "path": output, "message": message}) + return 0 + except Exception as exc: + _emit({"ok": False, "error": str(exc)}) + return 1 + + def cmd_add_zoom(args: dict) -> int: """Add an ease-in/ease-out punch-in zoom to one clip.""" path = str(args.get("path", "")) @@ -720,17 +822,10 @@ def cmd_analyze_voice(args: dict) -> int: num_speakers = str(args.get("num_speakers") or load_num_speakers() or "") try: - proj = parse_fcpxml(path) + media_paths = _project_media_paths(path) except Exception as exc: _emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"}) return 1 - tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None) - media_paths: list[str] = [] - if tl is not None: - for clip in getattr(tl, "clips", []): - mp = media_src_to_path(clip.media_path or "") - if mp and Path(mp).is_file() and mp not in media_paths: - media_paths.append(mp) if not media_paths: _emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."}) return 1 @@ -738,17 +833,41 @@ def cmd_analyze_voice(args: dict) -> int: from server import handle_build_voice_timeline messages: list[str] = [] + output_dir = str(args.get("output_dir") or "").strip() + existing: list[Path] = [] for mp in media_paths: + timeline_path = _voice_timeline_json_path(mp, output_dir) + if _load_cached_voice_timeline(timeline_path, mp) is not None: + existing.append(timeline_path) + if existing and len(existing) == len(media_paths) and not bool(args.get("force_reprocess", False)): + message = "# Voice Timeline Cache\n\n" + message += "Reaproveitando análise de voz existente. Nada foi reprocessado.\n\n" + for timeline_path in existing: + message += f"- **Timeline JSON**: {timeline_path}\n" + _emit({ + "ok": True, + "path": path, + "reused": True, + "timelines": [str(p) for p in existing], + "message": message, + }) + return 0 + + for mp in media_paths: + transcript_path = _transcript_json_path(mp, output_dir) + reused_prefix = "" + if _load_cached_transcript(transcript_path) is not None: + reused_prefix = f"# Cache\n\nReaproveitando transcrição existente: `{transcript_path}`\n\n" try: contents = asyncio.run(handle_build_voice_timeline({ "media_path": mp, "model": model, "language": language, "hf_token": token, "num_speakers": num_speakers, - "output_dir": args.get("output_dir"), + "output_dir": output_dir, })) except Exception as exc: _emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"}) return 1 - messages.append("\n".join(getattr(c, "text", str(c)) for c in contents)) + messages.append(reused_prefix + "\n".join(getattr(c, "text", str(c)) for c in contents)) _emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)}) return 0 @@ -938,9 +1057,23 @@ def cmd_set_diarization(args: dict) -> int: return 0 +def cmd_acoustics_capability(args: dict) -> int: + """Whether librosa (pitch/energy extraction) is installed in this venv. + + Surfaces `features_capability()` — previously computed but never + exposed to the app, so `layers.acoustics: false` in a voice timeline + had no explanation the user could act on. + """ + from fcpxml.voice_features import features_capability + ok, msg = features_capability() + _emit({"ok": True, "available": ok, "message": msg}) + return 0 + + def cmd_voice_analysis(args: dict) -> int: """Read the persisted voice-analysis settings (energy/emphasis/emotion).""" - _emit({"ok": True, **load_voice_analysis_config()}) + config = load_voice_analysis_config() + _emit({"ok": True, **config, "emphasis_threshold": config["emphasis_floor"]}) return 0 @@ -950,9 +1083,13 @@ def cmd_set_voice_analysis(args: dict) -> int: config = save_voice_analysis_config( energy_threshold=args.get("energy_threshold"), emphasis_weights=weights if isinstance(weights, dict) else None, - emphasis_threshold=args.get("emphasis_threshold"), + emphasis_floor=args.get("emphasis_threshold"), emotion_enabled=args.get("emotion_enabled"), emotion_sensitivity=args.get("emotion_sensitivity"), + zoom_scale=args.get("zoom_scale"), + zoom_mode=args.get("zoom_mode"), + zoom_ease_in=args.get("zoom_ease_in"), + zoom_ease_out=args.get("zoom_ease_out"), ) _emit({"ok": True, **config}) return 0 @@ -977,6 +1114,24 @@ def cmd_set_dynamic_subtitle_config(args: dict) -> int: return 0 +def cmd_plain_subtitle_config(args: dict) -> int: + """Read the persisted simple subtitle style.""" + _emit({"ok": True, **load_plain_subtitle_config()}) + return 0 + + +def cmd_set_plain_subtitle_config(args: dict) -> int: + """Persist simple subtitle style fields. Only the given fields change.""" + config = save_plain_subtitle_config(**{ + k: args.get(k) for k in ( + "font", "font_size", "font_color", "max_words", + "position_y", "uppercase", "keep_punctuation", "text_scale", + ) + }) + _emit({"ok": True, **config}) + return 0 + + def cmd_silence_config(args: dict) -> int: """Read the persisted silence thresholds (noise floor, duration, padding).""" _emit({"ok": True, **load_silence_config()}) @@ -1026,6 +1181,14 @@ def cmd_apply_voice_actions(args: dict) -> int: return 1 actions = loaded.get("actions") if isinstance(loaded, dict) else loaded + # The documented output format is {"source": ..., "actions": [...]} — + # callers passing that whole object inline (e.g. the wizard pasting the + # skill's JSON verbatim) need the same unwrap the actions_path branch + # above already does, or a well-formed payload gets rejected as + # "malformed" for having one extra layer of nesting. + if isinstance(actions, dict): + actions = actions.get("actions") + if not isinstance(actions, list) or not actions: _emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."}) return 1 @@ -1054,6 +1217,86 @@ def cmd_apply_voice_actions(args: dict) -> int: return 0 +def cmd_build_phrase_review(args: dict) -> int: + """Build the reviewable script (phrases + the AI's decisions) for the wizard. + + `voice_timeline` points at the _voice_timeline.json; `actions` carries the + decision list the model returned (inline, in any of the shapes the skill + emits). The review is always rebuilt from the current analysis, then the + decisions saved on a previous visit are laid back over it — reopening the + step must show the edits the user left there without freezing the acoustics + as they were when they left. + """ + from fcpxml.phrase_review import ( + build_phrase_review, + load_phrase_review, + merge_saved_decisions, + ) + + timeline_path = str(args.get("voice_timeline", "")) + if not timeline_path or not Path(timeline_path).exists(): + _emit({"ok": False, "error": "Análise de voz (voice_timeline.json) não encontrada."}) + return 1 + + try: + with open(timeline_path, encoding="utf-8") as fh: + timeline = json.load(fh) + except (OSError, ValueError) as exc: + _emit({"ok": False, "error": f"Erro ao ler a análise de voz: {exc}"}) + return 1 + + extra = [d for d in (args.get("output_dir"), args.get("media_dir")) if d] + review = build_phrase_review( + timeline, + args.get("actions"), + voice_timeline_path=timeline_path, + extra_dirs=extra, + ) + + saved = None if args.get("fresh") else load_phrase_review(timeline_path) + review = merge_saved_decisions(review, saved) + _emit({"ok": True, "reused": saved is not None, **review}) + return 0 + + +def cmd_save_phrase_review(args: dict) -> int: + """Persist the edited review and the actions derived from it.""" + from fcpxml.phrase_review import save_phrase_review + + timeline_path = str(args.get("voice_timeline", "")) + if not timeline_path: + _emit({"ok": False, "error": "Caminho da análise de voz não informado."}) + return 1 + + phrases = args.get("phrases") + if not isinstance(phrases, list): + _emit({"ok": False, "error": "Nenhuma frase para salvar."}) + return 1 + + review = { + "version": args.get("version", "1.0"), + "source": args.get("source", ""), + "duration": args.get("duration", 0.0), + "speakers": args.get("speakers", []), + "phrases": phrases, + "zooms": args.get("zooms", []), + } + try: + review_path, actions_path = save_phrase_review(timeline_path, review) + except OSError as exc: + _emit({"ok": False, "error": f"Erro ao salvar a revisão: {exc}"}) + return 1 + + _emit({ + "ok": True, + "review_path": str(review_path), + "actions_path": str(actions_path), + "emphasis_count": sum(1 for p in phrases if int(p.get("emphasis", 0) or 0) >= 1), + "removed_count": sum(1 for p in phrases if not p.get("active", True)), + }) + return 0 + + def cmd_project_config(args: dict) -> int: """Read the last project folder/file the app was working on.""" _emit({"ok": True, **load_project_config()}) @@ -1115,17 +1358,23 @@ def main() -> int: "remove_filler_words": cmd_remove_filler_words, "transcript_markers": cmd_transcript_markers, "generate_dynamic_subtitles": cmd_generate_dynamic_subtitles, + "generate_plain_subtitles": cmd_generate_plain_subtitles, "add_zoom": cmd_add_zoom, "zoom_clips": cmd_zoom_clips, "zoom_segments": cmd_zoom_segments, "rename_speakers": cmd_rename_speakers, "set_diarization": cmd_set_diarization, + "acoustics_capability": cmd_acoustics_capability, "voice_analysis": cmd_voice_analysis, "set_voice_analysis": cmd_set_voice_analysis, "analyze_voice": cmd_analyze_voice, "dynamic_subtitle_config": cmd_dynamic_subtitle_config, "set_dynamic_subtitle_config": cmd_set_dynamic_subtitle_config, + "plain_subtitle_config": cmd_plain_subtitle_config, + "set_plain_subtitle_config": cmd_set_plain_subtitle_config, "apply_voice_actions": cmd_apply_voice_actions, + "build_phrase_review": cmd_build_phrase_review, + "save_phrase_review": cmd_save_phrase_review, "project_config": cmd_project_config, "set_project_config": cmd_set_project_config, "silence_config": cmd_silence_config, diff --git a/code/Engine/docs/05_EXPERIENCIAS.md b/code/Engine/docs/05_EXPERIENCIAS.md index f72fc25..f085a2c 100644 --- a/code/Engine/docs/05_EXPERIENCIAS.md +++ b/code/Engine/docs/05_EXPERIENCIAS.md @@ -1183,6 +1183,51 @@ o outro; percentil entrega um punhado útil nos dois casos. --- +## 21 — 2026-08-19 — Teste travado no default antigo de `zoom scale` + +- **Sintoma:** `tests/test_voice_actions.py::test_default_scale_when_absent` + quebrando com `KeyError: 'scale'`, sem relação com a alteração em curso. +- **Causa raiz:** `parse_actions` deixou de carimbar `scale=1.3` quando o + parâmetro vem ausente, justamente para que + `server_tools/_shared.py` use o `zoom_scale` configurado pelo usuário. O + teste continuou afirmando o default antigo, então passou a acusar como erro + exatamente o comportamento desejado. +- **Solução adotada:** teste reescrito para o contrato novo — um `scale` + omitido tem que chegar ausente ao aplicador (`test_absent_scale_is_left_absent`). +- **Aprendizado:** quando um default sai do parser e vira configuração, o teste + que afirmava o valor antigo passa a defender o bug. Ao remover um default, + procure o teste que o fixava no mesmo commit — senão ele fica dizendo o + contrário do código, e a próxima pessoa perde tempo achando que quebrou algo. +- **Estado:** `resolvido` + +--- + +## 22 — 2026-08-19 — `VideoPlayer` (AVKit) derruba o app compilado por `swiftc` + +- **Sintoma:** "G-ART encerrou inesperadamente" (SIGABRT) toda vez que o + assistente entrava na etapa 5. Nada aparecia na tela antes do crash. +- **Causa raiz:** o app é montado invocando `swiftc` direto + (`MacApp/build_app.sh`), não pelo Xcode. Nesse modo o runtime não consegue + resolver a superclasse Objective-C de `VideoPlayer`: + `failed to demangle superclass of VideoPlayerView from mangled name + 'So12AVPlayerViewC'` → `getSuperclassMetadata` chama `fatalError`. É erro de + runtime, então a compilação passa limpa e o problema só aparece ao abrir a + view. +- **Solução adotada:** trocar `VideoPlayer` por um `AVPlayerLayer` dentro de um + `NSViewRepresentable` (`PlayerSurface`/`PlayerLayerView` em + `PhraseReviewView.swift`). Só depende de AVFoundation, que linka normalmente. + Os controles de transporte já viviam na barra da timeline, então não se perde + nada com a chrome do AVKit. +- **Aprendizado:** compilar limpo não prova que um componente de framework + existe em runtime neste build. Ao usar uma view SwiftUI que embrulha uma + classe AppKit/ObjC (AVKit, WebKit, MapKit), abra a tela de fato antes de + concluir. Um harness pequeno (`swiftc` com os mesmos fontes + um `@main` que + monta só aquela view e sai) reproduz o crash em segundos, sem precisar + navegar o app inteiro até lá. +- **Estado:** `resolvido` + +--- + ## Resumo rápido (índice) | # | Data | Problema | Estado | @@ -1205,5 +1250,7 @@ o outro; percentil entrega um punhado útil nos dois casos. | 18 | 2026-08-19 | Legendas dinâmicas geradas com `bold="0" fontFace="Bold"` não renderizam no FCP — negrito deve ser `bold="1"` (atributo) e itálico `fontFace`+`italic="1"` | `resolvido` | | 19 | 2026-08-19 | `output_dir` usado só como cerca de validação e nunca como destino — toda chamada entre pastas falhava acusando o caminho que ela mesma gerou | `resolvido` | | 20 | 2026-08-19 | `apply_voice_actions` ausente da ponte e do encadeamento do app — dava para analisar e legendar, não para cortar | `resolvido` | +| 21 | 2026-08-19 | Teste ainda afirmava o default `zoom scale=1.3` removido do parser (agora vem do `zoom_scale` do usuário) | `resolvido` | +| 22 | 2026-08-19 | `VideoPlayer` (AVKit) aborta em runtime no app compilado por `swiftc` — etapa 5 fechava o app; trocado por `AVPlayerLayer` | `resolvido` | > Mantenha o índice acima sempre sincronizado com as entradas mais recentes. diff --git a/code/MacApp/Sources/App.swift b/code/MacApp/Sources/App.swift index 23cef88..fb9e9ef 100644 --- a/code/MacApp/Sources/App.swift +++ b/code/MacApp/Sources/App.swift @@ -12,6 +12,7 @@ struct GArtApp: App { } enum ActiveTab: Hashable { + case wizard case project case captions case voiceAnalysis @@ -20,19 +21,23 @@ enum ActiveTab: Hashable { } struct ContentView: View { - @State private var activeTab: ActiveTab? = .project + @State private var activeTab: ActiveTab? = .wizard var body: some View { NavigationSplitView { List(selection: $activeTab) { - Label("Projeto", systemImage: "film") - .tag(ActiveTab.project) - Label("Legendas Dinâmicas", systemImage: "captions.bubble") - .tag(ActiveTab.captions) - Label("Análise de Voz", systemImage: "waveform") - .tag(ActiveTab.voiceAnalysis) - Label("Modelos", systemImage: "tray.and.arrow.down") - .tag(ActiveTab.models) + Label("Assistente", systemImage: "wand.and.stars") + .tag(ActiveTab.wizard) + Section("Avançado") { + Label("Projeto", systemImage: "film") + .tag(ActiveTab.project) + Label("Legendas", systemImage: "captions.bubble") + .tag(ActiveTab.captions) + Label("Análise de Voz", systemImage: "waveform") + .tag(ActiveTab.voiceAnalysis) + Label("Modelos", systemImage: "tray.and.arrow.down") + .tag(ActiveTab.models) + } Label("Sobre", systemImage: "info.circle") .tag(ActiveTab.about) } @@ -40,19 +45,22 @@ struct ContentView: View { .navigationSplitViewColumnWidth(min: 180, ideal: 200) } detail: { switch activeTab { + case .wizard, nil: + WizardView().id(UUID()) + .navigationTitle("Assistente") case .project: ProjectView().id(UUID()) .navigationTitle("Projeto") case .captions: CaptionsView().id(UUID()) - .navigationTitle("Legendas Dinâmicas") + .navigationTitle("Legendas") case .voiceAnalysis: VoiceAnalysisView().id(UUID()) .navigationTitle("Análise de Voz") case .models: ModelDownloadView().id(UUID()) .navigationTitle("Modelos") - case .about, nil: + case .about: AboutView() .navigationTitle("Sobre") } diff --git a/code/MacApp/Sources/CaptionsView.swift b/code/MacApp/Sources/CaptionsView.swift index ea959ea..7c807c9 100644 --- a/code/MacApp/Sources/CaptionsView.swift +++ b/code/MacApp/Sources/CaptionsView.swift @@ -19,6 +19,7 @@ import UniformTypeIdentifiers /// assunto. struct CaptionsView: View { @State private var config = CaptionStyleConfig.defaults + @State private var plainConfig = PlainSubtitleConfig.defaults @State private var isLoading = true @State private var errorMessage: String? @@ -53,6 +54,13 @@ struct CaptionsView: View { ) } + private func plainBound(_ keyPath: WritableKeyPath) -> Binding { + Binding( + get: { plainConfig[keyPath: keyPath] }, + set: { plainConfig[keyPath: keyPath] = $0; savePlain() } + ) + } + private func colorBound(_ keyPath: WritableKeyPath) -> Binding { Binding( get: { Color(rgbaString: config[keyPath: keyPath]) }, @@ -60,6 +68,13 @@ struct CaptionsView: View { ) } + private func plainColorBound(_ keyPath: WritableKeyPath) -> Binding { + Binding( + get: { Color(rgbaString: plainConfig[keyPath: keyPath]) }, + set: { plainConfig[keyPath: keyPath] = $0.fcpxmlColorString; savePlain() } + ) + } + var body: some View { HSplitView { previewColumn @@ -152,6 +167,7 @@ struct CaptionsView: View { positionSection bodySection emphasisSection + plainSubtitleSection calibrationSection } if let errorMessage { @@ -225,6 +241,35 @@ struct CaptionsView: View { } } + private var plainSubtitleSection: some View { + Section("Legenda comum") { + Picker("Fonte", selection: plainBound(\.font)) { + ForEach(fontChoices, id: \.self) { Text($0).tag($0) } + } + slider( + "Tamanho", + value: plainBound(\.fontSize), in: 28...300, step: 1, + readout: "\(Int(plainConfig.fontSize))pt", + help: "Tamanho da legenda comum editável no Final Cut." + ) + slider( + "Máximo de palavras", + value: plainBound(\.maxWords), in: 1...14, step: 1, + readout: "\(Int(plainConfig.maxWords))", + help: "Quantidade máxima de palavras por bloco de legenda." + ) + slider( + "Altura", + value: plainBound(\.positionY), in: -1200...300, step: 1, + readout: "\(Int(plainConfig.positionY))", + help: "Posição vertical da legenda comum no quadro; valores mais negativos descem." + ) + ColorPicker("Cor", selection: plainColorBound(\.fontColor), supportsOpacity: true) + Toggle("Usar letra maiúscula", isOn: plainBound(\.uppercase)) + Toggle("Manter vírgula e ponto", isOn: plainBound(\.keepPunctuation)) + } + } + private var calibrationSection: some View { Section { slider( @@ -305,8 +350,17 @@ struct CaptionsView: View { } else if let error { errorMessage = error } - isLoading = false - continuation.resume() + PythonBridge.call(command: "plain_subtitle_config") { plainResult, plainError in + DispatchQueue.main.async { + if let plainResult { + plainConfig = PlainSubtitleConfig(from: plainResult) + } else if let plainError { + errorMessage = plainError + } + isLoading = false + continuation.resume() + } + } } } } @@ -317,6 +371,12 @@ struct CaptionsView: View { DispatchQueue.main.async { errorMessage = error } } } + + private func savePlain() { + PythonBridge.call(command: "set_plain_subtitle_config", arguments: plainConfig.arguments()) { _, error in + DispatchQueue.main.async { errorMessage = error } + } + } } /// O estilo das legendas dinâmicas, no formato que a tela edita e o bridge @@ -402,6 +462,69 @@ struct CaptionStyleConfig { } } +struct PlainSubtitleConfig { + var font: String + var fontSize: Double + var fontColor: String + var maxWords: Double + var positionY: Double + var uppercase: Bool + var keepPunctuation: Bool + var textScale: Double + + static let defaults = PlainSubtitleConfig( + font: "Helvetica Neue", + fontSize: 82, + fontColor: "1 1 1 1", + maxWords: 7, + positionY: -820, + uppercase: false, + keepPunctuation: true, + textScale: 2.0 + ) + + init(from json: [String: Any]) { + let d = PlainSubtitleConfig.defaults + self.init( + font: json["font"] as? String ?? d.font, + fontSize: (json["font_size"] as? NSNumber)?.doubleValue ?? d.fontSize, + fontColor: json["font_color"] as? String ?? d.fontColor, + maxWords: (json["max_words"] as? NSNumber)?.doubleValue ?? d.maxWords, + positionY: (json["position_y"] as? NSNumber)?.doubleValue ?? d.positionY, + uppercase: json["uppercase"] as? Bool ?? d.uppercase, + keepPunctuation: json["keep_punctuation"] as? Bool ?? d.keepPunctuation, + textScale: (json["text_scale"] as? NSNumber)?.doubleValue ?? d.textScale + ) + } + + init( + font: String, fontSize: Double, fontColor: String, maxWords: Double, + positionY: Double, uppercase: Bool, keepPunctuation: Bool, textScale: Double + ) { + self.font = font + self.fontSize = fontSize + self.fontColor = fontColor + self.maxWords = maxWords + self.positionY = positionY + self.uppercase = uppercase + self.keepPunctuation = keepPunctuation + self.textScale = textScale + } + + func arguments() -> [String: Any] { + [ + "font": font, + "font_size": Int(fontSize), + "font_color": fontColor, + "max_words": Int(maxWords), + "position_y": positionY, + "uppercase": uppercase, + "keep_punctuation": keepPunctuation, + "text_scale": textScale, + ] + } +} + extension Color { /// Parses an FCPXML "R G B A" space-separated 0-1 string into a Color. init(rgbaString: String) { diff --git a/code/MacApp/Sources/ModelDownloadView.swift b/code/MacApp/Sources/ModelDownloadView.swift index 6fb4cf2..20f6b77 100644 --- a/code/MacApp/Sources/ModelDownloadView.swift +++ b/code/MacApp/Sources/ModelDownloadView.swift @@ -15,6 +15,11 @@ struct ModelDownloadView: View { @State private var hfTokenText: String = "" @State private var numSpeakersText: String = "" @State private var language: String = "auto" + @State private var acousticsAvailable: Bool? + @State private var acousticsMessage: String = "" + @State private var isInstallingAcoustics = false + @State private var acousticsInstallLog: String = "" + @State private var acousticsInstallError: String? private let languages: [(String, String)] = [ ("auto", "Detectar automaticamente"), @@ -33,6 +38,7 @@ struct ModelDownloadView: View { var body: some View { Form { storageSection + acousticsSection diarizationSection if let errorMessage { Section { @@ -65,7 +71,7 @@ struct ModelDownloadView: View { } } .formStyle(.grouped) - .task { await refresh() } + .task { await refresh(); checkAcoustics() } } // MARK: - Transcription language @@ -95,6 +101,98 @@ struct ModelDownloadView: View { PythonBridge.call(command: "set_language", arguments: ["language": code]) { _, _ in } } + // MARK: - Acoustic analysis (librosa) + + /// A ênfase de voz (pitch/energia) precisa do `librosa`, que é uma + /// dependência opcional — sem ela `layers.acoustics` vem `false` na + /// análise e a decisão de zoom fica sem base real. Antes disso só dava + /// pra descobrir lendo o JSON exportado; agora o app já diz e resolve. + private var acousticsSection: some View { + Section { + VStack(alignment: .leading, spacing: 10) { + if let acousticsAvailable { + Label( + acousticsMessage.isEmpty + ? (acousticsAvailable ? "Disponível" : "Indisponível") + : acousticsMessage, + systemImage: acousticsAvailable ? "checkmark.circle.fill" : "exclamationmark.triangle.fill" + ) + .font(.caption) + .foregroundStyle(acousticsAvailable ? Color.green : Color.orange) + } else { + Label("Verificando…", systemImage: "hourglass") + .font(.caption).foregroundStyle(.secondary) + } + + if acousticsAvailable == false { + Button { + installAcoustics() + } label: { + if isInstallingAcoustics { + HStack { ProgressView().controlSize(.small); Text("Instalando…") } + } else { + Label("Instalar (uv sync --all-extras)", systemImage: "arrow.down.circle") + } + } + .disabled(isInstallingAcoustics) + + if !acousticsInstallLog.isEmpty { + ScrollView { + Text(acousticsInstallLog) + .font(.system(.caption2, design: .monospaced)) + .foregroundStyle(.secondary) + .frame(maxWidth: .infinity, alignment: .leading) + } + .frame(height: 90) + .background(RoundedRectangle(cornerRadius: 6).fill(Color.secondary.opacity(0.06))) + } + if let acousticsInstallError { + Label(acousticsInstallError, systemImage: "xmark.circle.fill") + .font(.caption).foregroundStyle(.red) + } + } + } + } header: { + Text("Análise Acústica (zoom por voz)") + } footer: { + Text("Mede a energia e o tom de voz de verdade, para os candidatos a zoom da edição por voz. Sem isso, a análise ainda transcreve e decide cortes pelo texto — só o zoom fica sem base acústica.") + .font(.caption) + .foregroundStyle(.secondary) + } + } + + private func checkAcoustics() { + PythonBridge.call(command: "acoustics_capability") { result, err in + DispatchQueue.main.async { + guard let result, result["ok"] as? Bool == true else { return } + acousticsAvailable = result["available"] as? Bool + acousticsMessage = result["message"] as? String ?? "" + } + } + } + + private func installAcoustics() { + isInstallingAcoustics = true + acousticsInstallLog = "" + acousticsInstallError = nil + // --all-extras, não só "intelligence": `uv sync` substitui o + // ambiente pelos extras pedidos em vez de somar, então um sync + // parcial aqui derrubaria dev/transcribe/diarização já instalados. + PythonBridge.runUV(arguments: ["sync", "--all-extras"]) { line in + DispatchQueue.main.async { + acousticsInstallLog += (acousticsInstallLog.isEmpty ? "" : "\n") + line + } + } completion: { code, err in + DispatchQueue.main.async { + isInstallingAcoustics = false + if code != 0 { + acousticsInstallError = err ?? "Falha ao instalar." + } + checkAcoustics() + } + } + } + // MARK: - Diarization private var diarizationSection: some View { diff --git a/code/MacApp/Sources/Models.swift b/code/MacApp/Sources/Models.swift index 115c864..da6d4fa 100644 --- a/code/MacApp/Sources/Models.swift +++ b/code/MacApp/Sources/Models.swift @@ -116,6 +116,145 @@ struct ZoomClip: Identifiable { } } +/// One word inside a phrase, with the acoustics that justify an emphasis. +struct ReviewWord: Identifiable { + let id: Int + let text: String + let start: Double + let end: Double + let energy: Double + let emphasis: Double + + init(id: Int, json: [String: Any]) { + self.id = id + text = json["text"] as? String ?? "" + start = json["start"] as? Double ?? 0 + end = json["end"] as? Double ?? 0 + energy = json["energy"] as? Double ?? 0 + emphasis = json["emphasis"] as? Double ?? 0 + } +} + +/// A phrase in the review step — one spoken line plus the decision made about +/// it. Mirrors `fcpxml/phrase_review.py`; `emphasis` is 0–3 and everything +/// mutable here is what the editor is allowed to change. +struct ReviewPhrase: Identifiable { + let id: Int + let start: Double + let end: Double + var trimStart: Double + var trimEnd: Double + var text: String + let speaker: String + var active: Bool + var emphasis: Int + var track: String + let peakEmphasis: Double + let emotion: String + let emotionConfidence: Double + let takeBoundary: Bool + let gapBefore: Double + let reason: String + let words: [ReviewWord] + + static let trackScript = "roteiro" + static let trackBackstage = "bastidor" + + /// Delivery emotion as the analysis names it, in the user's language plus a + /// glyph — the label alone is too easy to skim past in a dense list. + static func emotionLabel(_ emotion: String) -> (String, String) { + switch emotion { + case "excited": return ("Empolgado", "flame") + case "tense": return ("Tenso", "bolt") + case "calm": return ("Calmo", "leaf") + case "reflective": return ("Reflexivo", "moon") + default: return ("Neutro", "circle") + } + } + + init(json: [String: Any]) { + id = json["index"] as? Int ?? 0 + start = json["start"] as? Double ?? 0 + end = json["end"] as? Double ?? 0 + trimStart = json["trim_start"] as? Double ?? (json["start"] as? Double ?? 0) + trimEnd = json["trim_end"] as? Double ?? (json["end"] as? Double ?? 0) + text = json["text"] as? String ?? "" + speaker = json["speaker"] as? String ?? "" + active = json["active"] as? Bool ?? true + emphasis = json["emphasis"] as? Int ?? 0 + track = json["track"] as? String ?? ReviewPhrase.trackScript + peakEmphasis = json["peak_emphasis"] as? Double ?? 0 + emotion = json["emotion"] as? String ?? "neutral" + emotionConfidence = json["emotion_confidence"] as? Double ?? 0 + takeBoundary = json["take_boundary"] as? Bool ?? false + gapBefore = json["gap_before"] as? Double ?? 0 + reason = json["reason"] as? String ?? "" + words = (json["words"] as? [[String: Any]] ?? []) + .enumerated().map { ReviewWord(id: $0.offset, json: $0.element) } + } + + var asJSON: [String: Any] { + [ + "index": id, + "start": start, + "end": end, + "trim_start": trimStart, + "trim_end": trimEnd, + "text": text, + "speaker": speaker, + "active": active, + "emphasis": emphasis, + "track": track, + "reason": reason, + ] + } + + var isBackstage: Bool { track == ReviewPhrase.trackBackstage } + var isTrimmed: Bool { trimStart > start + 0.001 || trimEnd < end - 0.001 } + var timecode: String { + String(format: "%02d:%02d", Int(start) / 60, Int(start) % 60) + } + + /// The word boundaries a trim handle is allowed to land on. + func snap(_ time: Double, edge: TrimEdge) -> Double { + let boundaries = words.map { edge == .start ? $0.start : $0.end }.filter { $0 > 0 } + guard let nearest = boundaries.min(by: { abs($0 - time) < abs($1 - time) }) else { + return time + } + return nearest + } +} + +enum TrimEdge { case start, end } + +/// A punch-in the editor placed by hand over an arbitrary range, next to the +/// whole-phrase zoom that an emphasis level produces. It stores only *when* — +/// the scale and the ramp come from the Voice Analysis settings at render time. +struct ManualZoom: Identifiable { + let id = UUID() + var start: Double + var end: Double + + /// Below this a punch-in has no room to ramp in and back out; the writer + /// rejects the window, so offering it would place nothing. + static let minimumDuration: Double = 0.4 + + init(start: Double, end: Double) { + self.start = start + self.end = end + } + + init?(json: [String: Any]) { + guard let start = json["start"] as? Double, let end = json["end"] as? Double, + end - start >= ManualZoom.minimumDuration + else { return nil } + self.start = start + self.end = end + } + + var asJSON: [String: Any] { ["start": start, "end": end] } +} + struct ZoomSegment: Identifiable { let id: Int let start: Double diff --git a/code/MacApp/Sources/PhraseReviewModel.swift b/code/MacApp/Sources/PhraseReviewModel.swift new file mode 100644 index 0000000..cfc4886 --- /dev/null +++ b/code/MacApp/Sources/PhraseReviewModel.swift @@ -0,0 +1,429 @@ +import AVFoundation +import Combine +import Foundation + +/// State behind the wizard's emphasis-review step. +/// +/// Holds the phrases, the selection, and the player — together, because they +/// are one thing to the user: clicking a phrase moves the playhead, playing +/// moves the selection, and skipping a removed line only works if whoever owns +/// playback also knows which lines are removed. +/// +/// The preview deliberately plays the *original* media and jumps over whatever +/// the edit removes, instead of rendering a cut first. Rendering to check a +/// toggle would put minutes between a decision and its result; jumping gives +/// the same reading instantly, and the real cut is generated later from the +/// exact same phrase list. +@MainActor +final class PhraseReviewModel: ObservableObject { + @Published var phrases: [ReviewPhrase] = [] + @Published var selection: Int? + @Published var isLoading = false + @Published var errorMessage: String? + @Published var currentTime: Double = 0 + @Published var isPlaying = false + @Published var pixelsPerSecond: Double = 40 + @Published var skipRemoved = true + @Published var zooms: [ManualZoom] = [] + /// In/out the editor dragged on the timeline, in source seconds. + @Published var rangeStart: Double? + @Published var rangeEnd: Double? + /// Aspect ratio of the footage as recorded. + @Published var videoAspect: Double = 16.0 / 9.0 + /// Aspect ratio the project delivers in, read from the .fcpxml. It is + /// routinely *not* the footage's: these takes are shot horizontal and + /// delivered vertical, so previewing the raw frame would show a crop the + /// audience never sees — and the emphasis decisions are about what lands on + /// screen. Nil until the project is known. + @Published var projectAspect: Double? + /// Whether the preview crops to the delivery frame. On by default whenever + /// the two aspects disagree. + @Published var matchProjectFraming = true + + /// What the preview should actually draw. + var previewAspect: Double { + guard matchProjectFraming, let projectAspect else { return videoAspect } + return projectAspect + } + + /// True when the delivery frame differs enough from the footage that the + /// preview is showing a crop rather than the whole take. + var isCropping: Bool { + guard matchProjectFraming, let projectAspect else { return false } + return abs(projectAspect - videoAspect) > 0.01 + } + + private(set) var source = "" + private(set) var sourcePath = "" + private(set) var duration: Double = 0 + private(set) var speakers: [String] = [] + private(set) var emotionAvailable = false + private(set) var player: AVPlayer? + + private var voiceTimelinePath = "" + private var timeObserver: Any? + private var playbackLimit: Double? + + let minPixelsPerSecond: Double = 8 + let maxPixelsPerSecond: Double = 400 + + deinit { + if let timeObserver, let player { + player.removeTimeObserver(timeObserver) + } + } + + // MARK: - Carregar + + /// Builds the review from the voice timeline plus whatever the AI decided. + /// A review saved on a previous visit wins — see `cmd_build_phrase_review`. + /// Reads the delivery format from the project so the preview can frame the + /// take the way it will actually be seen. + func loadProjectFormat(projectPath: String) { + PythonBridge.call(command: "inspect", arguments: ["path": projectPath]) { [weak self] result, _ in + Task { @MainActor in + guard let self, + let timelines = result?["timelines"] as? [[String: Any]], + let first = timelines.first, + let width = first["width"] as? Int, let height = first["height"] as? Int, + width > 0, height > 0 + else { return } + self.projectAspect = Double(width) / Double(height) + } + } + } + + func load(voiceTimelinePath: String, decisionsJSON: String, + outputFolder: String? = nil, mediaFolder: String? = nil) { + self.voiceTimelinePath = voiceTimelinePath + isLoading = true + errorMessage = nil + + var arguments: [String: Any] = ["voice_timeline": voiceTimelinePath] + if let outputFolder { arguments["output_dir"] = outputFolder } + if let mediaFolder { arguments["media_dir"] = mediaFolder } + if let data = decisionsJSON.data(using: .utf8), + let parsed = try? JSONSerialization.jsonObject(with: data) { + arguments["actions"] = parsed + } + + PythonBridge.call(command: "build_phrase_review", arguments: arguments) { [weak self] result, error in + Task { @MainActor in + guard let self else { return } + self.isLoading = false + if let error { + self.errorMessage = error + return + } + guard let result, result["ok"] as? Bool == true else { + self.errorMessage = result?["error"] as? String ?? "Não foi possível montar a revisão." + return + } + self.apply(result) + } + } + } + + private func apply(_ result: [String: Any]) { + source = result["source"] as? String ?? "" + // The timeline JSON stores only the media's file name; the bridge + // resolves it to something openable (see phrase_review.resolve_source). + sourcePath = result["source_path"] as? String ?? "" + duration = result["duration"] as? Double ?? 0 + speakers = result["speakers"] as? [String] ?? [] + emotionAvailable = result["emotion_available"] as? Bool ?? false + phrases = (result["phrases"] as? [[String: Any]] ?? []).map { ReviewPhrase(json: $0) } + zooms = (result["zooms"] as? [[String: Any]] ?? []).compactMap { ManualZoom(json: $0) } + selection = phrases.first?.id + if let errors = result["errors"] as? [String], !errors.isEmpty { + errorMessage = "A IA mandou \(errors.count) decisão(ões) que não deu para ler — o resto foi aplicado." + } + preparePlayer() + } + + /// Point the preview at a media file the user chose by hand — the way out + /// when the footage moved somewhere the automatic lookup can't reach. + func useMedia(at path: String) { + sourcePath = path + preparePlayer() + } + + private func preparePlayer() { + guard !sourcePath.isEmpty, FileManager.default.fileExists(atPath: sourcePath) else { + player = nil + return + } + if let timeObserver, let player { + player.removeTimeObserver(timeObserver) + self.timeObserver = nil + } + let asset = AVURLAsset(url: URL(fileURLWithPath: sourcePath)) + let player = AVPlayer(playerItem: AVPlayerItem(asset: asset)) + self.player = player + readAspect(from: asset) + // 60 Hz: the same observer drives the playhead *and* decides when to + // jump a removed stretch, so its period is the worst-case amount of cut + // material that can be heard before the skip lands. At 20 Hz that was an + // audible blip on every join. + let interval = CMTime(seconds: 1.0 / 60.0, preferredTimescale: 600) + timeObserver = player.addPeriodicTimeObserver(forInterval: interval, queue: .main) { [weak self] time in + Task { @MainActor in + self?.tick(time.seconds) + } + } + } + + /// The displayed aspect ratio, honouring the rotation the camera recorded. + /// A phone take is stored 1920×1080 with a 90° transform: reading + /// `naturalSize` alone would call a vertical video horizontal. + private func readAspect(from asset: AVURLAsset) { + Task { [weak self] in + guard let track = try? await asset.loadTracks(withMediaType: .video).first, + let size = try? await track.load(.naturalSize), + let transform = try? await track.load(.preferredTransform) + else { return } + let displayed = size.applying(transform) + let width = abs(displayed.width), height = abs(displayed.height) + guard width > 0, height > 0 else { return } + await MainActor.run { self?.videoAspect = width / height } + } + } + + // MARK: - Reprodução + + private func tick(_ time: Double) { + currentTime = time + guard isPlaying else { return } + + // Playing a single phrase or a marked range stops at its out point + // instead of running on into the rest of the take. + if let limit = playbackLimit, time >= limit { + pause() + seek(to: limit) + return + } + + if skipRemoved, let jump = nextKeptTime(after: time), jump > time { + seek(to: jump) + } + if let phrase = phrase(at: time), selection != phrase.id { + selection = phrase.id + } + } + + /// Where playback should resume when `time` lands on removed material. + /// Returns nil when the time is on material that survives. + func nextKeptTime(after time: Double) -> Double? { + for phrase in phrases where time >= phrase.start - 0.001 && time < phrase.end { + if !phrase.active { return phrase.end } + if time < phrase.trimStart { return phrase.trimStart } + if time >= phrase.trimEnd { return phrase.end } + return nil + } + return nil + } + + func togglePlay() { + if isPlaying { + pause() + } else { + playbackLimit = nil + play() + } + } + + private func play() { + guard let player else { return } + if skipRemoved, let jump = nextKeptTime(after: currentTime) { seek(to: jump) } + player.play() + isPlaying = true + } + + func pause() { + player?.pause() + isPlaying = false + playbackLimit = nil + } + + /// Play exactly one span and stop — how a cut is judged: in context, at + /// speed, without hunting for the out point by hand. + func playRange(from start: Double, to end: Double) { + guard end > start else { return } + seek(to: start) + playbackLimit = end + player?.play() + isPlaying = true + } + + func playSelectedPhrase() { + guard let selection, let phrase = phrases.first(where: { $0.id == selection }) + else { return } + playRange(from: phrase.active ? phrase.trimStart : phrase.start, + to: phrase.active ? phrase.trimEnd : phrase.end) + } + + func seek(to time: Double) { + currentTime = max(0, time) + player?.seek(to: CMTime(seconds: max(0, time), preferredTimescale: 600), + toleranceBefore: .zero, toleranceAfter: .zero) + } + + /// Move the playhead to a phrase and select it. + func goTo(phraseID: Int) { + guard let phrase = phrases.first(where: { $0.id == phraseID }) else { return } + selection = phraseID + seek(to: phrase.active ? phrase.trimStart : phrase.start) + } + + func phrase(at time: Double) -> ReviewPhrase? { + phrases.first { time >= $0.start && time < $0.end } + } + + func selectNeighbour(_ delta: Int) { + guard let selection, let index = phrases.firstIndex(where: { $0.id == selection }) else { + if let first = phrases.first { goTo(phraseID: first.id) } + return + } + let next = min(max(0, index + delta), phrases.count - 1) + goTo(phraseID: phrases[next].id) + } + + // MARK: - Edições + + private func update(_ id: Int, _ change: (inout ReviewPhrase) -> Void) { + guard let index = phrases.firstIndex(where: { $0.id == id }) else { return } + change(&phrases[index]) + } + + func setEmphasis(_ level: Int, for id: Int) { + update(id) { $0.emphasis = min(3, max(0, level)) } + } + + func toggleActive(_ id: Int) { + update(id) { $0.active.toggle() } + } + + func setTrack(_ track: String, for id: Int) { + update(id) { $0.track = track } + } + + func setText(_ text: String, for id: Int) { + update(id) { $0.text = text } + } + + /// Trim a phrase's head or tail, landing on a word boundary. + /// A trim that would swallow the whole line is refused — deactivating the + /// phrase is the way to remove it, and doing it by accident with a drag + /// would lose the emphasis decision along with the line. + func trim(_ id: Int, edge: TrimEdge, to time: Double) { + update(id) { phrase in + let snapped = phrase.snap(time, edge: edge) + switch edge { + case .start: + let value = min(max(phrase.start, snapped), phrase.trimEnd - 0.1) + if value < phrase.trimEnd { phrase.trimStart = value } + case .end: + let value = max(min(phrase.end, snapped), phrase.trimStart + 0.1) + if value > phrase.trimStart { phrase.trimEnd = value } + } + } + } + + func resetTrim(_ id: Int) { + update(id) { $0.trimStart = $0.start; $0.trimEnd = $0.end } + } + + /// Trim everything before/after a given word — the text-first way to cut, + /// since the editor reads the line and points at where it should begin. + func trimToWord(_ word: ReviewWord, edge: TrimEdge, in id: Int) { + trim(id, edge: edge, to: edge == .start ? word.start : word.end) + } + + // MARK: - Trecho marcado e zooms + + var hasRange: Bool { + guard let rangeStart, let rangeEnd else { return false } + return rangeEnd - rangeStart >= ManualZoom.minimumDuration + } + + var rangeSpan: (start: Double, end: Double)? { + guard let rangeStart, let rangeEnd, rangeEnd > rangeStart else { return nil } + return (rangeStart, rangeEnd) + } + + func setRange(from start: Double, to end: Double) { + rangeStart = min(start, end) + rangeEnd = max(start, end) + } + + func clearRange() { + rangeStart = nil + rangeEnd = nil + } + + /// Add a punch-in over the marked range. Scale and ramp are not stored: + /// they come from the "Análise de Voz" settings when the edit is rendered, + /// so changing the look there restyles every zoom at once. + func addZoomForRange() { + guard let span = rangeSpan, span.end - span.start >= ManualZoom.minimumDuration + else { return } + zooms.append(ManualZoom(start: span.start, end: span.end)) + zooms.sort { $0.start < $1.start } + clearRange() + } + + func addZoomForPhrase(_ id: Int) { + guard let phrase = phrases.first(where: { $0.id == id }) else { return } + zooms.append(ManualZoom(start: phrase.trimStart, end: phrase.trimEnd)) + zooms.sort { $0.start < $1.start } + } + + func removeZoom(_ id: UUID) { + zooms.removeAll { $0.id == id } + } + + func zoom(at time: Double) -> ManualZoom? { + zooms.first { time >= $0.start && time <= $0.end } + } + + func setEmphasisForAll(_ level: Int) { + for index in phrases.indices where phrases[index].active { + phrases[index].emphasis = level + } + } + + // MARK: - Resumo e gravação + + var emphasisCount: Int { phrases.filter { $0.active && $0.emphasis >= 1 }.count } + var removedCount: Int { phrases.filter { !$0.active }.count } + var keptDuration: Double { + phrases.filter { $0.active }.reduce(0) { $0 + ($1.trimEnd - $1.trimStart) } + } + + /// Persists the edited review plus the actions derived from it. Called when + /// the wizard advances — the render itself happens in the next step. + func save(completion: @escaping (String?) -> Void) { + guard !voiceTimelinePath.isEmpty, !phrases.isEmpty else { + completion(nil) + return + } + let arguments: [String: Any] = [ + "voice_timeline": voiceTimelinePath, + "source": source, + "duration": duration, + "speakers": speakers, + "phrases": phrases.map { $0.asJSON }, + "zooms": zooms.map { $0.asJSON }, + ] + PythonBridge.call(command: "save_phrase_review", arguments: arguments) { result, error in + Task { @MainActor in + if let error { + completion(nil) + _ = error + return + } + completion(result?["review_path"] as? String) + } + } + } +} diff --git a/code/MacApp/Sources/PhraseReviewView.swift b/code/MacApp/Sources/PhraseReviewView.swift new file mode 100644 index 0000000..26136e8 --- /dev/null +++ b/code/MacApp/Sources/PhraseReviewView.swift @@ -0,0 +1,413 @@ +import AVFoundation +import SwiftUI + +/// The video surface, as a plain `AVPlayerLayer` in an `NSView`. +/// +/// AVKit's `VideoPlayer` would be the obvious choice and is a trap here: this +/// app is built by invoking `swiftc` directly (see `MacApp/build_app.sh`), and +/// `_AVKit_SwiftUI` aborts at launch instantiating its generic metadata under +/// that build. A player layer needs only AVFoundation, which links cleanly — +/// and the transport controls live in the timeline's own toolbar anyway, so +/// nothing is lost by dropping AVKit's chrome. +private struct PlayerSurface: NSViewRepresentable { + let player: AVPlayer + /// When true the frame is filled and cropped instead of letterboxed — used + /// to preview horizontal footage inside a vertical delivery frame. + var fills: Bool + + func makeNSView(context: Context) -> PlayerLayerView { + let view = PlayerLayerView() + view.player = player + view.fills = fills + return view + } + + func updateNSView(_ view: PlayerLayerView, context: Context) { + if view.player !== player { view.player = player } + view.fills = fills + } +} + +final class PlayerLayerView: NSView { + private let playerLayer = AVPlayerLayer() + + var player: AVPlayer? { + get { playerLayer.player } + set { playerLayer.player = newValue } + } + + var fills: Bool = false { + didSet { playerLayer.videoGravity = fills ? .resizeAspectFill : .resizeAspect } + } + + override init(frame frameRect: NSRect) { + super.init(frame: frameRect) + wantsLayer = true + layer = CALayer() + layer?.backgroundColor = NSColor.black.cgColor + playerLayer.videoGravity = .resizeAspect + layer?.addSublayer(playerLayer) + } + + required init?(coder: NSCoder) { + super.init(coder: coder) + wantsLayer = true + layer = CALayer() + playerLayer.videoGravity = .resizeAspect + layer?.addSublayer(playerLayer) + } + + override func layout() { + super.layout() + playerLayer.frame = bounds + } +} + +/// The wizard's emphasis-review step, laid out like an editing room: preview on +/// top, timeline across the bottom, and the script as an inspector down the +/// right side. +/// +/// The arrangement is the point. Every decision here is about a *sentence*, so +/// the same phrase has to be legible in all three places at once — a block on +/// the timeline, a line of text in the inspector, and a moment in the preview. +/// Selecting in any one of them selects in the other two. +struct PhraseReviewView: View { + @ObservedObject var model: PhraseReviewModel + + var body: some View { + VSplitView { + HSplitView { + previewPane + .frame(minWidth: 320, idealWidth: 640) + inspectorPane + .frame(minWidth: 300, idealWidth: 360, maxWidth: 520) + } + .frame(minHeight: 240) + + TimelineTracksView(model: model) + .frame(minHeight: 190, idealHeight: 210) + } + .overlay { if model.isLoading { loadingOverlay } } + .focusable() + .onKeyPress(.space) { model.togglePlay(); return .handled } + .onKeyPress(.return) { model.playSelectedPhrase(); return .handled } + .onKeyPress(.leftArrow) { model.selectNeighbour(-1); return .handled } + .onKeyPress(.rightArrow) { model.selectNeighbour(1); return .handled } + .onKeyPress(characters: .decimalDigits) { press in + guard let level = Int(press.characters), (0...3).contains(level), + let selection = model.selection else { return .ignored } + model.setEmphasis(level, for: selection) + return .handled + } + } + + private var loadingOverlay: some View { + ZStack { + Color(nsColor: .windowBackgroundColor).opacity(0.85) + VStack(spacing: 10) { + ProgressView() + Text("Montando a revisão…").font(.callout).foregroundStyle(.secondary) + } + } + } + + // MARK: - Preview + + private var previewPane: some View { + VStack(spacing: 0) { + if let player = model.player { + // The footage here is usually vertical. Sizing the surface to + // the take's own aspect keeps a 9:16 frame as tall as the pane + // allows instead of shrinking it to fit a horizontal box. + // Framed to what the project delivers, not to what the camera + // recorded: these takes are shot horizontal and cut vertical, + // so the raw frame would show material the audience never sees. + ZStack { + Color.black + PlayerSurface(player: player, fills: model.isCropping) + .aspectRatio(model.previewAspect, contentMode: .fit) + .clipped() + } + .overlay(alignment: .topTrailing) { framingBadge } + } else { + ZStack { + Color.black.opacity(0.85) + VStack(spacing: 10) { + Image(systemName: "film.stack") + .font(.system(size: 28)).foregroundStyle(.secondary) + Text(model.source.isEmpty + ? "A análise de voz não registrou qual mídia foi usada." + : "Não achei \(model.source) na pasta do projeto.") + .font(.callout).foregroundStyle(.secondary) + Text("A revisão funciona igual sem o preview — ele só ajuda a conferir o corte.") + .font(.caption).foregroundStyle(.tertiary) + Button("Localizar a mídia…") { pickMedia() } + .buttonStyle(.bordered) + } + .multilineTextAlignment(.center) + .padding(.horizontal, 24) + } + } + Divider() + summaryBar + } + } + + private var summaryBar: some View { + HStack(spacing: 16) { + summaryItem("text.quote", "\(model.phrases.count) frases") + summaryItem("sparkles", "\(model.emphasisCount) com ênfase") + summaryItem("scissors", "\(model.removedCount) fora do corte") + summaryItem("clock", durationLabel(model.keptDuration)) + if !model.zooms.isEmpty { + summaryItem("plus.magnifyingglass", "\(model.zooms.count) zooms") + } + Spacer() + if let phrase = selectedPhrase, !phrase.reason.isEmpty { + Label(phrase.reason, systemImage: "brain") + .font(.caption).foregroundStyle(.secondary) + .lineLimit(1).truncationMode(.tail) + } + } + .padding(.horizontal, 14) + .padding(.vertical, 8) + } + + private func summaryItem(_ icon: String, _ text: String) -> some View { + Label(text, systemImage: icon).font(.caption).foregroundStyle(.secondary) + } + + private func durationLabel(_ seconds: Double) -> String { + String(format: "%02d:%02d finais", Int(seconds) / 60, Int(seconds) % 60) + } + + /// Says which frame is on screen, and lets the editor flip to the raw take. + /// Without it a centred crop looks like the footage itself, and someone + /// would judge framing on an approximation without knowing it. + @ViewBuilder + private var framingBadge: some View { + if model.projectAspect != nil, abs((model.projectAspect ?? 0) - model.videoAspect) > 0.01 { + Button { + model.matchProjectFraming.toggle() + } label: { + Label(model.matchProjectFraming ? "Enquadramento do projeto" : "Mídia original", + systemImage: model.matchProjectFraming ? "crop" : "rectangle.expand.vertical") + .font(.caption2) + } + .buttonStyle(.borderless) + .padding(6) + .background(Capsule().fill(.black.opacity(0.45))) + .foregroundStyle(.white) + .padding(8) + .help("A fonte é horizontal e o projeto é vertical — o preview mostra o corte central aproximado. O enquadramento real de cada clipe vem do Final Cut.") + } + } + + private func pickMedia() { + let panel = NSOpenPanel() + panel.canChooseFiles = true + panel.canChooseDirectories = false + panel.allowsMultipleSelection = false + panel.prompt = "Usar esta mídia" + panel.message = model.source.isEmpty + ? "Escolha o arquivo de vídeo desta gravação." + : "Escolha onde está \(model.source)." + if panel.runModal() == .OK, let url = panel.url { + model.useMedia(at: url.path) + } + } + + private var selectedPhrase: ReviewPhrase? { + guard let selection = model.selection else { return nil } + return model.phrases.first { $0.id == selection } + } + + // MARK: - Inspector de frases + + private var inspectorPane: some View { + VStack(spacing: 0) { + inspectorHeader + Divider() + List(selection: $model.selection) { + ForEach($model.phrases) { $phrase in + PhraseRow(phrase: $phrase, model: model) + .tag(phrase.id) + } + } + .listStyle(.inset) + .onChange(of: model.selection) { _, newValue in + if let newValue { model.goTo(phraseID: newValue) } + } + } + } + + private var inspectorHeader: some View { + VStack(alignment: .leading, spacing: 6) { + Text("Frases").font(.headline) + Text("Só as frases com ênfase recebem zoom e legenda dinâmica. O resto fica com legenda comum.") + .font(.caption).foregroundStyle(.secondary) + if !model.emotionAvailable { + Label("Emoção da fala não foi detectada nesta análise — ligue em Avançado → Análise de Voz e refaça o passo 3.", + systemImage: "waveform.path.ecg") + .font(.caption2).foregroundStyle(.secondary) + } + HStack(spacing: 8) { + Button("Limpar ênfases") { model.setEmphasisForAll(0) } + .buttonStyle(.link).font(.caption) + Spacer() + Text("0–3 no teclado · ← → navega") + .font(.caption2).foregroundStyle(.secondary) + } + } + .padding(12) + } +} + +/// One phrase in the inspector: the line as it will be said, plus every +/// decision attached to it. Kept in one row on purpose — jumping to a separate +/// detail pane to set a toggle would double the clicks on the most repeated +/// action in the screen. +private struct PhraseRow: View { + @Binding var phrase: ReviewPhrase + @ObservedObject var model: PhraseReviewModel + @State private var isEditing = false + + var body: some View { + VStack(alignment: .leading, spacing: 6) { + HStack(spacing: 6) { + Text(phrase.timecode) + .font(.system(.caption2, design: .monospaced)) + .foregroundStyle(.secondary) + if phrase.takeBoundary { + Image(systemName: "scissors.badge.ellipsis") + .font(.caption2).foregroundStyle(.orange) + .help("Nova tomada começa aqui") + } + if phrase.isTrimmed { + Image(systemName: "arrow.left.and.right.square") + .font(.caption2).foregroundStyle(.blue) + .help("Frase cortada nas pontas") + } + if model.emotionAvailable { + emotionChip + } + Spacer() + Toggle("", isOn: $phrase.active) + .toggleStyle(.switch) + .controlSize(.mini) + .labelsHidden() + .help(phrase.active ? "No corte" : "Fora do corte") + } + + if isEditing { + TextField("Texto da frase", text: $phrase.text, axis: .vertical) + .textFieldStyle(.roundedBorder) + .font(.callout) + .onSubmit { isEditing = false } + } else { + Text(phrase.text.isEmpty ? "(sem texto)" : phrase.text) + .font(.callout) + .foregroundStyle(phrase.active ? .primary : .secondary) + .strikethrough(!phrase.active) + .onTapGesture(count: 2) { isEditing = true } + } + + HStack(spacing: 8) { + Picker("", selection: $phrase.emphasis) { + ForEach(0..<4, id: \.self) { level in + Text(EmphasisPalette.label(level)).tag(level) + } + } + .pickerStyle(.segmented) + .controlSize(.mini) + .labelsHidden() + .disabled(!phrase.active) + + Picker("", selection: $phrase.track) { + Text("Roteiro").tag(ReviewPhrase.trackScript) + Text("Bastidor").tag(ReviewPhrase.trackBackstage) + } + .pickerStyle(.menu) + .controlSize(.mini) + .labelsHidden() + .frame(width: 92) + } + + if model.selection == phrase.id && !phrase.words.isEmpty { + wordTrimmer + } + } + .padding(.vertical, 4) + .opacity(phrase.active ? 1 : 0.55) + } + + /// The delivery emotion the acoustics suggest. Shown faded below its own + /// confidence: a guess the analysis is unsure about should not compete for + /// attention with the emphasis decision, which is the point of the row. + private var emotionChip: some View { + let (label, icon) = ReviewPhrase.emotionLabel(phrase.emotion) + return Label(label, systemImage: icon) + .font(.caption2) + .padding(.horizontal, 5) + .padding(.vertical, 1) + .background( + Capsule().fill(Color.secondary.opacity(0.12)) + ) + .foregroundStyle(phrase.emotionConfidence >= 0.5 ? .secondary : .tertiary) + .help("Emoção da entrega: \(label) — confiança \(Int(phrase.emotionConfidence * 100))%") + } + + /// Trimming by pointing at the transcript: click a word to start the phrase + /// there, option-click to end it there. Same edit as dragging the block's + /// edge on the timeline, but reachable while reading the line. + private var wordTrimmer: some View { + VStack(alignment: .leading, spacing: 4) { + HStack(spacing: 4) { + Text("Cortar pelas palavras").font(.caption2).foregroundStyle(.secondary) + Spacer() + if phrase.isTrimmed { + Button("Inteira") { model.resetTrim(phrase.id) } + .buttonStyle(.link).font(.caption2) + } + } + FlowWords(words: phrase.words, phrase: phrase) { word, edge in + model.trimToWord(word, edge: edge, in: phrase.id) + } + Text("Clique = começa aqui · ⌥clique = termina aqui") + .font(.caption2).foregroundStyle(.tertiary) + } + .padding(.top, 2) + } +} + +/// The phrase's words as wrapping chips, dimmed where they fall outside the trim. +private struct FlowWords: View { + let words: [ReviewWord] + let phrase: ReviewPhrase + let onTrim: (ReviewWord, TrimEdge) -> Void + + var body: some View { + // A LazyVGrid with adaptive columns wraps chips without a custom layout; + // phrases are short enough that the slight raggedness beats the cost of + // hand-rolling a flow layout here. + LazyVGrid(columns: [GridItem(.adaptive(minimum: 44), spacing: 3)], + alignment: .leading, spacing: 3) { + ForEach(words) { word in + let kept = word.start >= phrase.trimStart - 0.001 && word.end <= phrase.trimEnd + 0.001 + Text(word.text) + .font(.caption2) + .padding(.horizontal, 4) + .padding(.vertical, 2) + .background( + RoundedRectangle(cornerRadius: 3) + .fill(kept ? Color.accentColor.opacity(0.12) : Color.secondary.opacity(0.08)) + ) + .foregroundStyle(kept ? .primary : .secondary) + .strikethrough(!kept) + .onTapGesture { + onTrim(word, NSEvent.modifierFlags.contains(.option) ? .end : .start) + } + } + } + } +} diff --git a/code/MacApp/Sources/PythonBridge.swift b/code/MacApp/Sources/PythonBridge.swift index 94f94e3..27d149c 100644 --- a/code/MacApp/Sources/PythonBridge.swift +++ b/code/MacApp/Sources/PythonBridge.swift @@ -44,12 +44,26 @@ enum PythonBridge { return ["python3", scriptURL.path] } + /// `admin/models_api.py` lives outside `code/`, but its dependencies + /// (`pyproject.toml`, `.venv`) live inside it. `uv run` picks the + /// environment from the process's cwd, not from the script path — so + /// running with cwd at the repo root made `uv` create/use a second, + /// empty `.venv` there, silently ignoring everything installed into + /// `code/.venv` (this cost a real debugging session: librosa/pyannote + /// installed successfully but the app kept reporting them missing). + /// Every `uv run` must share the same cwd as `uv sync` to see the same + /// environment. static var workingDirectory: URL { - projectRoot + codeDirectory + } + + /// Directory containing `pyproject.toml` — where `uv sync` must run from. + static var codeDirectory: URL { + projectRoot.appendingPathComponent("code") } /// Locate `uv` on PATH or in common install locations. - private static func findUV() -> String? { + static func findUV() -> String? { if let onPath = which("uv") { return onPath } let candidates = [ "/usr/local/bin/uv", @@ -148,6 +162,59 @@ enum PythonBridge { } } + // MARK: - uv sync (installing optional extras, e.g. acoustic analysis) + + /// Runs `uv ` from `codeDirectory` (where `pyproject.toml` + /// lives), streaming each output line as plain text — used for + /// `sync --extra intelligence` so "Modelos" can install the librosa + /// extra without the user opening a terminal. + static func runUV(arguments: [String], + onLine: @escaping (String) -> Void, + completion: @escaping (Int, String?) -> Void) { + guard let uv = findUV() else { + completion(1, "uv não encontrado. Instale com: curl -LsSf https://astral.sh/uv/install.sh | sh") + return + } + let process = Process() + process.executableURL = URL(fileURLWithPath: "/usr/bin/env") + process.arguments = [uv] + arguments + process.currentDirectoryURL = codeDirectory + + let pipe = Pipe() + process.standardOutput = pipe + process.standardError = pipe + + var buffer = "" + let lock = NSLock() + pipe.fileHandleForReading.readabilityHandler = { handle in + let data = handle.availableData + guard !data.isEmpty, let s = String(data: data, encoding: .utf8) else { return } + lock.lock() + buffer += s + let parts = buffer.split(separator: "\n", omittingEmptySubsequences: false) + buffer = String(parts.last ?? "") + let lines = parts.dropLast() + lock.unlock() + for line in lines where !line.isEmpty { onLine(String(line)) } + } + + process.terminationHandler = { p in + pipe.fileHandleForReading.readabilityHandler = nil + lock.lock() + let last = buffer.trimmingCharacters(in: .whitespacesAndNewlines) + buffer = "" + lock.unlock() + if !last.isEmpty { onLine(last) } + completion(Int(p.terminationStatus), p.terminationStatus == 0 ? nil : "uv sync terminou com erro (código \(p.terminationStatus)).") + } + + do { + try process.run() + } catch { + completion(1, error.localizedDescription) + } + } + // MARK: - Convenience: single JSON result /// Runs a command and delivers the first parsed JSON document as the result. diff --git a/code/MacApp/Sources/TimelineTracksView.swift b/code/MacApp/Sources/TimelineTracksView.swift new file mode 100644 index 0000000..0ccb405 --- /dev/null +++ b/code/MacApp/Sources/TimelineTracksView.swift @@ -0,0 +1,506 @@ +import SwiftUI + +/// Colors shared by the timeline and the inspector, so a block and its row in +/// the list always read as the same thing. +enum EmphasisPalette { + static func color(_ level: Int) -> Color { + switch level { + case 1: return Color.blue + case 2: return Color.orange + case 3: return Color.pink + default: return Color.secondary + } + } + + static func label(_ level: Int) -> String { + switch level { + case 1: return "Leve" + case 2: return "Média" + case 3: return "Forte" + default: return "Sem" + } + } + + static func speakerColor(_ speaker: String, among speakers: [String]) -> Color { + let palette: [Color] = [.teal, .purple, .green, .indigo, .brown, .cyan] + guard let index = speakers.firstIndex(of: speaker) else { return .gray } + return palette[index % palette.count] + } +} + +/// The timeline strip: four stacked tracks over one shared time axis. +/// +/// Phrases are laid out as real views rather than drawn into a Canvas, because +/// every one of them is a target — click to select, drag its edge to trim, +/// right-click to change emphasis. The dense per-word energy track *is* a +/// Canvas: it has thousands of bars and nothing to hit. +struct TimelineTracksView: View { + @ObservedObject var model: PhraseReviewModel + + private let rulerHeight: CGFloat = 18 + private let phraseHeight: CGFloat = 46 + private let energyHeight: CGFloat = 34 + private let stripHeight: CGFloat = 12 + private let handleWidth: CGFloat = 8 + + private let gutterWidth: CGFloat = 92 + private let trackSpacing: CGFloat = 4 + + private var pps: CGFloat { CGFloat(model.pixelsPerSecond) } + private var contentWidth: CGFloat { max(320, CGFloat(model.duration) * pps) } + + /// Name, icon and height of each lane, in the order they stack. The gutter + /// and the tracks are built from this one list so a label can never drift + /// off the lane it names. + private var lanes: [(label: String, icon: String, height: CGFloat)] { + [ + ("", "", rulerHeight), + ("Zooms", "plus.magnifyingglass", stripHeight + 6), + ("Frases", "text.quote", phraseHeight), + ("Energia", "waveform", energyHeight), + ("Emoção", "face.smiling", stripHeight), + ("Locutor", "person.wave.2", stripHeight), + ("Roteiro", "list.bullet.rectangle", stripHeight), + ] + } + + var body: some View { + VStack(spacing: 0) { + toolbar + Divider() + HStack(alignment: .top, spacing: 0) { + gutter + Divider() + timelineScroller + } + } + .background(Color(nsColor: .underPageBackgroundColor)) + } + + /// Fixed column naming each lane. Without it the stripes are six colours + /// with no way to tell which one is emotion and which one is the speaker. + private var gutter: some View { + VStack(alignment: .leading, spacing: trackSpacing) { + ForEach(lanes.indices, id: \.self) { index in + let lane = lanes[index] + HStack(spacing: 4) { + if !lane.icon.isEmpty { + Image(systemName: lane.icon).font(.system(size: 9)) + } + Text(lane.label).font(.system(size: 10)) + Spacer(minLength: 0) + } + .foregroundStyle(.secondary) + .frame(height: lane.height, alignment: .center) + } + } + .padding(.horizontal, 8) + .padding(.vertical, 8) + .frame(width: gutterWidth, alignment: .leading) + } + + private var timelineScroller: some View { + ScrollViewReader { proxy in + ScrollView([.horizontal]) { + ZStack(alignment: .topLeading) { + VStack(alignment: .leading, spacing: trackSpacing) { + ruler + zoomTrack + phraseTrack + energyTrack + emotionTrack + speakerTrack + scriptTrack + } + .frame(width: contentWidth, alignment: .leading) + rangeOverlay + playhead + // Anchors the auto-scroll: one invisible marker per + // phrase, so selecting a line off-screen brings it in. + ForEach(model.phrases) { phrase in + Color.clear + .frame(width: 1, height: 1) + .offset(x: x(phrase.start)) + .id(phrase.id) + } + } + .padding(.vertical, 8) + .contentShape(Rectangle()) + .gesture(scrubGesture) + .contextMenu { timelineMenu } + } + .onChange(of: model.selection) { _, newValue in + guard let newValue else { return } + withAnimation(.easeOut(duration: 0.2)) { + proxy.scrollTo(newValue, anchor: .center) + } + } + } + } + + // MARK: - Barra de controles + + private var toolbar: some View { + HStack(spacing: 12) { + Button { + model.togglePlay() + } label: { + Image(systemName: model.isPlaying ? "pause.fill" : "play.fill") + } + .buttonStyle(.borderless) + .help("Reproduzir (espaço)") + .disabled(model.player == nil) + + Text(timecode(model.currentTime)) + .font(.system(.caption, design: .monospaced)) + .foregroundStyle(.secondary) + + Button { + model.playSelectedPhrase() + } label: { + Image(systemName: "play.rectangle") + } + .buttonStyle(.borderless) + .help("Tocar só a frase selecionada (⏎)") + .disabled(model.player == nil || model.selection == nil) + + Toggle("Pular removidos", isOn: $model.skipRemoved) + .toggleStyle(.checkbox) + .font(.caption) + .help("Durante a reprodução, salta os trechos desativados — mostra como o corte ficou.") + + Button { + model.addZoomForRange() + } label: { + Label("Zoom no trecho", systemImage: "plus.magnifyingglass") + } + .buttonStyle(.borderless) + .font(.caption) + .disabled(!model.hasRange) + .help("Arraste na timeline para marcar um trecho e crie um zoom nele. A escala vem de Análise de Voz.") + + Spacer() + + legend + + Spacer() + + Image(systemName: "minus.magnifyingglass").foregroundStyle(.secondary) + Slider(value: $model.pixelsPerSecond, + in: model.minPixelsPerSecond...model.maxPixelsPerSecond) + .frame(width: 130) + Image(systemName: "plus.magnifyingglass").foregroundStyle(.secondary) + } + .padding(.horizontal, 12) + .padding(.vertical, 8) + } + + private var legend: some View { + HStack(spacing: 10) { + ForEach(0..<4, id: \.self) { level in + HStack(spacing: 4) { + RoundedRectangle(cornerRadius: 2) + .fill(EmphasisPalette.color(level)) + .frame(width: 10, height: 10) + Text(EmphasisPalette.label(level)).font(.caption2) + } + } + } + .foregroundStyle(.secondary) + } + + // MARK: - Trilhas + + private var ruler: some View { + Canvas { context, size in + let step = tickStep() + var time = 0.0 + while time <= model.duration { + let position = x(time) + context.stroke( + Path { $0.move(to: CGPoint(x: position, y: size.height - 6)) + $0.addLine(to: CGPoint(x: position, y: size.height)) }, + with: .color(.secondary.opacity(0.5)) + ) + context.draw( + Text(timecode(time)).font(.system(size: 9, design: .monospaced)) + .foregroundColor(.secondary), + at: CGPoint(x: position + 18, y: 6) + ) + time += step + } + } + .frame(width: contentWidth, height: rulerHeight) + } + + private var phraseTrack: some View { + ZStack(alignment: .topLeading) { + RoundedRectangle(cornerRadius: 4) + .fill(Color.secondary.opacity(0.06)) + .frame(width: contentWidth, height: phraseHeight) + ForEach(model.phrases) { phrase in + phraseBlock(phrase) + } + } + .frame(width: contentWidth, height: phraseHeight, alignment: .topLeading) + } + + @ViewBuilder + private func phraseBlock(_ phrase: ReviewPhrase) -> some View { + let isSelected = model.selection == phrase.id + let color = EmphasisPalette.color(phrase.emphasis) + let fullWidth = max(2, width(from: phrase.start, to: phrase.end)) + let keptWidth = max(1, width(from: phrase.trimStart, to: phrase.trimEnd)) + + ZStack(alignment: .topLeading) { + // The whole line, dim — what is there before the edit. + RoundedRectangle(cornerRadius: 4) + .fill(color.opacity(phrase.active ? 0.15 : 0.10)) + .frame(width: fullWidth, height: phraseHeight) + + // What survives: the kept span, drawn solid over it. + RoundedRectangle(cornerRadius: 4) + .fill(color.opacity(phrase.active ? 0.55 : 0.12)) + .frame(width: keptWidth, height: phraseHeight) + .offset(x: width(from: phrase.start, to: phrase.trimStart)) + + Text(phrase.text) + .font(.system(size: 10)) + .lineLimit(2) + .padding(.horizontal, 4) + .frame(width: fullWidth, height: phraseHeight, alignment: .topLeading) + .foregroundStyle(phrase.active ? .primary : .secondary) + .strikethrough(!phrase.active) + + RoundedRectangle(cornerRadius: 4) + .stroke(isSelected ? Color.accentColor : color.opacity(0.4), + lineWidth: isSelected ? 2 : 1) + .frame(width: fullWidth, height: phraseHeight) + + if isSelected && phrase.active { + trimHandle(phrase, edge: .start) + trimHandle(phrase, edge: .end) + } + } + .frame(width: fullWidth, height: phraseHeight, alignment: .topLeading) + .offset(x: x(phrase.start)) + .contentShape(Rectangle()) + .onTapGesture { model.goTo(phraseID: phrase.id) } + .contextMenu { phraseMenu(phrase) } + .help(phrase.reason.isEmpty ? phrase.text : "\(phrase.text)\n— \(phrase.reason)") + } + + private func trimHandle(_ phrase: ReviewPhrase, edge: TrimEdge) -> some View { + let offset = edge == .start + ? width(from: phrase.start, to: phrase.trimStart) + : width(from: phrase.start, to: phrase.trimEnd) - handleWidth + return RoundedRectangle(cornerRadius: 2) + .fill(Color.accentColor) + .frame(width: handleWidth, height: phraseHeight) + .offset(x: offset) + .gesture( + DragGesture(minimumDistance: 1) + .onChanged { value in + let time = phrase.start + Double((value.location.x) / pps) + model.trim(phrase.id, edge: edge, to: time) + } + ) + .help(edge == .start ? "Arraste para cortar o começo (pula de palavra em palavra)" + : "Arraste para cortar o fim (pula de palavra em palavra)") + } + + @ViewBuilder + private func phraseMenu(_ phrase: ReviewPhrase) -> some View { + Button("Tocar esta frase") { + model.goTo(phraseID: phrase.id) + model.playSelectedPhrase() + } + Button(phrase.active ? "Remover do corte" : "Trazer de volta") { + model.toggleActive(phrase.id) + } + Button("Adicionar zoom nesta frase") { model.addZoomForPhrase(phrase.id) } + Divider() + ForEach(0..<4, id: \.self) { level in + Button("Ênfase: \(EmphasisPalette.label(level))") { + model.setEmphasis(level, for: phrase.id) + } + } + Divider() + Button(phrase.isBackstage ? "Marcar como roteiro" : "Marcar como bastidor") { + model.setTrack(phrase.isBackstage ? ReviewPhrase.trackScript : ReviewPhrase.trackBackstage, + for: phrase.id) + } + if phrase.isTrimmed { + Divider() + Button("Desfazer corte da frase") { model.resetTrim(phrase.id) } + } + } + + /// Per-word energy/emphasis, straight from the voice timeline — the closest + /// thing to a waveform without opening the audio again. + private var energyTrack: some View { + Canvas { context, size in + for phrase in model.phrases { + for word in phrase.words { + let start = x(word.start) + let barWidth = max(1, width(from: word.start, to: word.end) - 1) + let height = size.height * CGFloat(max(0.04, word.energy)) + let rect = CGRect(x: start, y: size.height - height, + width: barWidth, height: height) + let color = word.emphasis >= 0.65 ? Color.pink + : word.emphasis >= 0.45 ? Color.orange + : Color.secondary + context.fill(Path(rect), + with: .color(color.opacity(phrase.active ? 0.6 : 0.2))) + } + } + } + .frame(width: contentWidth, height: energyHeight) + .background(RoundedRectangle(cornerRadius: 4).fill(Color.secondary.opacity(0.06))) + } + + private var speakerTrack: some View { + stripTrack { phrase in + EmphasisPalette.speakerColor(phrase.speaker, among: model.speakers) + } + } + + private var scriptTrack: some View { + stripTrack { phrase in phrase.isBackstage ? Color.gray : Color.mint } + } + + private func stripTrack(_ color: @escaping (ReviewPhrase) -> Color) -> some View { + Canvas { context, size in + for phrase in model.phrases { + let rect = CGRect(x: x(phrase.start), y: 0, + width: max(1, width(from: phrase.start, to: phrase.end)), + height: size.height) + context.fill(Path(roundedRect: rect, cornerRadius: 2), + with: .color(color(phrase).opacity(phrase.active ? 0.7 : 0.2))) + } + } + .frame(width: contentWidth, height: stripHeight) + } + + private var playhead: some View { + Rectangle() + .fill(Color.red) + .frame(width: 1.5) + .offset(x: x(model.currentTime)) + .allowsHitTesting(false) + } + + /// One gesture, two meanings, decided by whether the mouse moved: a click + /// parks the playhead, a drag marks in/out. Splitting them across separate + /// controls would mean choosing a tool before every action, which is + /// exactly the ceremony this screen is meant to avoid. + private var scrubGesture: some Gesture { + DragGesture(minimumDistance: 0) + .onChanged { value in + let from = Double(value.startLocation.x / pps) + let to = Double(value.location.x / pps) + if abs(value.translation.width) > 3 { + model.setRange(from: from, to: to) + model.seek(to: min(from, to)) + } else { + model.clearRange() + model.seek(to: to) + } + } + } + + /// The marked in/out, drawn over every track so the span reads against the + /// phrases and the energy at once. + private var rangeOverlay: some View { + Group { + if let span = model.rangeSpan { + Rectangle() + .fill(Color.accentColor.opacity(0.18)) + .overlay(Rectangle().stroke(Color.accentColor.opacity(0.6), lineWidth: 1)) + .frame(width: max(1, width(from: span.start, to: span.end))) + .offset(x: x(span.start)) + .allowsHitTesting(false) + } + } + } + + @ViewBuilder + private var timelineMenu: some View { + if model.hasRange, let span = model.rangeSpan { + Button("Adicionar zoom no trecho (\(secondsLabel(span.end - span.start)))") { + model.addZoomForRange() + } + Button("Tocar o trecho") { model.playRange(from: span.start, to: span.end) } + Button("Limpar seleção") { model.clearRange() } + } else { + Text("Arraste na timeline para marcar um trecho") + } + if let zoom = model.zoom(at: model.currentTime) { + Divider() + Button("Remover o zoom daqui") { model.removeZoom(zoom.id) } + } + } + + private func secondsLabel(_ seconds: Double) -> String { + String(format: "%.1fs", seconds) + } + + /// Punch-ins, on their own lane above the script: they are a second layer + /// over the same time, not a property of a phrase. + private var zoomTrack: some View { + ZStack(alignment: .topLeading) { + RoundedRectangle(cornerRadius: 3) + .fill(Color.secondary.opacity(0.06)) + .frame(width: contentWidth, height: stripHeight + 6) + ForEach(model.zooms) { zoom in + RoundedRectangle(cornerRadius: 3) + .fill(Color.yellow.opacity(0.55)) + .overlay( + Image(systemName: "plus.magnifyingglass") + .font(.system(size: 8)).foregroundStyle(.black.opacity(0.6)) + ) + .frame(width: max(6, width(from: zoom.start, to: zoom.end)), + height: stripHeight + 6) + .offset(x: x(zoom.start)) + .help("Zoom marcado — \(secondsLabel(zoom.end - zoom.start)). A escala vem de Análise de Voz.") + .contextMenu { + Button("Remover este zoom") { model.removeZoom(zoom.id) } + } + } + } + .frame(width: contentWidth, height: stripHeight + 6, alignment: .topLeading) + } + + /// Delivery emotion per phrase — the fourth signal to read against the text. + private var emotionTrack: some View { + stripTrack { phrase in + switch phrase.emotion { + case "excited": return .orange + case "tense": return .red + case "calm": return .blue + case "reflective": return .purple + default: return .secondary + } + } + } + + // MARK: - Escala + + private func x(_ time: Double) -> CGFloat { CGFloat(time) * pps } + + private func width(from: Double, to: Double) -> CGFloat { + max(0, CGFloat(to - from) * pps) + } + + /// Ruler spacing that keeps labels ~80pt apart at any zoom. + private func tickStep() -> Double { + let candidates: [Double] = [1, 2, 5, 10, 15, 30, 60, 120, 300, 600] + let wanted = 80 / Double(pps) + return candidates.first { $0 >= wanted } ?? 600 + } + + private func timecode(_ seconds: Double) -> String { + let total = Int(seconds.rounded(.down)) + return String(format: "%02d:%02d", total / 60, total % 60) + } +} diff --git a/code/MacApp/Sources/TranscriptionView.swift b/code/MacApp/Sources/TranscriptionView.swift index feee256..5770df4 100644 --- a/code/MacApp/Sources/TranscriptionView.swift +++ b/code/MacApp/Sources/TranscriptionView.swift @@ -271,7 +271,7 @@ struct TranscriptionView: View { Toggle("Marcar o que foi dito na timeline", isOn: $batchMarkers) Divider() - Toggle("Exportar legendas SRT", isOn: $batchSubtitles) + Toggle("Gerar legenda comum (texto editável no FCP)", isOn: $batchSubtitles) Divider() batchOptionRow( @@ -793,7 +793,7 @@ struct TranscriptionView: View { if batchFillers { operations.append("remove_filler_words") } if batchPhrases { operations.append("edit_by_transcript") } if batchMarkers { operations.append("transcript_markers") } - if batchSubtitles { operations.append("export_srt") } + if batchSubtitles { operations.append("generate_plain_subtitles") } // Runs last, on the timing already cut by any earlier steps (see the // "Abrir no Final Cut Pro" fallback chain and exportSubtitles()'s own // preference for `processedPath` — same reasoning). @@ -834,7 +834,7 @@ struct TranscriptionView: View { } let nextPath = result?["path"] as? String ?? currentPath if operation == "remove_silences" { processedPath = nextPath } - if operation == "export_srt" { subtitlePaths = result?["paths"] as? [String] ?? [] } + if operation == "generate_plain_subtitles" { subtitlePaths = [nextPath] } if operation == "generate_dynamic_subtitles" { dynamicSubtitlesPath = nextPath } processBatchStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder) } diff --git a/code/MacApp/Sources/VoiceAnalysisView.swift b/code/MacApp/Sources/VoiceAnalysisView.swift index 51b3842..f072911 100644 --- a/code/MacApp/Sources/VoiceAnalysisView.swift +++ b/code/MacApp/Sources/VoiceAnalysisView.swift @@ -23,6 +23,7 @@ struct VoiceAnalysisView: View { } else { energySection emphasisSection + zoomSection weightsSection emotionSection resetSection @@ -90,6 +91,44 @@ struct VoiceAnalysisView: View { } } + private var zoomSection: some View { + Section { + sliderRow( + title: "Zoom na ênfase", + value: $config.zoomScale, + range: 1.0...3.0, + readout: "\(Int(config.zoomScale * 100))%", + help: "Fator aplicado nos punch-ins de ênfase. 130% equivale a escala 1,30 no Final Cut." + ) + Picker("Movimento", selection: $config.zoomMode) { + Text("Zoom in e out").tag("in_out") + Text("Só zoom in").tag("in") + Text("Só zoom out").tag("out") + } + .onChange(of: config.zoomMode) { _, _ in save() } + sliderRow( + title: "Velocidade do zoom in", + value: $config.zoomEaseIn, + range: 0.05...2.0, + readout: String(format: "%.2fs", config.zoomEaseIn), + help: "Duração da entrada do zoom. Menor é mais rápido." + ) + sliderRow( + title: "Velocidade do zoom out", + value: $config.zoomEaseOut, + range: 0.01...2.0, + readout: String(format: "%.2fs", config.zoomEaseOut), + help: "Duração da saída do zoom. Menor é mais seco." + ) + } header: { + Text("Zoom de Ênfase") + } footer: { + Text("Esses valores viram o padrão para ações de zoom que não trouxerem scale/ease/ease_out no JSON da edição por voz.") + .font(.caption) + .foregroundStyle(.secondary) + } + } + // MARK: - Emoção private var emotionSection: some View { @@ -129,13 +168,14 @@ struct VoiceAnalysisView: View { title: String, value: Binding, range: ClosedRange = 0...1, + readout: String? = nil, help: String? = nil ) -> some View { VStack(alignment: .leading, spacing: 2) { HStack { Text(title) Spacer() - Text(String(format: "%.2f", value.wrappedValue)) + Text(readout ?? String(format: "%.2f", value.wrappedValue)) .monospacedDigit() .foregroundStyle(.secondary) } @@ -188,6 +228,10 @@ struct VoiceAnalysisConfig { var weightDuration: Double var emotionEnabled: Bool var emotionSensitivity: Double + var zoomScale: Double + var zoomMode: String + var zoomEaseIn: Double + var zoomEaseOut: Double static let defaults = VoiceAnalysisConfig( energyThreshold: 0.5, @@ -198,7 +242,11 @@ struct VoiceAnalysisConfig { weightPause: 0.15, weightDuration: 0.10, emotionEnabled: false, - emotionSensitivity: 0.5 + emotionSensitivity: 0.5, + zoomScale: 1.30, + zoomMode: "in_out", + zoomEaseIn: 0.25, + zoomEaseOut: 0.04 ) init( @@ -210,7 +258,11 @@ struct VoiceAnalysisConfig { weightPause: Double, weightDuration: Double, emotionEnabled: Bool, - emotionSensitivity: Double + emotionSensitivity: Double, + zoomScale: Double, + zoomMode: String, + zoomEaseIn: Double, + zoomEaseOut: Double ) { self.energyThreshold = energyThreshold self.emphasisThreshold = emphasisThreshold @@ -221,6 +273,10 @@ struct VoiceAnalysisConfig { self.weightDuration = weightDuration self.emotionEnabled = emotionEnabled self.emotionSensitivity = emotionSensitivity + self.zoomScale = zoomScale + self.zoomMode = zoomMode + self.zoomEaseIn = zoomEaseIn + self.zoomEaseOut = zoomEaseOut } /// Lê a resposta do bridge, caindo no padrão para qualquer campo ausente. @@ -236,7 +292,11 @@ struct VoiceAnalysisConfig { weightPause: weights["pause_before"] as? Double ?? defaults.weightPause, weightDuration: weights["duration"] as? Double ?? defaults.weightDuration, emotionEnabled: json["emotion_enabled"] as? Bool ?? defaults.emotionEnabled, - emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity + emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity, + zoomScale: json["zoom_scale"] as? Double ?? defaults.zoomScale, + zoomMode: json["zoom_mode"] as? String ?? defaults.zoomMode, + zoomEaseIn: json["zoom_ease_in"] as? Double ?? defaults.zoomEaseIn, + zoomEaseOut: json["zoom_ease_out"] as? Double ?? defaults.zoomEaseOut ) } @@ -253,6 +313,10 @@ struct VoiceAnalysisConfig { ], "emotion_enabled": emotionEnabled, "emotion_sensitivity": emotionSensitivity, + "zoom_scale": zoomScale, + "zoom_mode": zoomMode, + "zoom_ease_in": zoomEaseIn, + "zoom_ease_out": zoomEaseOut, ] } } diff --git a/code/MacApp/Sources/WizardView.swift b/code/MacApp/Sources/WizardView.swift new file mode 100644 index 0000000..649b6c5 --- /dev/null +++ b/code/MacApp/Sources/WizardView.swift @@ -0,0 +1,808 @@ +import SwiftUI +import AppKit + +/// Guia passo a passo do fluxo completo: projeto → transcrição → análise de +/// voz → copiar para o chat e trazer as decisões → revisar as ênfases → +/// processamento final. Existe para que o usuário não precise entender a ordem +/// certa de botões espalhados em várias abas — cada etapa só libera a próxima +/// quando o passo anterior terminou, e a "ponte" com o chat (que hoje exigia +/// sair do app e escolher um arquivo na mão) vira copiar/colar assistido +/// dentro da própria tela. +enum WizardStep: Int, CaseIterable, Identifiable { + case projeto, transcricao, analise, exportarChat, revisar, finalizar, concluido + var id: Int { rawValue } + + var titulo: String { + switch self { + case .projeto: return "Projeto" + case .transcricao: return "Transcrever" + case .analise: return "Analisar voz" + case .exportarChat: return "Decisões da IA" + case .revisar: return "Revisar ênfases" + case .finalizar: return "Processar" + case .concluido: return "Concluído" + } + } +} + +struct WizardView: View { + @State private var step: WizardStep = .projeto + + // Passo 1 — projeto + @State private var outputFolder: String? + @State private var projectPath: String? + @State private var catalog: Catalog? + + // Passo 2 — transcrição + @State private var isTranscribing = false + @State private var transcribeProgress: Double = 0 + @State private var transcribeStage = "" + @State private var transcribeResults: [TranscriptResult] = [] + + // Passo 3 — análise de voz + @State private var isAnalyzing = false + @State private var voiceTimelinePath: String? + @State private var voiceAnalysisMessage = "" + @State private var acousticsAvailable: Bool? + @State private var showVoiceTimelineReuseAlert = false + @State private var existingVoiceTimelinePath: String? + + // Passo 4 — enviar ao chat e trazer as decisões de volta + @State private var copiedFeedback = "" + @State private var decisionsText = "" + @State private var isApplyingDecisions = false + @State private var appliedPath: String? + @State private var skippedVoiceEdit = false + + // Passo 5 — revisar ênfases + @StateObject private var reviewModel = PhraseReviewModel() + @State private var reviewLoadedFor: String? + @State private var phraseReviewPath: String? + + // Passo 6 — processamento final + @State private var finalSilences = true + @State private var finalFillers = false + @State private var finalSubtitles = true + @State private var finalDynamicSubtitles = false + @State private var isFinalizing = false + @State private var finalStatus = "" + @State private var finalPath: String? + + @State private var errorMessage: String? + + var body: some View { + VStack(spacing: 0) { + stepperHeader + .padding(.horizontal, 24) + .padding(.top, 20) + .padding(.bottom, 16) + + Divider() + + // A revisão é uma sala de edição, não um formulário: ela precisa da + // largura toda e rola por conta própria (timeline horizontal, lista + // vertical). As demais etapas continuam na coluna estreita, que é o + // que mantém um passo a passo legível. + if step == .revisar { + revisarStep + } else { + ScrollView { + VStack(alignment: .leading, spacing: 18) { + if let errorMessage, !errorMessage.isEmpty { + Label(errorMessage, systemImage: "exclamationmark.triangle.fill") + .foregroundStyle(.red) + .padding(.top, 4) + } + content + } + .padding(24) + .frame(maxWidth: 640, alignment: .leading) + .frame(maxWidth: .infinity) + } + } + + Divider() + navFooter + .padding(.horizontal, 24) + .padding(.vertical, 16) + } + .task { + loadProjectConfig() + await loadCatalog() + } + .alert("Análise de voz já existe", isPresented: $showVoiceTimelineReuseAlert) { + Button("Usar existente") { + if let existingVoiceTimelinePath { + voiceTimelinePath = existingVoiceTimelinePath + voiceAnalysisMessage = "Reaproveitando análise existente: \(existingVoiceTimelinePath)" + } + } + Button("Reprocessar") { + analyzeVoice(forceReprocess: true) + } + Button("Cancelar", role: .cancel) {} + } message: { + Text("Já existe um arquivo voice_timeline para este projeto. Quer manter o processamento anterior para ganhar tempo?") + } + } + + // MARK: - Cabeçalho com os passos + + private var stepperHeader: some View { + HStack(spacing: 6) { + ForEach(WizardStep.allCases) { s in + HStack(spacing: 6) { + ZStack { + Circle() + .fill(colorFor(s)) + .frame(width: 24, height: 24) + if s.rawValue < step.rawValue { + Image(systemName: "checkmark") + .font(.caption2.weight(.bold)) + .foregroundStyle(.white) + } else { + Text("\(s.rawValue + 1)") + .font(.caption2.weight(.bold)) + .foregroundStyle(s == step ? .white : .secondary) + } + } + Text(s.titulo) + .font(.caption) + .foregroundStyle(s == step ? .primary : .secondary) + .fontWeight(s == step ? .semibold : .regular) + } + if s != WizardStep.allCases.last { + Rectangle() + .fill(s.rawValue < step.rawValue ? Color.accentColor : Color.secondary.opacity(0.25)) + .frame(height: 2) + .frame(maxWidth: .infinity) + } + } + } + } + + private func colorFor(_ s: WizardStep) -> Color { + if s.rawValue < step.rawValue { return .accentColor } + if s == step { return .accentColor } + return Color.secondary.opacity(0.25) + } + + // MARK: - Conteúdo por etapa + + @ViewBuilder + private var content: some View { + switch step { + case .projeto: projetoStep + case .transcricao: transcricaoStep + case .analise: analiseStep + case .exportarChat: exportarChatStep + case .revisar: revisarStep + case .finalizar: finalizarStep + case .concluido: concluidoStep + } + } + + private var projetoStep: some View { + VStack(alignment: .leading, spacing: 16) { + Text("1. Escolha o projeto").font(.title3.weight(.semibold)) + Text("A pasta é onde tudo o que for gerado nesse fluxo fica salvo. O arquivo é o .fcpxml exportado do Final Cut Pro.") + .font(.callout).foregroundStyle(.secondary) + + fieldRow(icon: "folder", label: outputFolder ?? "Nenhuma pasta selecionada", isSet: outputFolder != nil) { + pickOutputFolder() + } + fieldRow(icon: "doc.text", label: projectPath.map { URL(fileURLWithPath: $0).lastPathComponent } ?? "Nenhum arquivo selecionado", isSet: projectPath != nil) { + pickProjectFile() + } + + if looksLikeGeneratedFile(projectPath) { + Label("Esse arquivo parece já ter sido processado por este fluxo (o nome tem um sufixo como \"_voice_edit\" ou \"_silence_removed\"). Rodar o wizard de novo em cima dele reaplica os cortes por cima de cortes já feitos. Selecione o .fcpxml original do Final Cut, a menos que a intenção seja mesmo reprocessar.", + systemImage: "exclamationmark.triangle.fill") + .font(.caption).foregroundStyle(.orange) + } + + if (catalog?.installedCount ?? 0) == 0 { + Label("Nenhum modelo de transcrição instalado. Baixe um na aba \"Modelos\" antes de continuar.", + systemImage: "exclamationmark.triangle.fill") + .font(.caption).foregroundStyle(.orange) + } + } + } + + private var transcricaoStep: some View { + VStack(alignment: .leading, spacing: 16) { + Text("2. Transcreva o áudio").font(.title3.weight(.semibold)) + Text("Roda localmente com o modelo escolhido na aba Modelos. Vira a base de tudo que vem depois — o corte por voz, as legendas, os marcadores.") + .font(.callout).foregroundStyle(.secondary) + + Button { + startTranscription() + } label: { + if isTranscribing { + HStack { ProgressView().controlSize(.small); Text(transcribeStage.isEmpty ? "Transcrevendo…" : transcribeStage) } + .frame(maxWidth: .infinity) + } else { + Label(transcribeResults.isEmpty ? "Transcrever" : "Transcrever novamente", systemImage: "waveform") + .frame(maxWidth: .infinity) + } + } + .buttonStyle(.borderedProminent) + .controlSize(.large) + .disabled(isTranscribing || projectPath == nil || outputFolder == nil) + + if isTranscribing { + VStack(alignment: .leading, spacing: 6) { + ProgressView(value: transcribeProgress) + Text("\(Int(transcribeProgress * 100))%").font(.caption).foregroundStyle(.secondary).monospacedDigit() + } + } + + if !transcribeResults.isEmpty { + ForEach(transcribeResults, id: \.media) { r in + VStack(alignment: .leading, spacing: 4) { + HStack { + Image(systemName: "checkmark.circle.fill").foregroundStyle(.green) + Text(r.media).font(.body.weight(.medium)) + Spacer() + Text("\(r.language) · \(r.words) palavras").font(.caption).foregroundStyle(.secondary) + } + Text(r.preview).font(.caption).foregroundStyle(.secondary).lineLimit(2) + } + .padding(12) + .background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06))) + } + } + } + } + + private var analiseStep: some View { + VStack(alignment: .leading, spacing: 16) { + Text("3. Analise a voz").font(.title3.weight(.semibold)) + Text("Gera o JSON com transcrição, locutor e intensidade (pitch/energia/ritmo) por palavra — é esse arquivo que o chat lê para decidir o que cortar. Não corta nada sozinho.") + .font(.callout).foregroundStyle(.secondary) + + Button { + analyzeVoice() + } label: { + if isAnalyzing { + HStack { ProgressView().controlSize(.small); Text("Analisando…") }.frame(maxWidth: .infinity) + } else { + Label(voiceTimelinePath == nil ? "Analisar voz" : "Analisar novamente", systemImage: "waveform.badge.magnifyingglass") + .frame(maxWidth: .infinity) + } + } + .buttonStyle(.borderedProminent) + .controlSize(.large) + .disabled(isAnalyzing || projectPath == nil || outputFolder == nil) + + if let voiceTimelinePath { + VStack(alignment: .leading, spacing: 6) { + Label("Análise pronta", systemImage: "checkmark.circle.fill").foregroundStyle(.green) + Text(voiceTimelinePath).font(.caption).foregroundStyle(.secondary).lineLimit(1).truncationMode(.middle) + } + .padding(12) + .background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06))) + + if acousticsAvailable == false { + VStack(alignment: .leading, spacing: 4) { + Label("Sem análise acústica real", systemImage: "exclamationmark.triangle.fill") + .font(.caption.weight(.semibold)).foregroundStyle(.orange) + Text("Falta o componente \"librosa\" — os cortes ainda são decididos pelo texto, mas o chat não vai propor zoom com confiança. Instale em Avançado → Modelos → \"Análise Acústica\", e refaça esta etapa depois.") + .font(.caption).foregroundStyle(.secondary) + } + .padding(12) + .background(RoundedRectangle(cornerRadius: 8).fill(Color.orange.opacity(0.08))) + } + } + } + } + + private var exportarChatStep: some View { + VStack(alignment: .leading, spacing: 16) { + Text("4. Envie para o chat decidir os cortes").font(.title3.weight(.semibold)) + Text("Esta é a única etapa manual que sobra: o julgamento de qual tomada usar, onde dar zoom e o que escrever na tela é feito pela IA numa conversa, não por um botão. Copie abaixo, cole numa sessão do Claude e peça pra rodar a skill \"editar-por-voz\".") + .font(.callout).foregroundStyle(.secondary) + + if let voiceTimelinePath { + Button { + copyForChat(path: voiceTimelinePath) + } label: { + Label("Copiar para colar no chat", systemImage: "doc.on.clipboard") + .frame(maxWidth: .infinity) + } + .buttonStyle(.borderedProminent) + .controlSize(.large) + + if !copiedFeedback.isEmpty { + Label(copiedFeedback, systemImage: "checkmark.circle.fill") + .font(.caption).foregroundStyle(.green) + } + + VStack(alignment: .leading, spacing: 8) { + Text("O que é copiado").font(.caption.weight(.semibold)).foregroundStyle(.secondary) + Text("Um pedido pronto + o conteúdo de \(URL(fileURLWithPath: voiceTimelinePath).lastPathComponent), já formatado. É só colar (⌘V) numa conversa com o Claude.") + .font(.caption).foregroundStyle(.secondary) + } + .padding(12) + .background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06))) + + Divider().padding(.vertical, 4) + + Text("Cole aqui o que o chat devolveu").font(.callout.weight(.semibold)) + Text("Na próxima etapa essas decisões aparecem já marcadas na timeline, frase por frase, para você lapidar.") + .font(.caption).foregroundStyle(.secondary) + + HStack { + Button { + if let s = NSPasteboard.general.string(forType: .string) { + decisionsText = s + } + } label: { + Label("Colar da área de transferência", systemImage: "list.clipboard") + } + Spacer() + if !decisionsText.isEmpty { + Label(jsonIsValid ? "JSON válido" : "JSON inválido", + systemImage: jsonIsValid ? "checkmark.circle.fill" : "xmark.circle.fill") + .font(.caption) + .foregroundStyle(jsonIsValid ? .green : .red) + } + } + + TextEditor(text: $decisionsText) + .font(.system(.caption, design: .monospaced)) + .frame(minHeight: 140) + .padding(8) + .background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06))) + .overlay(RoundedRectangle(cornerRadius: 8).stroke(Color.secondary.opacity(0.2))) + + Button { + applyDecisions() + } label: { + if isApplyingDecisions { + HStack { ProgressView().controlSize(.small); Text("Aplicando…") } + .frame(maxWidth: .infinity) + } else { + Label("Aplicar decisões", systemImage: "checkmark.seal") + .frame(maxWidth: .infinity) + } + } + .buttonStyle(.borderedProminent) + .controlSize(.large) + .disabled(isApplyingDecisions || !jsonIsValid) + + if let appliedPath { + Label("Decisões aplicadas — \(URL(fileURLWithPath: appliedPath).lastPathComponent)", + systemImage: "checkmark.circle.fill") + .font(.caption).foregroundStyle(.green) + } + + Divider() + Button("Pular esta etapa (revisar as ênfases direto, sem passar pela IA)") { + skippedVoiceEdit = true + appliedPath = nil + decisionsText = "" + } + .buttonStyle(.plain) + .font(.caption) + .foregroundStyle(.secondary) + } else { + Label("Volte ao passo anterior e rode a análise de voz primeiro.", systemImage: "exclamationmark.triangle.fill") + .font(.caption).foregroundStyle(.orange) + } + } + } + + /// Etapa 5 — a sala de edição. Diferente das outras, não é um formulário + /// dentro da coluna do assistente: ocupa a janela toda e se carrega sozinha + /// na primeira vez que aparece para aquela análise de voz. + private var revisarStep: some View { + Group { + if voiceTimelinePath != nil { + PhraseReviewView(model: reviewModel) + } else { + VStack(spacing: 8) { + Label("Volte ao passo 3 e rode a análise de voz primeiro.", + systemImage: "exclamationmark.triangle.fill") + .foregroundStyle(.orange) + } + .frame(maxWidth: .infinity, maxHeight: .infinity) + } + } + .onAppear { loadReviewIfNeeded() } + } + + private var finalizarStep: some View { + VStack(alignment: .leading, spacing: 16) { + Text("6. Finalize o corte").font(.title3.weight(.semibold)) + Text("Últimos passos automáticos, sem decisão envolvida — rodam com os parâmetros já configurados na aba \"Análise de Voz\" / \"Legendas Dinâmicas\".") + .font(.callout).foregroundStyle(.secondary) + + Toggle("Remover silêncios do áudio", isOn: $finalSilences) + Toggle("Remover palavras de preenchimento", isOn: $finalFillers) + Toggle("Gerar legenda comum (texto editável no FCP)", isOn: $finalSubtitles) + Toggle("Gerar legendas dinâmicas (estilo configurado na aba própria)", isOn: $finalDynamicSubtitles) + + Button { + finalizeProcessing() + } label: { + if isFinalizing { + HStack { ProgressView().controlSize(.small); Text(finalStatus.isEmpty ? "Processando…" : finalStatus) } + .frame(maxWidth: .infinity) + } else { + Label("Processar", systemImage: "play.fill").frame(maxWidth: .infinity) + } + } + .buttonStyle(.borderedProminent) + .controlSize(.large) + .disabled(isFinalizing || (!finalSilences && !finalFillers && !finalSubtitles && !finalDynamicSubtitles)) + + if !finalStatus.isEmpty && !isFinalizing { + Text(finalStatus).font(.caption).foregroundStyle(.secondary) + } + } + } + + private var concluidoStep: some View { + VStack(alignment: .leading, spacing: 16) { + Label("Concluído", systemImage: "checkmark.seal.fill") + .font(.title3.weight(.semibold)) + .foregroundStyle(.green) + if let finalPath { + Text(finalPath).font(.caption).foregroundStyle(.secondary).lineLimit(1).truncationMode(.middle) + HStack { + Button("Abrir no Final Cut Pro") { NSWorkspace.shared.open(URL(fileURLWithPath: finalPath)) } + .buttonStyle(.borderedProminent) + Button("Mostrar no Finder") { + NSWorkspace.shared.activateFileViewerSelecting([URL(fileURLWithPath: finalPath)]) + } + } + } + Divider().padding(.vertical, 8) + Button("Começar outro projeto") { resetWizard() } + } + } + + // MARK: - Navegação + + private var navFooter: some View { + HStack { + if step != .projeto && step != .concluido { + Button("Voltar") { goBack() } + } + Spacer() + if step != .concluido { + Button(step == .finalizar ? "Concluir" : "Continuar") { goNext() } + .buttonStyle(.borderedProminent) + .disabled(!canAdvance) + } + } + } + + private var canAdvance: Bool { + switch step { + case .projeto: return outputFolder != nil && projectPath != nil + case .transcricao: return !transcribeResults.isEmpty + case .analise: return voiceTimelinePath != nil + case .exportarChat: return appliedPath != nil || skippedVoiceEdit + // Revisar é opcional: a sugestão da IA já é utilizável como veio, então + // o botão nunca trava aqui — o passo existe para lapidar, não para + // exigir mais uma confirmação. + case .revisar: return true + case .finalizar: return finalPath != nil && !isFinalizing + case .concluido: return false + } + } + + private func goNext() { + guard let next = WizardStep(rawValue: step.rawValue + 1) else { return } + // Sair da revisão grava o que foi decidido (e as ações derivadas dela) + // ao lado da análise de voz. Nada é renderizado aqui: a etapa 6 é que + // lê esse arquivo para dar zoom e legenda dinâmica só nas ênfases. + if step == .revisar { + reviewModel.save { path in + phraseReviewPath = path + } + } + step = next + } + + private func goBack() { + guard let prev = WizardStep(rawValue: step.rawValue - 1) else { return } + step = prev + } + + private func resetWizard() { + step = .projeto + transcribeResults = [] + voiceTimelinePath = nil + voiceAnalysisMessage = "" + decisionsText = "" + appliedPath = nil + skippedVoiceEdit = false + reviewLoadedFor = nil + phraseReviewPath = nil + finalStatus = "" + finalPath = nil + errorMessage = nil + } + + // MARK: - Componentes auxiliares + + @ViewBuilder + private func fieldRow(icon: String, label: String, isSet: Bool, action: @escaping () -> Void) -> some View { + HStack { + Image(systemName: icon).foregroundStyle(isSet ? .primary : .secondary) + Text(label).lineLimit(1).truncationMode(.middle).foregroundStyle(isSet ? .primary : .secondary) + Spacer() + Button("Escolher…", action: action) + } + .padding(12) + .background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06))) + } + + /// Todo output do fluxo carrega um destes sufixos no nome (ver + /// `_derived_output` / suffixes usados por `apply_voice_actions`, + /// `remove_silences`, `generate_dynamic_subtitles` em + /// `admin/models_api.py`). Selecionar um deles como "o projeto" no passo + /// 1 é o erro que gerou arquivos como `_voice_edit_voice_edit_...`: os + /// cortes de voz assumem timestamps da mídia ORIGINAL, então reaplicá-los + /// sobre um arquivo já cortado desloca tudo silenciosamente. + private static let generatedSuffixes = [ + "_voice_edit", "_silence_removed", "_dynamic_subtitles", + "_transcript_edit", "_fillers_removed", "_markers", + ] + + private func looksLikeGeneratedFile(_ path: String?) -> Bool { + guard let path else { return false } + let stem = URL(fileURLWithPath: path).deletingPathExtension().lastPathComponent + return Self.generatedSuffixes.contains { stem.contains($0) } + } + + private var jsonIsValid: Bool { + guard let data = decisionsText.data(using: .utf8), !decisionsText.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { return false } + return (try? JSONSerialization.jsonObject(with: data)) != nil + } + + // MARK: - Ações — Python bridge + + private func loadProjectConfig() { + PythonBridge.call(command: "project_config") { result, _ in + DispatchQueue.main.async { + guard let result, result["ok"] as? Bool == true else { return } + if let folder = result["folder"] as? String, !folder.isEmpty { outputFolder = folder } + if let file = result["file"] as? String, !file.isEmpty { projectPath = file } + } + } + } + + private func loadCatalog() async { + PythonBridge.call(command: "catalog") { result, _ in + DispatchQueue.main.async { + if let result { catalog = Catalog(json: result) } + } + } + } + + private func pickOutputFolder() { + let panel = NSOpenPanel() + panel.canChooseFiles = false + panel.canChooseDirectories = true + panel.allowsMultipleSelection = false + panel.prompt = "Usar esta pasta" + panel.message = "Escolha a pasta onde os resultados serão salvos." + if panel.runModal() == .OK, let url = panel.url { + outputFolder = url.path + PythonBridge.call(command: "set_project_config", arguments: ["folder": url.path]) { _, _ in } + } + } + + private func pickProjectFile() { + let panel = NSOpenPanel() + panel.canChooseFiles = true + panel.canChooseDirectories = false + panel.allowsMultipleSelection = false + panel.prompt = "Selecionar" + panel.message = "Selecione o arquivo (.fcpxml) ou o bundle (.fcpxmld) exportado pelo Final Cut Pro." + if panel.runModal() == .OK, let url = panel.url { + let ext = url.pathExtension.lowercased() + if ext == "fcpxml" || ext == "fcpxmld" || ext == "xml" { + projectPath = url.path + PythonBridge.call(command: "set_project_config", arguments: ["file": url.path]) { _, _ in } + } else { + errorMessage = "Selecione um arquivo .fcpxml, .fcpxmld ou .xml do Final Cut Pro." + } + } + } + + private func startTranscription() { + guard let projectPath, let outputFolder else { return } + isTranscribing = true + errorMessage = nil + transcribeResults = [] + transcribeProgress = 0 + PythonBridge.run(command: "transcribe", arguments: ["path": projectPath, "output_dir": outputFolder]) { obj in + DispatchQueue.main.async { + let type = obj["type"] as? String + if type == "progress" { + transcribeProgress = (obj["fraction"] as? NSNumber)?.doubleValue ?? 0 + transcribeStage = obj["stage"] as? String ?? "" + } else if type == "error" { + errorMessage = obj["message"] as? String ?? "Erro na transcrição." + } else if type == "result", let arr = obj["transcripts"] as? [[String: Any]] { + transcribeResults = arr.map(TranscriptResult.init) + } + } + } completion: { code, err in + DispatchQueue.main.async { + isTranscribing = false + transcribeProgress = 1 + if code != 0 && transcribeResults.isEmpty { + errorMessage = err ?? "A transcrição falhou." + } + } + } + } + + private func analyzeVoice(forceReprocess: Bool = false) { + guard let projectPath, let outputFolder else { return } + isAnalyzing = true + errorMessage = nil + PythonBridge.call(command: "analyze_voice", arguments: [ + "path": projectPath, + "output_dir": outputFolder, + "force_reprocess": forceReprocess, + ]) { result, err in + DispatchQueue.main.async { + isAnalyzing = false + guard result?["ok"] as? Bool == true else { + errorMessage = result?["error"] as? String ?? err ?? "Falha ao analisar a voz." + return + } + if result?["reused"] as? Bool == true, !forceReprocess { + let timelines = result?["timelines"] as? [String] ?? [] + existingVoiceTimelinePath = timelines.first ?? extractPath(from: result?["message"] as? String ?? "", marker: "**Timeline JSON**:") + showVoiceTimelineReuseAlert = true + return + } + let message = result?["message"] as? String ?? "" + voiceAnalysisMessage = message + if let path = extractPath(from: message, marker: "**Timeline JSON**:") { + voiceTimelinePath = path + } else { + voiceTimelinePath = nil + // ok:true não garante que a análise gerou timeline — se + // não houver fala detectável no áudio, o Python volta com + // sucesso mas sem "Timeline JSON" na mensagem. Sem isso + // aqui, a etapa parecia não fazer nada. + errorMessage = "A análise terminou mas não encontrou fala reconhecível no áudio. Mensagem do motor: " + (message.isEmpty ? "(vazia)" : message) + } + checkAcoustics() + } + } + } + + /// A ênfase de voz (energia/tom) depende do `librosa`, dependência + /// opcional. Sem ela, a análise ainda transcreve e corta pelo texto, + /// mas nunca deveria propor zoom — por isso avisamos aqui, no ponto + /// onde o usuário sentiria falta, em vez de só na aba Modelos. + private func checkAcoustics() { + PythonBridge.call(command: "acoustics_capability") { result, _ in + DispatchQueue.main.async { + guard let result, result["ok"] as? Bool == true else { return } + acousticsAvailable = result["available"] as? Bool + } + } + } + + /// Localiza uma linha markdown do tipo "- **Marker**: valor" (usado nas + /// mensagens do bridge Python) e devolve o valor. Aceita o marcador de + /// lista "- " opcional antes dos asteriscos. + private func extractPath(from message: String, marker: String) -> String? { + for line in message.split(separator: "\n") { + var trimmed = Substring(line.trimmingCharacters(in: .whitespaces)) + if trimmed.hasPrefix("- ") { trimmed = trimmed.dropFirst(2) } + if trimmed.hasPrefix(marker) { + return trimmed.dropFirst(marker.count).trimmingCharacters(in: .whitespaces) + } + } + return nil + } + + private func copyForChat(path: String) { + guard let content = try? String(contentsOfFile: path, encoding: .utf8) else { + errorMessage = "Não foi possível ler \(path)." + return + } + let prompt = """ + Use a skill "editar-por-voz" para decidir os cortes deste projeto a partir da timeline de voz abaixo. Devolva só o JSON de decisões (cortes, zooms, textos, marcadores) pronto para eu colar de volta no app. + + ```json + \(content) + ``` + """ + let pasteboard = NSPasteboard.general + pasteboard.clearContents() + pasteboard.setString(prompt, forType: .string) + copiedFeedback = "Copiado — cole (⌘V) numa conversa com o Claude." + } + + /// Monta a revisão uma vez por análise de voz. Voltar e avançar de novo não + /// recarrega: isso jogaria fora as edições manuais em silêncio, que é + /// exatamente o que esta tela existe para preservar. + private func loadReviewIfNeeded() { + guard let voiceTimelinePath, reviewLoadedFor != voiceTimelinePath else { return } + reviewLoadedFor = voiceTimelinePath + // A pasta do projeto e a do .fcpxml entram como onde procurar a mídia: + // a análise de voz guarda só o nome do arquivo, não o caminho. + reviewModel.load( + voiceTimelinePath: voiceTimelinePath, + decisionsJSON: decisionsText, + outputFolder: outputFolder, + mediaFolder: projectPath.map { URL(fileURLWithPath: $0).deletingLastPathComponent().path } + ) + if let projectPath { reviewModel.loadProjectFormat(projectPath: projectPath) } + } + + private func applyDecisions() { + guard let projectPath, let outputFolder, + let data = decisionsText.data(using: .utf8), + let parsed = try? JSONSerialization.jsonObject(with: data) else { return } + isApplyingDecisions = true + errorMessage = nil + PythonBridge.call(command: "apply_voice_actions", arguments: [ + "path": projectPath, + "output_dir": outputFolder, + "actions": parsed, + ]) { result, err in + DispatchQueue.main.async { + isApplyingDecisions = false + guard result?["ok"] as? Bool == true else { + errorMessage = result?["error"] as? String ?? err ?? "Falha ao aplicar as decisões." + return + } + appliedPath = result?["path"] as? String ?? projectPath + skippedVoiceEdit = false + } + } + } + + private func finalizeProcessing() { + guard let outputFolder else { return } + let startPath = appliedPath ?? projectPath + guard let startPath else { return } + var operations: [String] = [] + if finalSilences { operations.append("remove_silences") } + if finalFillers { operations.append("remove_filler_words") } + if finalSubtitles { operations.append("generate_plain_subtitles") } + if finalDynamicSubtitles { operations.append("generate_dynamic_subtitles") } + guard !operations.isEmpty else { return } + isFinalizing = true + errorMessage = nil + finalStatus = "Iniciando…" + finalizeStep(operations, index: 0, currentPath: startPath, outputFolder: outputFolder) + } + + private func finalizeStep(_ operations: [String], index: Int, currentPath: String, outputFolder: String) { + guard index < operations.count else { + isFinalizing = false + finalStatus = "Processamento concluído." + finalPath = currentPath + return + } + let operation = operations[index] + finalStatus = "Processando: \(operation)…" + PythonBridge.call(command: operation, arguments: ["path": currentPath, "output_dir": outputFolder]) { result, err in + DispatchQueue.main.async { + guard result?["ok"] as? Bool == true else { + isFinalizing = false + errorMessage = result?["error"] as? String ?? err ?? "Falha em \(operation)." + finalStatus = "Processamento interrompido." + return + } + let nextPath = result?["path"] as? String ?? currentPath + finalizeStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder) + } + } + } +} diff --git a/code/fcpxml/model_manager.py b/code/fcpxml/model_manager.py index 2f73468..18e211d 100644 --- a/code/fcpxml/model_manager.py +++ b/code/fcpxml/model_manager.py @@ -382,6 +382,10 @@ DEFAULT_VOICE_ANALYSIS_CONFIG: dict = { "emphasis_floor": 0.25, "emotion_enabled": False, "emotion_sensitivity": 0.5, + "zoom_scale": 1.30, + "zoom_mode": "in_out", + "zoom_ease_in": 0.25, + "zoom_ease_out": 0.04, } @@ -402,12 +406,28 @@ def load_voice_analysis_config() -> dict: stored = _load_config().get("voice_analysis") if not isinstance(stored, dict): return cfg - for key in ("energy_threshold", "peak_percentile", "emphasis_floor", "emotion_sensitivity"): + for key in ( + "energy_threshold", "peak_percentile", "emphasis_floor", + "emotion_sensitivity", "zoom_scale", "zoom_ease_in", "zoom_ease_out", + ): if key in stored: try: - cfg[key] = max(0.0, min(1.0, float(stored[key]))) + value = float(stored[key]) + if key == "zoom_scale": + cfg[key] = max(1.0, min(3.0, value)) + elif key.startswith("zoom_ease"): + cfg[key] = max(0.01, min(5.0, value)) + else: + cfg[key] = max(0.0, min(1.0, value)) except (TypeError, ValueError): pass + if "emphasis_threshold" in stored and "emphasis_floor" not in stored: + try: + cfg["emphasis_floor"] = max(0.0, min(1.0, float(stored["emphasis_threshold"]))) + except (TypeError, ValueError): + pass + if stored.get("zoom_mode") in ("in_out", "in", "out"): + cfg["zoom_mode"] = stored["zoom_mode"] if "emotion_enabled" in stored: cfg["emotion_enabled"] = bool(stored["emotion_enabled"]) weights = stored.get("emphasis_weights") @@ -428,6 +448,10 @@ def save_voice_analysis_config( emphasis_floor: float | None = None, emotion_enabled: bool | None = None, emotion_sensitivity: float | None = None, + zoom_scale: float | None = None, + zoom_mode: str | None = None, + zoom_ease_in: float | None = None, + zoom_ease_out: float | None = None, ) -> dict: """Persist voice-analysis thresholds/weights. Only given fields change. @@ -446,6 +470,14 @@ def save_voice_analysis_config( cfg["emotion_enabled"] = bool(emotion_enabled) if emotion_sensitivity is not None: cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity))) + if zoom_scale is not None: + cfg["zoom_scale"] = max(1.0, min(3.0, float(zoom_scale))) + if zoom_mode in ("in_out", "in", "out"): + cfg["zoom_mode"] = zoom_mode + if zoom_ease_in is not None: + cfg["zoom_ease_in"] = max(0.01, min(5.0, float(zoom_ease_in))) + if zoom_ease_out is not None: + cfg["zoom_ease_out"] = max(0.01, min(5.0, float(zoom_ease_out))) if emphasis_weights is not None: for key, value in emphasis_weights.items(): if key in cfg["emphasis_weights"] and value is not None: @@ -536,6 +568,73 @@ def save_dynamic_subtitle_config(**fields) -> dict: return cfg +DEFAULT_PLAIN_SUBTITLE_CONFIG: dict = { + "font": "Helvetica Neue", + "font_size": 82, + "font_color": "1 1 1 1", + "max_words": 7, + "position_y": -820.0, + "uppercase": False, + "keep_punctuation": True, + "text_scale": 2.0, +} + + +def load_plain_subtitle_config() -> dict: + """Persisted style for simple editable FCPXML title subtitles.""" + cfg = dict(DEFAULT_PLAIN_SUBTITLE_CONFIG) + stored = _load_config().get("plain_subtitles") + if not isinstance(stored, dict): + return cfg + for key in ("position_y", "text_scale"): + if key in stored: + try: + cfg[key] = float(stored[key]) + except (TypeError, ValueError): + pass + for key in ("font_size", "max_words"): + if key in stored: + try: + cfg[key] = int(stored[key]) + except (TypeError, ValueError): + pass + for key in ("font", "font_color"): + if key in stored and isinstance(stored[key], str) and stored[key]: + cfg[key] = stored[key] + for key in ("uppercase", "keep_punctuation"): + if key in stored: + cfg[key] = bool(stored[key]) + cfg["max_words"] = max(1, int(cfg["max_words"])) + return cfg + + +def save_plain_subtitle_config(**fields) -> dict: + """Persist simple subtitle style fields. Only given fields change.""" + cfg = load_plain_subtitle_config() + for key, value in fields.items(): + if key not in DEFAULT_PLAIN_SUBTITLE_CONFIG or value is None: + continue + if isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], bool): + cfg[key] = bool(value) + elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], float): + try: + cfg[key] = float(value) + except (TypeError, ValueError): + continue + elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], int): + try: + cfg[key] = int(value) + except (TypeError, ValueError): + continue + else: + cfg[key] = str(value) + cfg["max_words"] = max(1, int(cfg["max_words"])) + data = _load_config() + data["plain_subtitles"] = cfg + _write_config(data) + return cfg + + # Mirrors the silence thresholds the detection/removal handlers use when no # argument is passed (server_tools/qc.py). Persisted so the app's slider and # any later run agree without threading three fields through every call. diff --git a/code/fcpxml/phrase_review.py b/code/fcpxml/phrase_review.py new file mode 100644 index 0000000..f7d4fd1 --- /dev/null +++ b/code/fcpxml/phrase_review.py @@ -0,0 +1,547 @@ +"""Phrase review — the human pass between the AI's decisions and the render. + +A voice timeline says *how* every line was spoken; a list of voice actions says +what the model decided to do about it. Neither is reviewable on its own: the +timeline has no editorial intent, and the action list is a set of timecodes with +no text attached. This module joins them into the one view an editor can +actually judge — the script, phrase by phrase, each carrying the decision that +was made about it. + +The phrase is the unit on purpose. Emphasis, in this pipeline, is not a property +of a word but of a line: an emphasized phrase gets a punch-in and a dynamic +caption, everything else gets a plain caption. Keeping the same granularity in +the review, the JSON, and the render means a toggle in the UI maps to exactly +one editorial outcome, with nothing to reconcile in between. + +Trimming stays inside the phrase for the same reason. A line is rarely wrong as +a whole — it has a false start, or a trailing "né" — so each phrase carries a +``trim_start``/``trim_end`` pair that rides on word boundaries. Editing a cut +therefore means picking a word, never hunting for a frame, and a partial cut +from the model arrives as a trim instead of being rounded away. + +Round-tripping is the other half of the contract. :func:`build_phrase_review` +derives the review from actions, :func:`phrase_review_to_actions` derives +actions back from the edited review, and everything the editor touched wins over +what was inferred — so re-opening the screen shows what was left there, not a +re-derivation that quietly discards the edits. +""" + +import json +from pathlib import Path +from typing import Any, Dict, List, Optional, Sequence, Tuple + +from .voice_actions import VoiceAction, merge_cut_ranges, parse_actions + +PHRASE_REVIEW_VERSION = "1.0" + +# Emphasis is stored 0-3 rather than as a float so the UI, the JSON and the +# render agree on the same discrete decision. The thresholds map the continuous +# `peak_emphasis` of the voice timeline onto those levels when the model gave no +# explicit direction for a phrase. +EMPHASIS_LEVELS = (0, 1, 2, 3) +EMPHASIS_THRESHOLDS = (0.25, 0.45, 0.65) + +# Zoom scale applied per emphasis level when the review is turned back into +# actions. Level 0 never produces a zoom. The values stay inside +# voice_actions.MIN_ZOOM_SCALE..MAX_ZOOM_SCALE. +ZOOM_SCALE_BY_LEVEL = {1: 1.15, 2: 1.3, 3: 1.5} + +# A phrase only survives if most of it does. Speech boundaries from a transcript +# are approximate, so a cut clipping a fraction of a second off the tail is a +# trim, not a removal — treating that as "phrase deleted" would grey out lines +# that are still fully audible. +CUT_COVERAGE_TO_DEACTIVATE = 0.6 + +# A punch-in shorter than this has no time to ramp in and back out — the writer +# rejects the window anyway (see the zoom ease-in/ease-out shape), so refusing +# it here turns a silent drop at render time into nothing being placed at all. +MIN_ZOOM_DURATION = 0.4 + +TRACK_SCRIPT = "roteiro" +TRACK_BACKSTAGE = "bastidor" +TRACKS = (TRACK_SCRIPT, TRACK_BACKSTAGE) + + +def resolve_source( + source: str, voice_timeline_path: str, extra_dirs: Sequence[str] = () +) -> str: + """The playable path for a timeline's ``source``, or "" when it's gone. + + The voice timeline stores only the media's *file name* — it is written to be + read by a model, where a machine-specific absolute path is noise. That makes + it useless for opening a preview, so the file is looked up where it can + actually be: beside its own timeline JSON first (that is where + ``analyze_voice`` writes it), then in whatever project folders the caller + knows about. + """ + if not source: + return "" + candidate = Path(source) + if candidate.is_absolute() and candidate.is_file(): + return str(candidate) + + directories = [Path(voice_timeline_path).parent] if voice_timeline_path else [] + directories += [Path(d) for d in extra_dirs if d] + for directory in directories: + found = directory / candidate.name + if found.is_file(): + return str(found) + return "" + + +def _overlap(a_start: float, a_end: float, b_start: float, b_end: float) -> float: + """Seconds shared by two spans (0.0 when they don't touch).""" + return max(0.0, min(a_end, b_end) - max(a_start, b_start)) + + +def _cut_coverage( + start: float, end: float, cuts: Sequence[Tuple[float, float]] +) -> float: + """Fraction of ``start``-``end`` that falls inside ``cuts`` (0-1).""" + span = end - start + if span <= 0: + return 0.0 + removed = sum(_overlap(start, end, c_start, c_end) for c_start, c_end in cuts) + return min(1.0, removed / span) + + +def snap_to_words( + time: float, words: Sequence[dict], fallback: float, edge: str +) -> float: + """Move ``time`` onto the nearest word boundary of this phrase. + + Trims are expressed by pointing at a word, so a trim handle that landed + mid-word would cut a syllable in half. ``edge`` is ``"in"`` (snap to word + starts) or ``"out"`` (snap to word ends); with no word timings available the + time is left as-is. + """ + boundaries = [ + float(word.get("start" if edge == "in" else "end", 0.0)) for word in words + ] + boundaries = [b for b in boundaries if b > 0] + if not boundaries: + return fallback + return min(boundaries, key=lambda b: abs(b - time)) + + +def _trim_from_cuts( + start: float, + end: float, + words: Sequence[dict], + cuts: Sequence[Tuple[float, float]], +) -> Tuple[float, float]: + """Read a partial cut over this phrase as a head/tail trim. + + Only cuts that touch an edge become trims: a cut carved out of the middle of + a line has no representation here (the phrase is the unit), so it is left + for the whole-phrase coverage rule to decide. + """ + trim_start, trim_end = start, end + for cut_start, cut_end in cuts: + if _overlap(start, end, cut_start, cut_end) <= 0: + continue + if cut_start <= trim_start < cut_end < end: + trim_start = snap_to_words(cut_end, words, cut_end, "in") + if start < cut_start < trim_end <= cut_end: + trim_end = snap_to_words(cut_start, words, cut_start, "out") + if trim_end <= trim_start: + return start, end + return trim_start, trim_end + + +def _level_from_peak(peak: float) -> int: + """Map a 0-1 ``peak_emphasis`` onto a 0-3 level.""" + for level, threshold in enumerate(EMPHASIS_THRESHOLDS): + if peak < threshold: + return level + return 3 + + +def _level_from_scale(scale: Optional[float]) -> int: + """Map a zoom's scale factor back onto a 0-3 level. + + The model is free to send any scale inside the allowed range, so this picks + the nearest level rather than requiring one of our own three values. + """ + if scale is None: + return 2 + best = 1 + smallest = None + for level, level_scale in ZOOM_SCALE_BY_LEVEL.items(): + distance = abs(level_scale - float(scale)) + if smallest is None or distance < smallest: + smallest, best = distance, level + return best + + +def _emphasis_from_actions( + start: float, + end: float, + actions: Sequence[VoiceAction], +) -> Tuple[Optional[int], str]: + """The level the model asked for on this phrase, and why. + + A ``zoom`` or ``text`` action anywhere inside the phrase is read as "this + line is the emphasis" — the model places them on the word that carries the + point, not on the whole line, so requiring a full-span match would find + nothing. Returns ``(None, "")`` when no action touches the phrase. + """ + level: Optional[int] = None + reason = "" + for action in actions: + if action.kind not in ("zoom", "text"): + continue + if _overlap(start, end, action.start, action.end) <= 0: + continue + if action.kind == "zoom": + candidate = _level_from_scale(action.params.get("scale")) + else: + candidate = 2 + if level is None or candidate > level: + level = candidate + reason = action.reason + return level, reason + + +def _cut_reason( + start: float, end: float, actions: Sequence[VoiceAction] +) -> str: + """The reason given for the cut that removes this phrase.""" + for action in actions: + if action.kind != "cut": + continue + if _overlap(start, end, action.start, action.end) > 0 and action.reason: + return action.reason + return "" + + +def build_phrase_review( + timeline: dict, + actions: Any = None, + voice_timeline_path: str = "", + extra_dirs: Sequence[str] = (), +) -> dict: + """Join a voice timeline with the AI's actions into a reviewable script. + + ``actions`` accepts whatever :func:`~.voice_actions.parse_actions` accepts — + a bare list, ``{"actions": [...]}``, or ``None`` when there is no AI pass and + the review starts from the acoustics alone. Malformed rows are skipped and + reported in ``errors`` rather than raising, matching the rest of the + decision pipeline. + """ + parsed, errors = parse_actions(actions) if actions else ([], []) + cuts = merge_cut_ranges(parsed) + + phrases: List[dict] = [] + for index, segment in enumerate(timeline.get("segments", [])): + start = float(segment.get("start", 0.0)) + end = float(segment.get("end", 0.0)) + peak = float(segment.get("peak_emphasis", 0.0)) + take_boundary = bool(segment.get("take_boundary", False)) + + words = list(segment.get("words", [])) + coverage = _cut_coverage(start, end, cuts) + active = coverage < CUT_COVERAGE_TO_DEACTIVATE + trim_start, trim_end = ( + _trim_from_cuts(start, end, words, cuts) if active else (start, end) + ) + + asked_level, asked_reason = _emphasis_from_actions(start, end, parsed) + if asked_level is not None: + emphasis, reason = asked_level, asked_reason + else: + emphasis = _level_from_peak(peak) + reason = f"ênfase {peak:.2f}" if emphasis else "" + if not active: + # A removed line carries the reason it was removed; the emphasis it + # would have had is kept so re-activating it restores the decision. + reason = _cut_reason(start, end, parsed) or reason + + phrases.append( + { + "index": index, + "start": round(start, 3), + "end": round(end, 3), + "trim_start": round(trim_start, 3), + "trim_end": round(trim_end, 3), + "text": str(segment.get("text", "")).strip(), + "speaker": str(segment.get("speaker", "")), + "active": active, + "emphasis": emphasis, + "track": TRACK_BACKSTAGE if (not active and take_boundary) else TRACK_SCRIPT, + "peak_emphasis": round(peak, 3), + # Delivery emotion is a heuristic over the acoustics (see + # voice_timeline._emotion_for_word) and only means anything when + # the analysis actually ran — `emotion_available` below is what + # separates "spoken flat" from "never measured". + "emotion": str(segment.get("emotion", "neutral")), + "emotion_confidence": round( + float(segment.get("emotion_confidence", 0.0)), 3 + ), + "take_boundary": take_boundary, + "gap_before": round(float(segment.get("gap_before", 0.0)), 3), + "reason": reason, + "words": [ + { + "text": str(word.get("text", "")), + "start": round(float(word.get("start", 0.0)), 3), + "end": round(float(word.get("end", 0.0)), 3), + "energy": round(float(word.get("energy", 0.0)), 3), + "emphasis": round(float(word.get("emphasis", 0.0)), 3), + } + for word in words + ], + } + ) + + source = timeline.get("source", "") + layers = timeline.get("layers", {}) if isinstance(timeline.get("layers"), dict) else {} + return { + "version": PHRASE_REVIEW_VERSION, + "source": source, + "source_path": resolve_source(source, voice_timeline_path, extra_dirs), + "duration": round(phrases[-1]["end"], 3) if phrases else 0.0, + "speakers": timeline.get("speakers", []), + "emotion_available": bool(layers.get("emotion", False)), + "phrases": phrases, + # Punch-ins the editor places by hand on an arbitrary range, alongside + # the whole-phrase zoom that an emphasis level produces. Both end up as + # zoom actions; this one exists because the moment worth punching into + # is not always a whole sentence. + "zooms": [], + "errors": errors, + } + + +def _coerce_zoom(raw: Any) -> Optional[Dict[str, float]]: + """Normalize one manually placed zoom range.""" + if not isinstance(raw, dict): + return None + try: + start = float(raw.get("start")) + end = float(raw.get("end")) + except (TypeError, ValueError): + return None + if end - start < MIN_ZOOM_DURATION: + return None + return {"start": start, "end": end} + + +def _coerce_phrase(raw: Any, index: int) -> Optional[Dict[str, Any]]: + """Normalize one edited phrase row coming back from the UI.""" + if not isinstance(raw, dict): + return None + try: + start = float(raw.get("start")) + end = float(raw.get("end")) + except (TypeError, ValueError): + return None + if end <= start: + return None + try: + emphasis = int(raw.get("emphasis", 0)) + except (TypeError, ValueError): + emphasis = 0 + try: + trim_start = float(raw.get("trim_start", start)) + trim_end = float(raw.get("trim_end", end)) + except (TypeError, ValueError): + trim_start, trim_end = start, end + # A trim that escaped the phrase, or inverted, is treated as no trim at all: + # the UI is the only thing that writes these, and silently discarding a bad + # pair keeps a rounding slip from deleting material the editor kept. + if not (start <= trim_start < trim_end <= end): + trim_start, trim_end = start, end + track = str(raw.get("track", TRACK_SCRIPT)) + return { + "index": int(raw.get("index", index)), + "start": start, + "end": end, + "trim_start": trim_start, + "trim_end": trim_end, + "text": str(raw.get("text", "")).strip(), + "speaker": str(raw.get("speaker", "")), + "active": bool(raw.get("active", True)), + "emphasis": min(3, max(0, emphasis)), + "track": track if track in TRACKS else TRACK_SCRIPT, + "reason": str(raw.get("reason", "")), + } + + +def phrase_review_to_actions(review: dict) -> dict: + """Turn an edited review back into the action list the applier consumes. + + Every deactivated phrase becomes a ``cut``, a trimmed one becomes a cut over + the head and/or tail it lost, and every emphasized one becomes a ``zoom`` + scaled by its level. The emphasis flags ride along in ``emphasis_spans`` so + the caption step can give those lines the dynamic treatment and everything + else the plain one, without re-deriving the decision from the acoustics. + """ + phrases = [ + coerced + for index, raw in enumerate(review.get("phrases", [])) + if (coerced := _coerce_phrase(raw, index)) is not None + ] + + actions: List[dict] = [] + emphasis_spans: List[dict] = [] + for phrase in phrases: + if not phrase["active"]: + actions.append( + VoiceAction( + kind="cut", + start=phrase["start"], + end=phrase["end"], + reason=phrase["reason"] or "desativada na revisão", + speaker=phrase["speaker"], + ).as_dict() + ) + continue + + # Head and tail the editor trimmed off — each becomes its own cut, so a + # false start disappears without taking the line with it. + for trim_start, trim_end, where in ( + (phrase["start"], phrase["trim_start"], "início"), + (phrase["trim_end"], phrase["end"], "fim"), + ): + if trim_end - trim_start <= 0: + continue + actions.append( + VoiceAction( + kind="cut", + start=trim_start, + end=trim_end, + reason=f"trecho do {where} da frase removido na revisão", + speaker=phrase["speaker"], + ).as_dict() + ) + + if phrase["emphasis"] >= 1: + actions.append( + VoiceAction( + kind="zoom", + start=phrase["trim_start"], + end=phrase["trim_end"], + params={"scale": ZOOM_SCALE_BY_LEVEL[phrase["emphasis"]]}, + reason=phrase["reason"] or f"ênfase nível {phrase['emphasis']}", + speaker=phrase["speaker"], + ).as_dict() + ) + emphasis_spans.append( + { + "start": phrase["trim_start"], + "end": phrase["trim_end"], + "level": phrase["emphasis"], + "text": phrase["text"], + } + ) + + # Hand-placed punch-ins carry no scale on purpose: an omitted scale lets the + # applier use the shape configured in "Análise de Voz" (zoom_scale, ease in + # and out), so changing that setting restyles every manual zoom instead of + # leaving a scale frozen into each one at the moment it was drawn. + for raw in review.get("zooms", []): + zoom = _coerce_zoom(raw) + if zoom is None: + continue + actions.append( + VoiceAction( + kind="zoom", + start=zoom["start"], + end=zoom["end"], + reason="zoom marcado na revisão", + ).as_dict() + ) + + return { + "source": review.get("source", ""), + "actions": actions, + "emphasis_spans": emphasis_spans, + } + + +def merge_saved_decisions(review: dict, saved: Optional[dict]) -> dict: + """Lay a previously saved review's decisions over a freshly built one. + + Only the editorial fields travel — active, emphasis, track, text, trims. + Everything else (words, emotion, energy) is re-derived from the current + analysis, so re-running the voice pass with better settings improves the + screen instead of being masked by a stale copy of itself, and the saved file + never has to carry a duplicate of data it does not own. + + Phrases are matched by index *and* start time: if the analysis changed + enough to move a line, the old decision for that slot is dropped rather than + applied to a different sentence. + """ + if not saved: + return review + + review["zooms"] = [ + zoom for raw in saved.get("zooms", []) if (zoom := _coerce_zoom(raw)) is not None + ] + + by_index = {} + for raw in saved.get("phrases", []): + if isinstance(raw, dict) and "index" in raw: + by_index[raw["index"]] = raw + + for phrase in review["phrases"]: + previous = by_index.get(phrase["index"]) + if previous is None: + continue + if abs(float(previous.get("start", -1)) - phrase["start"]) > 0.25: + continue + phrase["active"] = bool(previous.get("active", phrase["active"])) + phrase["emphasis"] = min(3, max(0, int(previous.get("emphasis", phrase["emphasis"])))) + track = str(previous.get("track", phrase["track"])) + phrase["track"] = track if track in TRACKS else phrase["track"] + if previous.get("text"): + phrase["text"] = str(previous["text"]) + trim_start = float(previous.get("trim_start", phrase["trim_start"])) + trim_end = float(previous.get("trim_end", phrase["trim_end"])) + if phrase["start"] <= trim_start < trim_end <= phrase["end"]: + phrase["trim_start"], phrase["trim_end"] = trim_start, trim_end + + return review + + +def review_paths(voice_timeline_path: str) -> Tuple[Path, Path]: + """Where the review and its derived actions live, next to the timeline. + + Both files sit beside the ``_voice_timeline.json`` they came from and are + named after it, so a project folder stays readable and re-running the wizard + on the same take overwrites its own files instead of accumulating copies. + """ + base = Path(voice_timeline_path) + stem = base.stem + if stem.endswith("_voice_timeline"): + stem = stem[: -len("_voice_timeline")] + return ( + base.with_name(f"{stem}_phrase_review.json"), + base.with_name(f"{stem}_phrase_actions.json"), + ) + + +def save_phrase_review(voice_timeline_path: str, review: dict) -> Tuple[Path, Path]: + """Write the edited review and the actions derived from it. Returns both paths.""" + review_path, actions_path = review_paths(voice_timeline_path) + review_path.write_text( + json.dumps(review, ensure_ascii=False, indent=2), encoding="utf-8" + ) + actions_path.write_text( + json.dumps(phrase_review_to_actions(review), ensure_ascii=False, indent=2), + encoding="utf-8", + ) + return review_path, actions_path + + +def load_phrase_review(voice_timeline_path: str) -> Optional[dict]: + """The review saved earlier for this timeline, or ``None`` if there is none.""" + review_path, _ = review_paths(voice_timeline_path) + if not review_path.is_file(): + return None + try: + data = json.loads(review_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return None + return data if isinstance(data, dict) else None diff --git a/code/fcpxml/transcribe.py b/code/fcpxml/transcribe.py index a14ba6b..f4ab00f 100755 --- a/code/fcpxml/transcribe.py +++ b/code/fcpxml/transcribe.py @@ -28,8 +28,11 @@ ALLOWED_MODELS = ( ) # Conservative by default: interjections that are near-universally filler. +# Portuguese "um"/"uma" are usually articles/numerals inside real phrases +# ("de um jeito") rather than discardable hesitations, so only cut them when +# the caller explicitly opts in through the fillers argument. # "like" / "so" / "actually" are speech, not noise, unless the user opts in. -DEFAULT_FILLERS = ("um", "uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm") +DEFAULT_FILLERS = ("uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm") _NORM_RE = re.compile(r"[^\w']+") diff --git a/code/fcpxml/voice_actions.py b/code/fcpxml/voice_actions.py index 42c5bdc..eacc818 100644 --- a/code/fcpxml/voice_actions.py +++ b/code/fcpxml/voice_actions.py @@ -82,21 +82,36 @@ def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]: params = dict(params) if isinstance(params, dict) else {} if kind == "zoom": - try: - scale = float(params.get("scale", 1.3)) - except (TypeError, ValueError): - return None, f"{where}: zoom scale must be a number" - if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE): - return None, ( - f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}" - ) - params["scale"] = scale + if "scale" in params and params.get("scale") is not None: + try: + scale = float(params["scale"]) + except (TypeError, ValueError): + return None, f"{where}: zoom scale must be a number" + if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE): + return None, ( + f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}" + ) + params["scale"] = scale if kind == "text": content = str(params.get("content", "")).strip() if not content: return None, f"{where}: text action needs params.content" params["content"] = content[:MAX_TEXT_LENGTH] + # Style is optional — omitted fields fall back to the "Legendas + # Dinâmicas" emphasis style at apply time (see _apply_placed_action), + # so a callout matches the captions' look without the caller having + # to know or repeat that configuration. Anything given here wins. + for key in ("font", "font_color", "face"): + if key in params and not isinstance(params[key], str): + del params[key] + if "font_size" in params: + try: + params["font_size"] = int(params["font_size"]) + except (TypeError, ValueError): + del params["font_size"] + if "bold" in params: + params["bold"] = bool(params["bold"]) return ( VoiceAction( diff --git a/code/fcpxml/voice_timeline.py b/code/fcpxml/voice_timeline.py index 6cd3a81..1bd9c2d 100644 --- a/code/fcpxml/voice_timeline.py +++ b/code/fcpxml/voice_timeline.py @@ -56,12 +56,20 @@ VALUE_SCALES = { "rate_delta": "0-1, how much the local speaking rate departs from the average", "pause_before": "seconds of silence immediately before the word", "emphasis": "0-1 combined index; high values are punch-in/highlight candidates", + "emotion": "heuristic label from delivery: neutral, excited, tense, calm, reflective", + "emotion_confidence": "0-1 confidence in the heuristic emotion label", + "arousal": "0-1 vocal activation from energy/rate/pitch movement", + "valence": "0-1 rough positive tone; lower values suggest tension/weight", }, "segment": { "gap_before": "seconds of silence before this line", "take_boundary": "true when the gap is long enough that the take likely restarted here", "avg_energy": "0-1 mean loudness across the line", "peak_emphasis": "0-1 highest emphasis of any word in the line", + "emotion": "dominant delivery emotion across the line", + "emotion_confidence": "0-1 confidence in the dominant segment emotion", + "arousal": "0-1 mean vocal activation across the line", + "valence": "0-1 mean rough positive tone across the line", }, } @@ -92,11 +100,74 @@ def _round_word(word: dict) -> dict: "rate_delta": round(word.get("rate_delta", 0.0), 3), "pause_before": round(word.get("pause_before", 0.0), 3), "emphasis": round(word.get("emphasis", 0.0), 3), + "emotion": word.get("emotion", "neutral"), + "emotion_confidence": round(word.get("emotion_confidence", 0.0), 3), + "arousal": round(word.get("arousal", 0.0), 3), + "valence": round(word.get("valence", 0.5), 3), "energy_raw": word.get("energy"), "pitch_hz": word.get("pitch_hz"), } +def _emotion_for_word(word: dict, enabled: bool, sensitivity: float) -> dict: + """Classify delivery emotion from normalized acoustic features. + + This is deliberately a local heuristic rather than a claimed clinical + emotion model. It gives the editor a useful signal about delivery shape + while degrading predictably when acoustic extraction is unavailable. + """ + if not enabled: + return { + "emotion": "neutral", + "emotion_confidence": 0.0, + "arousal": 0.0, + "valence": 0.5, + } + + energy = float(word.get("energy_norm", 0.0)) + pitch = float(word.get("pitch_delta", 0.0)) + rate = float(word.get("rate_delta", 0.0)) + pause = min(float(word.get("pause_before", 0.0)) / 2.0, 1.0) + emphasis = float(word.get("emphasis", 0.0)) + + arousal = max(0.0, min(1.0, energy * 0.45 + pitch * 0.25 + rate * 0.20 + emphasis * 0.10)) + valence = max(0.0, min(1.0, 0.55 + energy * 0.15 - pause * 0.20 - rate * 0.10)) + + if arousal >= 0.68 and valence >= 0.50: + label = "excited" + confidence = arousal + elif arousal >= 0.58 and valence < 0.50: + label = "tense" + confidence = max(arousal, 1.0 - valence) + elif arousal <= 0.28 and pause >= 0.25: + label = "reflective" + confidence = max(1.0 - arousal, pause) + elif arousal <= 0.35: + label = "calm" + confidence = 1.0 - arousal + else: + label = "neutral" + confidence = 1.0 - abs(arousal - 0.5) * 2.0 + + confidence = max(0.0, min(1.0, confidence)) + if confidence < sensitivity: + label = "neutral" + return { + "emotion": label, + "emotion_confidence": confidence, + "arousal": arousal, + "valence": valence, + } + + +def annotate_emotions(words: Sequence[dict], enabled: bool, sensitivity: float) -> List[dict]: + """Attach heuristic emotion labels to enriched word rows.""" + return [ + {**w, **_emotion_for_word(w, enabled, sensitivity)} + for w in words + ] + + def enrich_words( words: Sequence[dict], pitch_track: Optional[Sequence] = None, @@ -166,6 +237,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict] in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end] energies = [w["energy_norm"] for w in in_seg] emphases = [w["emphasis"] for w in in_seg] + arousals = [w.get("arousal", 0.0) for w in in_seg] + valences = [w.get("valence", 0.5) for w in in_seg] + emotions = [w.get("emotion", "neutral") for w in in_seg] + dominant = max(set(emotions), key=emotions.count) if emotions else "neutral" + emotion_confidences = [ + w.get("emotion_confidence", 0.0) for w in in_seg if w.get("emotion") == dominant + ] gap = max(0.0, start - previous_end) rows.append( { @@ -181,6 +259,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict] "take_boundary": gap >= TAKE_BOUNDARY_GAP, "avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0, "peak_emphasis": round(max(emphases), 3) if emphases else 0.0, + "emotion": dominant, + "emotion_confidence": ( + round(sum(emotion_confidences) / len(emotion_confidences), 3) + if emotion_confidences else 0.0 + ), + "arousal": round(sum(arousals) / len(arousals), 3) if arousals else 0.0, + "valence": round(sum(valences) / len(valences), 3) if valences else 0.5, "words": [_round_word(w) for w in in_seg], } ) @@ -425,6 +510,8 @@ def build_voice_timeline( weights: EmphasisWeights = EmphasisWeights(), peak_percentile: float = 0.02, emphasis_floor: float = 0.25, + emotion_enabled: bool = False, + emotion_sensitivity: float = 0.5, progress_cb: Optional[Callable[[float, str], None]] = None, ) -> dict: """Build the consolidated voice timeline for one media file. @@ -445,6 +532,7 @@ def build_voice_timeline( report(0.5, "Calculando ênfase...") words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights) + words = annotate_emotions(words, emotion_enabled, emotion_sensitivity) report(0.7, "Identificando participantes...") tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None @@ -465,6 +553,7 @@ def build_voice_timeline( "transcript": bool(transcript.get("words")), "acoustics": pitch_track is not None or energy_track is not None, "speakers": tracks is not None, + "emotion": bool(emotion_enabled), }, "scales": VALUE_SCALES, "summary": _summary( diff --git a/code/fcpxml/writer.py b/code/fcpxml/writer.py index c853df2..8df945d 100755 --- a/code/fcpxml/writer.py +++ b/code/fcpxml/writer.py @@ -2218,13 +2218,23 @@ class FCPXMLModifier: seg_start: 'TimeValue', seg_duration: 'TimeValue', ) -> None: - """Remove markers/keywords from *clip* that fall outside the segment range. + """Remove markers/keywords/titles from *clip* that fall outside the segment range. After ``split_clip`` deepcopy's the original clip into each segment, every segment inherits all child elements. Markers whose ``start`` falls outside ``[seg_start, seg_start + seg_duration)`` are phantom duplicates and must be removed. Keywords that partially overlap get their ``start``/``duration`` clamped to the segment boundaries. + + A lane-nested ```` (a "text" voice action's on-screen callout, + or a caption from an earlier `generate_dynamic_subtitles` pass) is + the same kind of phantom duplicate, just keyed on ``offset`` instead + of ``start`` — its offset lives in the same source-media coordinate + space as a marker's ``start`` (see ``add_text_title``/``add_marker``, + both anchored at ``parent.start``). Left unfiltered, every further + cut (silence removal, filler removal) duplicates it into every + resulting piece, so the same word shows up several times across the + edited timeline instead of once where it was placed. """ seg_end = seg_start + seg_duration to_remove = [] @@ -2234,6 +2244,10 @@ class FCPXMLModifier: child_start = TimeValue.from_timecode(child.get('start', '0s')) if child_start < seg_start or child_start >= seg_end: to_remove.append(child) + elif tag == 'title': + title_offset = TimeValue.from_timecode(child.get('offset', '0s')) + if title_offset < seg_start or title_offset >= seg_end: + to_remove.append(child) elif tag == 'keyword': kw_start = TimeValue.from_timecode(child.get('start', '0s')) kw_dur = TimeValue.from_timecode(child.get('duration', '0s')) @@ -2304,6 +2318,7 @@ class FCPXMLModifier: self._filter_children_for_segment( new_clip, current_start, segment_duration ) + self._reassign_text_style_ids(new_clip) spine.insert(clip_index + len(new_clips), new_clip) new_clips.append(new_clip) @@ -2404,6 +2419,7 @@ class FCPXMLModifier: new_clip.set('start', seg_start.to_fcpxml()) new_clip.set('duration', seg_duration.to_fcpxml()) self._filter_children_for_segment(new_clip, seg_start, seg_duration) + self._reassign_text_style_ids(new_clip) spine.insert(clip_index + len(new_clips), new_clip) new_clips.append(new_clip) current_offset = current_offset + seg_duration @@ -2946,6 +2962,7 @@ class FCPXMLModifier: ('-469658744/1000000000s', '0'), ('12328542033/1000000000s', '1'), ) + _TEXT_SIZE_KEY = '9999/10003/13260/3296672360/5/3296672362/3' def _ensure_text_title_effect(self, resources: ET.Element) -> str: """Return the resource id of the "Text" (Basic Text) effect, creating it if absent.""" @@ -2995,6 +3012,34 @@ class FCPXMLModifier: self._text_style_ids.add(candidate) return candidate + def _reassign_text_style_ids(self, clip: ET.Element) -> None: + """Give every ``<text-style-def>`` inside a just-deepcopy'd *clip* a + fresh document-unique id, repointing any ``<text-style ref="...">`` + in the same subtree that pointed at the old one. + + ``split_clip``/``cut_clip_ranges`` deepcopy the clip once per + resulting segment, so a clip carrying a ``<title>`` (from a "text" + voice action) keeps the exact same ``text-style-def id`` in every + copy. A single cut is harmless — but the batch chain re-cuts the + same clip at each step (silence removal, filler removal, dynamic + subtitles), and every pass multiplies the duplicate, so the DTD + validator eventually rejects the file with "ID ... already + defined". Regenerating here, at the only place copies are made, + fixes it for every caller instead of each one having to remember to. + """ + for style_def in clip.findall('.//text-style-def'): + old_id = style_def.get('id') + if not old_id: + continue + slug = old_id[3:] if old_id.startswith('ts_') else old_id + slug = re.sub(r'_\d+$', '', slug) # drop a prior _<N> counter + new_id = self._unique_text_style_id(slug) + if new_id == old_id: + continue + style_def.set('id', new_id) + for ref_el in clip.findall(f".//text-style[@ref='{old_id}']"): + ref_el.set('ref', new_id) + def _make_text_title_clip( self, effect_id: str, @@ -3012,6 +3057,8 @@ class FCPXMLModifier: face: Optional[str] = None, kerning: Optional[float] = None, font_scale: float = TEXT_TEMPLATE_FONT_SCALE, + animated: bool = True, + size_param: Optional[float] = None, ) -> ET.Element: """Build a standalone ``<title>`` clip from the "Text" (Basic Text) template. @@ -3042,9 +3089,12 @@ class FCPXMLModifier: param.set('key', key) param.set('value', value) + animation_params = {'Opacity', 'Speed', 'Apply Speed'} for param_name, param_key, param_value in self._TEXT_TITLE_PARAMS: + if not animated and param_name in animation_params: + continue _add_param(param_name, param_key, param_value) - if param_name == 'Speed': + if animated and param_name == 'Speed': # "Custom Speed" lands between "Speed" and "Apply Speed" and # carries a <keyframeAnimation> child instead of a value. cs = ET.SubElement(elem, 'param') @@ -3056,6 +3106,9 @@ class FCPXMLModifier: kf.set('time', kf_time) kf.set('value', kf_value) + if size_param is not None: + _add_param('Size', self._TEXT_SIZE_KEY, f"{float(size_param):g}") + text_el = ET.SubElement(elem, 'text') ts_id = self._unique_text_style_id(name) run = ET.SubElement(text_el, 'text-style') @@ -3109,6 +3162,10 @@ class FCPXMLModifier: font_size: int = 196, font_color: str = '1 1 1 1', bold: bool = True, + face: Optional[str] = None, + animated: bool = True, + font_scale: float = TEXT_TEMPLATE_FONT_SCALE, + size_param: Optional[float] = None, ) -> ET.Element: """Add a single static "Text" (Basic Text) title over *parent_clip*. @@ -3142,6 +3199,10 @@ class FCPXMLModifier: font_size=font_size, font_color=font_color, bold=bold, + face=face, + animated=animated, + font_scale=font_scale, + size_param=size_param, ) _dtd_insert(parent, title) return title diff --git a/code/server.py b/code/server.py index 0215603..d04eb92 100644 --- a/code/server.py +++ b/code/server.py @@ -154,6 +154,7 @@ from server_tools.roles import ( ) from server_tools.subtitles import ( handle_generate_dynamic_subtitles, + handle_generate_plain_subtitles, handle_validate_subtitle_layout, ) from server_tools.timeline import ( @@ -320,6 +321,7 @@ __all__ = [ "handle_save_voice_analysis_config", "handle_validate_subtitle_layout", "handle_generate_dynamic_subtitles", + "handle_generate_plain_subtitles", "handle_push_to_fcp", "handle_list_fcp_libraries", ] diff --git a/code/server_tools/_shared.py b/code/server_tools/_shared.py index cf359c3..fcb6354 100644 --- a/code/server_tools/_shared.py +++ b/code/server_tools/_shared.py @@ -15,6 +15,7 @@ from typing import Any, Sequence from mcp.types import TextContent from fcpxml.media_intel import media_src_to_path +from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config from fcpxml.models import ( DuplicateGroup, FlashFrame, @@ -25,6 +26,7 @@ from fcpxml.models import ( ) from fcpxml.parser import FCPXMLParser from fcpxml.rough_cut import RoughCutGenerator +from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe from fcpxml.writer import FCPXMLModifier @@ -639,28 +641,79 @@ def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str: rel_end = action.end - clip_start if action.kind == "zoom": + config = load_voice_analysis_config() # Only forward an explicit ease — otherwise add_zoom's own default # (a fast ramp in, instant snap back out) is what should apply. - zoom_args = {} - if action.params.get("ease") is not None: - zoom_args["ease"] = float(action.params["ease"]) - if action.params.get("ease_out") is not None: - zoom_args["ease_out"] = float(action.params["ease_out"]) + zoom_args = { + "ease": float(action.params.get("ease", config["zoom_ease_in"])), + "ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])), + } + mode = str(action.params.get("mode", config["zoom_mode"])) + if mode == "in": + zoom_args["hold_at_end"] = True + zoom_args["start_at_peak"] = False + elif mode == "out": + zoom_args["hold_at_end"] = False + zoom_args["start_at_peak"] = True + elif mode == "in_out": + zoom_args["hold_at_end"] = False + zoom_args["start_at_peak"] = False modifier.add_zoom( clip_id=clip_el, start=rel_start, end=rel_end, - scale=float(action.params.get("scale", 1.3)), + scale=float(action.params.get("scale", config["zoom_scale"])), **zoom_args, ) - return f"zoom {action.params.get('scale', 1.3):.2f}x" + return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x" if action.kind == "text": + # Default to the "Legendas Dinâmicas" emphasis style (the font used + # to highlight a word in the captions) rather than a hardcoded + # Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of + # the video's on-screen text instead of looking like a stray default + # title. Any of these the action itself specifies still wins. + subtitle_cfg = load_dynamic_subtitle_config() + font = action.params.get("font", subtitle_cfg["emphasis_font"]) + face = action.params.get("face", subtitle_cfg["emphasis_face"]) + font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0) + requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"])) + requested_kerning = float(action.params.get("kerning", 0.0) or 0.0) + + # Voice-action callouts are not part of the dynamic subtitle block. + # When omitted, put them above the subtitle band and shrink wide + # phrases to the title-safe width. The previous default (Position 0 0, + # full emphasis size) made long callouts like "PRÓTESES DE SILICONE" + # collide with captions and run off both sides of a vertical frame. + emitted_size = requested_size * font_scale + emitted_kerning = requested_kerning * font_scale + safe_width = modifier.frame_width() * 0.90 + width = measure_text( + action.params["content"], + emitted_size, + bold=bool(action.params.get("bold", False)), + kerning=emitted_kerning, + font=font, + face=face, + ) + font_size = requested_size + if width > safe_width and width > 0: + font_size = max(32, int(requested_size * safe_width / width)) + position = action.params.get("position") + if not position: + position = f"0 {modifier.frame_height() * 0.23:g}" + modifier.add_text_title( clip_el, action.params["content"], offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(), duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(), + position=position, + font=font, + font_size=font_size, + font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]), + face=face, + bold=action.params.get("bold", False), ) return f"text \"{action.params['content'][:24]}\"" diff --git a/code/server_tools/subtitles.py b/code/server_tools/subtitles.py index 4f009a5..11c1e9a 100644 --- a/code/server_tools/subtitles.py +++ b/code/server_tools/subtitles.py @@ -6,13 +6,14 @@ Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalo from __future__ import annotations import json +import re from pathlib import Path from typing import Sequence from mcp.types import TextContent, Tool from fcpxml.media_intel import media_src_to_path -from fcpxml.model_manager import load_dynamic_subtitle_config +from fcpxml.model_manager import load_dynamic_subtitle_config, load_plain_subtitle_config from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle from fcpxml.writer import FCPXMLModifier from server_tools._shared import ( @@ -69,9 +70,76 @@ TOOLS = [ "required": ["filepath"] } ), + Tool( + name="generate_plain_subtitles", + description="Generate simple editable FCPXML text-title subtitles, synchronized to transcript words but without visual build-in/build-out effects. Words are grouped into short blocks, placed at a configurable vertical position, and written as static Text titles rather than SRT captions.", + inputSchema={ + "type": "object", + "properties": { + "filepath": {"type": "string", "description": "Path to FCPXML file"}, + "clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"}, + "model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"}, + "language": {"type": "string", "description": "ISO language code hint (e.g. 'pt'); auto-detected if omitted"}, + "font": {"type": "string", "description": "Text font family. Falls back to saved plain-subtitle config."}, + "font_size": {"type": "integer", "description": "Font size in canvas points. Falls back to saved plain-subtitle config."}, + "font_color": {"type": "string", "description": "RGBA (0-1, space-separated). Falls back to saved plain-subtitle config."}, + "max_words": {"type": "integer", "description": "Maximum words per subtitle block. Falls back to saved plain-subtitle config."}, + "position_y": {"type": "number", "description": "Vertical title position in canvas points; negative sits lower in frame."}, + "uppercase": {"type": "boolean", "description": "Render text in uppercase."}, + "keep_punctuation": {"type": "boolean", "description": "Keep punctuation such as comma and period."}, + "text_scale": {"type": "number", "description": "Template font-size scale. Falls back to saved plain-subtitle config."}, + "output_path": {"type": "string", "description": "Output path (default: adds _plain_subtitles suffix)"}, + }, + "required": ["filepath"] + } + ), ] +_PUNCT_RE = re.compile(r"[^\w\sÀ-ÖØ-öø-ÿ]", re.UNICODE) + + +def _words_overlapping_clip(words: Sequence[dict], start: float, end: float) -> list[dict]: + """Return transcript words that overlap a source window, rebased to it.""" + clip_words: list[dict] = [] + for w in words: + word_start = float(w.get("start", 0.0)) + word_end = float(w.get("end", word_start)) + if word_end <= start or word_start >= end: + continue + clip_words.append( + { + "word": w.get("word", ""), + "start": max(0.0, word_start - start), + "end": max(0.0, min(word_end, end) - start), + } + ) + return clip_words + + +def _plain_word_text(word: str, *, uppercase: bool, keep_punctuation: bool) -> str: + text = str(word or "").strip() + if not keep_punctuation: + text = _PUNCT_RE.sub("", text) + text = re.sub(r"\s+", " ", text).strip() + return text.upper() if uppercase else text + + +def _plain_subtitle_blocks(words: Sequence[dict], max_words: int) -> list[list[dict]]: + blocks: list[list[dict]] = [] + pending: list[dict] = [] + for word in words: + if not str(word.get("word", "")).strip(): + continue + pending.append(word) + if len(pending) >= max(1, max_words): + blocks.append(pending) + pending = [] + if pending: + blocks.append(pending) + return blocks + + async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]: """Validate title/subtitle layout for spatial collisions and safe-area containment (collision.validate_titles over every <title> in the file).""" @@ -210,15 +278,7 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() window_end = clip_source_start + clip_duration - clip_words = [ - { - "word": w.get("word", ""), - "start": float(w.get("start", 0.0)) - clip_source_start, - "end": float(w.get("end", 0.0)) - clip_source_start, - } - for w in data.get("words", []) - if clip_source_start <= float(w.get("start", 0.0)) < window_end - ] + clip_words = _words_overlapping_clip(data.get("words", []), clip_source_start, window_end) if not clip_words: skipped.append((name, "no words in clip's source range")) continue @@ -277,7 +337,113 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon return _text_result(result) +async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextContent]: + """Generate static, editable title subtitles from word-level transcripts.""" + model = arguments.get("model", "base") + language = arguments.get("language") + output_dir = arguments.get("output_dir") + clip_filter = arguments.get("clip_name") + + saved = load_plain_subtitle_config() + font = arguments.get("font") or saved["font"] + font_size = int(arguments.get("font_size", saved["font_size"])) + font_color = arguments.get("font_color") or saved["font_color"] + max_words = max(1, int(arguments.get("max_words", saved["max_words"]))) + position_y = float(arguments.get("position_y", saved["position_y"])) + uppercase = bool(arguments.get("uppercase", saved["uppercase"])) + keep_punctuation = bool(arguments.get("keep_punctuation", saved["keep_punctuation"])) + + filepath, output_path, modifier = _setup_modifier(arguments, "_plain_subtitles") + + added: list[tuple[str, int, int]] = [] + skipped: list[tuple[str, str]] = [] + spine_clips = [el for _, el in modifier._iter_spine_clips()] + for el in spine_clips: + name = el.get("name", "") + if clip_filter and name != clip_filter: + continue + src = modifier.resources.get(el.get("ref", ""), {}).get("src", "") + media_path = media_src_to_path(src) + if not media_path or not Path(media_path).is_file(): + skipped.append((name, "media file missing")) + continue + data, reason = _load_or_transcribe(media_path, model, language, output_dir) + if data is None: + skipped.append((name, reason)) + continue + + clip_source_start = modifier.source_file_start(el).to_seconds() + clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() + clip_words = _words_overlapping_clip( + data.get("words", []), clip_source_start, clip_source_start + clip_duration + ) + if not clip_words: + skipped.append((name, "no words in clip's source range")) + continue + + blocks = _plain_subtitle_blocks(clip_words, max_words) + created = 0 + for block in blocks: + parts = [ + _plain_word_text(w.get("word", ""), uppercase=uppercase, keep_punctuation=keep_punctuation) + for w in block + ] + text = " ".join(p for p in parts if p).strip() + if not text: + continue + start = max(0.0, min(float(w.get("start", 0.0)) for w in block)) + end = max(float(w.get("end", start)) for w in block) + duration = max(end - start, modifier.frame_duration_fraction()) + modifier.add_text_title( + el, + text, + offset=f"{start:.6f}s", + duration=f"{duration:.6f}s", + lane=20, + position=f"0 {position_y:g}", + font=font, + font_size=font_size, + font_color=font_color, + bold=True, + face=None, + font_scale=1.0, + size_param=font_size, + ) + created += 1 + if created: + added.append((name, created, len(clip_words))) + + if not added: + text = "# Plain Subtitles\n\nNo subtitles generated — file unchanged (nothing saved)." + if skipped: + text += "\n\n## Skipped Clips\n" + _markdown_table( + ["Clip", "Reason"], [[n, r] for n, r in skipped] + ) + return _text_result(text) + + modifier.save(output_path) + total_titles = sum(lines for _, lines, _ in added) + total_words = sum(words for _, _, words in added) + result = "# Plain Subtitles Generated\n\n## Summary\n" + result += ( + f"- **Clips Captioned**: {len(added)}\n" + f"- **Title Clips**: {total_titles}\n" + f"- **Total Words**: {total_words}\n\n" + ) + result += _markdown_table( + ["Clip", "Title Clips", "Words"], + [[n, str(lines), str(words)] for n, lines, words in added], + ) + if skipped: + result += "\n## Skipped Clips\n" + _markdown_table( + ["Clip", "Reason"], [[n, r] for n, r in skipped] + ) + result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*" + return _text_result(result) + + HANDLERS = { "validate_subtitle_layout": handle_validate_subtitle_layout, "generate_dynamic_subtitles": handle_generate_dynamic_subtitles, + "generate_plain_subtitles": handle_generate_plain_subtitles, } diff --git a/code/server_tools/transcript.py b/code/server_tools/transcript.py index 7a20020..761e6b6 100644 --- a/code/server_tools/transcript.py +++ b/code/server_tools/transcript.py @@ -67,12 +67,12 @@ TOOLS = [ ), Tool( name="remove_filler_words", - description="Cut filler words (um, uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.", + description="Cut filler interjections (uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'um', 'uma', 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.", inputSchema={ "type": "object", "properties": { "filepath": {"type": "string", "description": "Path to FCPXML file"}, - "fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: um, uh, uhh, umm, erm, ehm, mmm, hmm, mhm)"}, + "fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: uh, uhh, umm, erm, ehm, mmm, hmm, mhm; pass um/uma explicitly if desired)"}, "clip_name": {"type": "string", "description": "Only clean the clip with this name"}, "model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"}, "padding": {"type": "number", "default": 0.02, "description": "Seconds to widen each cut on both sides (0-2, default 0.02)"}, diff --git a/code/server_tools/voice.py b/code/server_tools/voice.py index 8d63d1f..edbc346 100644 --- a/code/server_tools/voice.py +++ b/code/server_tools/voice.py @@ -348,8 +348,9 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]: language = arguments.get("language") token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers() + output_dir = arguments.get("output_dir") - transcript, reason = _load_or_transcribe(media_path, model, language) + transcript, reason = _load_or_transcribe(media_path, model, language, output_dir) if transcript is None: return _text_result( f"# Voice Timeline\n\nCould not obtain a transcript " @@ -365,9 +366,10 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]: weights=EmphasisWeights.from_dict(config["emphasis_weights"]), peak_percentile=config["peak_percentile"], emphasis_floor=config["emphasis_floor"], + emotion_enabled=config["emotion_enabled"], + emotion_sensitivity=config["emotion_sensitivity"], ) - output_dir = arguments.get("output_dir") json_path = Path(_validate_output_path( str(voice_timeline_path(media_path, output_dir)), anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent), @@ -398,6 +400,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]: "yes" if layers["acoustics"] else "FAILED — every acoustic value is 0", ], ["Speakers", "yes" if layers["speakers"] else "not run — single default speaker"], + ["Emotion", "yes" if layers.get("emotion") else "not run"], ], ) + "\n" diff --git a/code/tests/test_dynamic_subtitles.py b/code/tests/test_dynamic_subtitles.py index 5fd6c45..527c48b 100644 --- a/code/tests/test_dynamic_subtitles.py +++ b/code/tests/test_dynamic_subtitles.py @@ -29,6 +29,7 @@ from fcpxml.text_layout import ( ink_extent, ) from fcpxml.writer import FCPXMLModifier +from server_tools.subtitles import _words_overlapping_clip SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml" def font_points(style) -> float: @@ -51,6 +52,21 @@ WORDS = [ ] +def test_words_overlapping_clip_keeps_word_that_starts_just_before_in_point(): + words = [ + {"word": "Aquela", "start": 2.03, "end": 2.69}, + {"word": "mama", "start": 2.69, "end": 2.89}, + {"word": "fora", "start": 10.0, "end": 10.2}, + ] + + clip_words = _words_overlapping_clip(words, 2.0437166666666666, 3.0) + + assert clip_words == [ + {"word": "Aquela", "start": 0.0, "end": pytest.approx(0.6462833333333332)}, + {"word": "mama", "start": pytest.approx(0.6462833333333332), "end": pytest.approx(0.8462833333333334)}, + ] + + @pytest.fixture def temp_fcpxml(): with tempfile.NamedTemporaryFile(suffix=".fcpxml", delete=False) as f: diff --git a/code/tests/test_phrase_review.py b/code/tests/test_phrase_review.py new file mode 100644 index 0000000..a18bbc4 --- /dev/null +++ b/code/tests/test_phrase_review.py @@ -0,0 +1,479 @@ +"""Tests for the phrase review model (voice timeline + AI actions → editable script).""" + +import json + +import pytest + +from fcpxml.phrase_review import ( + TRACK_BACKSTAGE, + TRACK_SCRIPT, + ZOOM_SCALE_BY_LEVEL, + build_phrase_review, + load_phrase_review, + merge_saved_decisions, + phrase_review_to_actions, + resolve_source, + review_paths, + save_phrase_review, + snap_to_words, +) + + +def _words(spans, emphasis=0.0): + return [ + { + "text": f"w{i}", + "start": start, + "end": end, + "energy": 0.5, + "emphasis": emphasis, + } + for i, (start, end) in enumerate(spans) + ] + + +def _timeline(segments): + return {"source": "/tmp/take.mov", "speakers": ["SPEAKER_00"], "segments": segments} + + +def _segment(start, end, text="linha", peak=0.1, take_boundary=False, words=None): + return { + "start": start, + "end": end, + "text": text, + "speaker": "SPEAKER_00", + "peak_emphasis": peak, + "take_boundary": take_boundary, + "gap_before": 0.0, + "words": words if words is not None else _words([(start, end)]), + } + + +class TestBuildFromAcoustics: + def test_emphasis_levels_follow_peak_thresholds(self): + review = build_phrase_review( + _timeline( + [ + _segment(0, 1, peak=0.10), + _segment(1, 2, peak=0.30), + _segment(2, 3, peak=0.50), + _segment(3, 4, peak=0.90), + ] + ) + ) + assert [p["emphasis"] for p in review["phrases"]] == [0, 1, 2, 3] + + def test_every_phrase_starts_active_without_actions(self): + review = build_phrase_review(_timeline([_segment(0, 1), _segment(1, 2)])) + assert all(p["active"] for p in review["phrases"]) + assert all(p["track"] == TRACK_SCRIPT for p in review["phrases"]) + + def test_carries_text_speaker_and_words(self): + review = build_phrase_review(_timeline([_segment(0, 2, text=" olá ")])) + phrase = review["phrases"][0] + assert phrase["text"] == "olá" + assert phrase["speaker"] == "SPEAKER_00" + assert phrase["words"][0]["text"] == "w0" + assert review["duration"] == 2.0 + + +class TestCutsDeactivate: + def test_fully_cut_phrase_is_inactive(self): + review = build_phrase_review( + _timeline([_segment(0, 2), _segment(2, 4)]), + {"actions": [{"kind": "cut", "start": 0, "end": 2, "reason": "gaguejou"}]}, + ) + assert review["phrases"][0]["active"] is False + assert review["phrases"][0]["reason"] == "gaguejou" + assert review["phrases"][1]["active"] is True + + def test_small_overlap_keeps_the_phrase(self): + # 0.2s off a 2s line is a trim, not a removal. + review = build_phrase_review( + _timeline([_segment(1, 3, words=_words([(1, 1.2), (1.2, 3)]))]), + {"actions": [{"kind": "cut", "start": 0.5, "end": 1.2}]}, + ) + assert review["phrases"][0]["active"] is True + + def test_majority_overlap_deactivates(self): + review = build_phrase_review( + _timeline([_segment(0, 2)]), + {"actions": [{"kind": "cut", "start": 0, "end": 1.5}]}, + ) + assert review["phrases"][0]["active"] is False + + def test_inactive_after_take_boundary_is_backstage(self): + review = build_phrase_review( + _timeline([_segment(10, 12, take_boundary=True)]), + {"actions": [{"kind": "cut", "start": 10, "end": 12}]}, + ) + assert review["phrases"][0]["track"] == TRACK_BACKSTAGE + + +class TestTrimFromPartialCuts: + def test_head_cut_becomes_a_trim_snapped_to_a_word(self): + review = build_phrase_review( + _timeline([_segment(1, 4, words=_words([(1, 1.4), (1.4, 4)]))]), + {"actions": [{"kind": "cut", "start": 0.8, "end": 1.35}]}, + ) + phrase = review["phrases"][0] + assert phrase["active"] is True + assert phrase["trim_start"] == 1.4 # snapped to the second word's start + assert phrase["trim_end"] == 4.0 + + def test_tail_cut_becomes_a_trim(self): + review = build_phrase_review( + _timeline([_segment(0, 3, words=_words([(0, 2.5), (2.5, 3)]))]), + {"actions": [{"kind": "cut", "start": 2.6, "end": 3.5}]}, + ) + phrase = review["phrases"][0] + assert phrase["trim_start"] == 0.0 + assert phrase["trim_end"] == 2.5 + + def test_untouched_phrase_trims_to_its_own_bounds(self): + review = build_phrase_review(_timeline([_segment(0, 2)])) + phrase = review["phrases"][0] + assert (phrase["trim_start"], phrase["trim_end"]) == (0.0, 2.0) + + +class TestAIDirectionWins: + def test_zoom_action_sets_the_level_over_the_heuristic(self): + review = build_phrase_review( + _timeline([_segment(0, 2, peak=0.05)]), + { + "actions": [ + { + "kind": "zoom", + "start": 0.5, + "end": 0.9, + "params": {"scale": 1.5}, + "reason": "virada da história", + } + ] + }, + ) + phrase = review["phrases"][0] + assert phrase["emphasis"] == 3 + assert phrase["reason"] == "virada da história" + + def test_text_action_marks_emphasis(self): + review = build_phrase_review( + _timeline([_segment(0, 2, peak=0.0)]), + { + "actions": [ + { + "kind": "text", + "start": 0.5, + "end": 1.0, + "params": {"content": "3x mais rápido"}, + } + ] + }, + ) + assert review["phrases"][0]["emphasis"] == 2 + + def test_highest_level_wins_when_several_actions_overlap(self): + review = build_phrase_review( + _timeline([_segment(0, 4)]), + { + "actions": [ + {"kind": "zoom", "start": 0.2, "end": 0.5, "params": {"scale": 1.15}}, + {"kind": "zoom", "start": 2.0, "end": 2.4, "params": {"scale": 1.5}}, + ] + }, + ) + assert review["phrases"][0]["emphasis"] == 3 + + def test_malformed_rows_are_reported_not_fatal(self): + review = build_phrase_review( + _timeline([_segment(0, 2)]), + {"actions": [{"kind": "voar", "start": 0, "end": 1}]}, + ) + assert len(review["errors"]) == 1 + assert review["phrases"][0]["active"] is True + + +class TestBackToActions: + def test_inactive_phrase_becomes_a_cut(self): + review = build_phrase_review(_timeline([_segment(0, 2), _segment(2, 4)])) + review["phrases"][0]["active"] = False + result = phrase_review_to_actions(review) + cuts = [a for a in result["actions"] if a["kind"] == "cut"] + assert len(cuts) == 1 + assert (cuts[0]["start"], cuts[0]["end"]) == (0.0, 2.0) + + def test_emphasis_becomes_a_zoom_and_a_span(self): + review = build_phrase_review(_timeline([_segment(0, 2)])) + review["phrases"][0]["emphasis"] = 2 + result = phrase_review_to_actions(review) + zooms = [a for a in result["actions"] if a["kind"] == "zoom"] + assert zooms[0]["params"]["scale"] == ZOOM_SCALE_BY_LEVEL[2] + assert result["emphasis_spans"] == [ + {"start": 0.0, "end": 2.0, "level": 2, "text": "linha"} + ] + + def test_level_zero_produces_nothing(self): + review = build_phrase_review(_timeline([_segment(0, 2)])) + review["phrases"][0]["emphasis"] = 0 + result = phrase_review_to_actions(review) + assert result["actions"] == [] + assert result["emphasis_spans"] == [] + + def test_trim_becomes_head_and_tail_cuts(self): + review = build_phrase_review( + _timeline([_segment(0, 4, words=_words([(0, 1), (1, 3), (3, 4)]))]) + ) + review["phrases"][0]["trim_start"] = 1.0 + review["phrases"][0]["trim_end"] = 3.0 + result = phrase_review_to_actions(review) + spans = [(a["start"], a["end"]) for a in result["actions"] if a["kind"] == "cut"] + assert spans == [(0.0, 1.0), (3.0, 4.0)] + + def test_inactive_phrase_is_cut_whole_ignoring_its_trim(self): + review = build_phrase_review(_timeline([_segment(0, 4)])) + review["phrases"][0].update({"active": False, "trim_start": 1.0, "trim_end": 3.0}) + result = phrase_review_to_actions(review) + assert [(a["start"], a["end"]) for a in result["actions"]] == [(0.0, 4.0)] + + def test_zoom_follows_the_trimmed_span(self): + review = build_phrase_review(_timeline([_segment(0, 4)])) + review["phrases"][0].update({"emphasis": 1, "trim_start": 1.0, "trim_end": 3.0}) + result = phrase_review_to_actions(review) + zoom = next(a for a in result["actions"] if a["kind"] == "zoom") + assert (zoom["start"], zoom["end"]) == (1.0, 3.0) + + def test_impossible_trim_is_ignored(self): + review = build_phrase_review(_timeline([_segment(0, 4)])) + review["phrases"][0].update({"trim_start": 3.0, "trim_end": 1.0}) + result = phrase_review_to_actions(review) + assert result["actions"] == [] + + def test_emphasis_out_of_range_is_clamped(self): + review = build_phrase_review(_timeline([_segment(0, 2)])) + review["phrases"][0]["emphasis"] = 99 + result = phrase_review_to_actions(review) + assert result["actions"][0]["params"]["scale"] == ZOOM_SCALE_BY_LEVEL[3] + + def test_rows_that_make_no_sense_are_skipped(self): + result = phrase_review_to_actions( + {"phrases": ["nope", {"start": 5, "end": 1}, {"start": 0, "end": 1}]} + ) + assert result["actions"] == [] + + +class TestManualZooms: + def test_manual_zoom_becomes_an_action_without_a_scale(self): + review = build_phrase_review(_timeline([_segment(0, 10)])) + review["zooms"] = [{"start": 2.0, "end": 4.0}] + result = phrase_review_to_actions(review) + zoom = next(a for a in result["actions"] if a["kind"] == "zoom") + assert (zoom["start"], zoom["end"]) == (2.0, 4.0) + # Sem scale: o aplicador usa o zoom_scale configurado pelo usuário. + assert "scale" not in zoom["params"] + + def test_zoom_shorter_than_the_ramp_is_refused(self): + review = build_phrase_review(_timeline([_segment(0, 10)])) + review["zooms"] = [{"start": 2.0, "end": 2.1}] + assert phrase_review_to_actions(review)["actions"] == [] + + def test_manual_zoom_coexists_with_phrase_emphasis(self): + review = build_phrase_review(_timeline([_segment(0, 10)])) + review["phrases"][0]["emphasis"] = 2 + review["zooms"] = [{"start": 2.0, "end": 4.0}] + zooms = [a for a in phrase_review_to_actions(review)["actions"] if a["kind"] == "zoom"] + assert len(zooms) == 2 + + def test_malformed_zoom_rows_are_skipped(self): + review = build_phrase_review(_timeline([_segment(0, 10)])) + review["zooms"] = ["nope", {"start": 5}, {"start": 4, "end": 1}] + assert phrase_review_to_actions(review)["actions"] == [] + + def test_saved_zooms_are_restored(self): + review = merge_saved_decisions( + build_phrase_review(_timeline([_segment(0, 10)])), + {"phrases": [], "zooms": [{"start": 1.0, "end": 3.0}]}, + ) + assert review["zooms"] == [{"start": 1.0, "end": 3.0}] + + def test_new_review_starts_with_no_manual_zooms(self): + assert build_phrase_review(_timeline([_segment(0, 2)]))["zooms"] == [] + + +class TestRoundTrip: + def test_review_survives_actions_and_back(self): + timeline = _timeline( + [_segment(0, 2, peak=0.9), _segment(2, 4), _segment(4, 6, peak=0.5)] + ) + first = build_phrase_review(timeline) + first["phrases"][1]["active"] = False + actions = phrase_review_to_actions(first) + + second = build_phrase_review(timeline, actions) + assert [p["active"] for p in second["phrases"]] == [True, False, True] + assert [p["emphasis"] for p in second["phrases"]] == [3, 0, 2] + + +class TestSnapToWords: + def test_snaps_to_the_nearest_start(self): + words = _words([(1.0, 1.5), (1.5, 2.0)]) + assert snap_to_words(1.6, words, 1.6, "in") == 1.5 + + def test_snaps_to_the_nearest_end(self): + words = _words([(1.0, 1.5), (1.5, 2.0)]) + assert snap_to_words(1.9, words, 1.9, "out") == 2.0 + + def test_falls_back_without_word_timings(self): + assert snap_to_words(1.2, [], 3.4, "in") == 3.4 + + +class TestResolveSource: + def test_finds_the_media_beside_its_timeline(self, tmp_path): + media = tmp_path / "take.mov" + media.write_bytes(b"0") + timeline = tmp_path / "take_voice_timeline.json" + assert resolve_source("take.mov", str(timeline)) == str(media) + + def test_falls_back_to_the_project_folder(self, tmp_path): + media_dir = tmp_path / "midia" + media_dir.mkdir() + media = media_dir / "take.mov" + media.write_bytes(b"0") + timeline = tmp_path / "json" / "take_voice_timeline.json" + assert resolve_source("take.mov", str(timeline), [str(media_dir)]) == str(media) + + def test_absolute_path_is_used_as_is(self, tmp_path): + media = tmp_path / "take.mov" + media.write_bytes(b"0") + assert resolve_source(str(media), "") == str(media) + + def test_missing_media_resolves_to_empty(self, tmp_path): + assert resolve_source("take.mov", str(tmp_path / "x_voice_timeline.json")) == "" + + def test_stale_absolute_path_still_finds_the_file_by_name(self, tmp_path): + # The fixture's source is an absolute path that no longer exists (the + # everyday case: the project moved). Falling back to the file name next + # to the timeline is what keeps the preview working after a move. + media = tmp_path / "take.mov" + media.write_bytes(b"0") + assert resolve_source("/tmp/gone/take.mov", str(tmp_path / "t.json")) == str(media) + + def test_review_carries_the_resolved_path(self, tmp_path): + media = tmp_path / "take.mov" + media.write_bytes(b"0") + review = build_phrase_review( + {**_timeline([_segment(0, 1)]), "source": "take.mov"}, + voice_timeline_path=str(tmp_path / "take_voice_timeline.json"), + ) + assert review["source_path"] == str(media) + + def test_review_without_media_reports_no_path(self, tmp_path): + review = build_phrase_review( + _timeline([_segment(0, 1)]), + voice_timeline_path=str(tmp_path / "take_voice_timeline.json"), + ) + assert review["source_path"] == "" + + +class TestEmotion: + def test_segment_emotion_reaches_the_phrase(self): + segment = _segment(0, 2) + segment["emotion"] = "excited" + segment["emotion_confidence"] = 0.72 + review = build_phrase_review(_timeline([segment])) + assert review["phrases"][0]["emotion"] == "excited" + assert review["phrases"][0]["emotion_confidence"] == 0.72 + + def test_defaults_to_neutral_when_absent(self): + review = build_phrase_review(_timeline([_segment(0, 2)])) + assert review["phrases"][0]["emotion"] == "neutral" + assert review["phrases"][0]["emotion_confidence"] == 0.0 + + def test_availability_comes_from_the_analysis_layers(self): + assert build_phrase_review(_timeline([_segment(0, 1)]))["emotion_available"] is False + timeline = {**_timeline([_segment(0, 1)]), "layers": {"emotion": True}} + assert build_phrase_review(timeline)["emotion_available"] is True + + +class TestMergeSavedDecisions: + def test_saved_decisions_win_over_the_derivation(self): + timeline = _timeline([_segment(0, 2, peak=0.9), _segment(2, 4)]) + saved = { + "phrases": [ + {"index": 0, "start": 0.0, "emphasis": 0, "active": False, + "track": TRACK_BACKSTAGE, "text": "corrigido"}, + ] + } + review = merge_saved_decisions(build_phrase_review(timeline), saved) + first = review["phrases"][0] + assert (first["emphasis"], first["active"]) == (0, False) + assert first["track"] == TRACK_BACKSTAGE + assert first["text"] == "corrigido" + assert review["phrases"][1]["active"] is True + + def test_fresh_analysis_fields_are_not_overwritten(self): + segment = _segment(0, 2, peak=0.9) + segment["emotion"] = "tense" + review = merge_saved_decisions( + build_phrase_review(_timeline([segment])), + {"phrases": [{"index": 0, "start": 0.0, "emphasis": 1}]}, + ) + assert review["phrases"][0]["emotion"] == "tense" + assert review["phrases"][0]["peak_emphasis"] == 0.9 + + def test_decision_is_dropped_when_the_line_moved(self): + review = merge_saved_decisions( + build_phrase_review(_timeline([_segment(10, 12, peak=0.9)])), + {"phrases": [{"index": 0, "start": 0.0, "active": False}]}, + ) + assert review["phrases"][0]["active"] is True + + def test_saved_trim_is_restored(self): + review = merge_saved_decisions( + build_phrase_review(_timeline([_segment(0, 4)])), + {"phrases": [{"index": 0, "start": 0.0, "trim_start": 1.0, "trim_end": 3.0}]}, + ) + assert (review["phrases"][0]["trim_start"], review["phrases"][0]["trim_end"]) == (1.0, 3.0) + + def test_impossible_saved_trim_is_ignored(self): + review = merge_saved_decisions( + build_phrase_review(_timeline([_segment(0, 4)])), + {"phrases": [{"index": 0, "start": 0.0, "trim_start": 9.0, "trim_end": 12.0}]}, + ) + assert (review["phrases"][0]["trim_start"], review["phrases"][0]["trim_end"]) == (0.0, 4.0) + + def test_no_saved_review_is_a_no_op(self): + review = build_phrase_review(_timeline([_segment(0, 2)])) + assert merge_saved_decisions(review, None) is review + + +class TestPersistence: + def test_paths_are_named_after_the_timeline(self, tmp_path): + timeline_path = tmp_path / "take_voice_timeline.json" + review_path, actions_path = review_paths(str(timeline_path)) + assert review_path.name == "take_phrase_review.json" + assert actions_path.name == "take_phrase_actions.json" + + def test_save_writes_both_files_and_load_reads_it_back(self, tmp_path): + timeline_path = tmp_path / "take_voice_timeline.json" + review = build_phrase_review(_timeline([_segment(0, 2)])) + review["phrases"][0]["emphasis"] = 3 + + review_path, actions_path = save_phrase_review(str(timeline_path), review) + assert review_path.is_file() and actions_path.is_file() + + written = json.loads(actions_path.read_text(encoding="utf-8")) + assert written["actions"][0]["kind"] == "zoom" + + assert load_phrase_review(str(timeline_path))["phrases"][0]["emphasis"] == 3 + + def test_load_returns_none_when_absent_or_broken(self, tmp_path): + timeline_path = tmp_path / "take_voice_timeline.json" + assert load_phrase_review(str(timeline_path)) is None + + review_path, _ = review_paths(str(timeline_path)) + review_path.write_text("{ not json", encoding="utf-8") + assert load_phrase_review(str(timeline_path)) is None + + +if __name__ == "__main__": + pytest.main([__file__, "-v"]) diff --git a/code/tests/test_transcribe.py b/code/tests/test_transcribe.py index cf52913..805b232 100755 --- a/code/tests/test_transcribe.py +++ b/code/tests/test_transcribe.py @@ -40,7 +40,7 @@ WORDS = [ w("the", 2.5, 2.6), w("show", 2.65, 3.0), w("is", 3.05, 3.15), - w("um", 3.2, 3.5), + w("uh", 3.2, 3.5), w("great.", 3.6, 4.0), ] @@ -60,7 +60,7 @@ class TestNormalizeWord: class TestFindPhraseSpans: def test_single_word_multiple_hits(self): spans = find_phrase_spans(WORDS, "um") - assert spans == [(0.4, 0.6), (3.2, 3.5)] + assert spans == [(0.4, 0.6)] def test_multi_word_phrase(self): spans = find_phrase_spans(WORDS, "welcome to the show") @@ -84,9 +84,13 @@ class TestFindPhraseSpans: class TestFindFillerSpans: def test_default_fillers(self): spans = find_filler_spans(WORDS) - assert (0.4, 0.6) in spans + assert (0.4, 0.6) not in spans assert (3.2, 3.5) in spans + def test_um_can_still_be_explicit(self): + spans = find_filler_spans(WORDS, fillers=("um",)) + assert spans == [(0.4, 0.6)] + def test_multi_word_filler(self): spans = find_filler_spans(WORDS, fillers=("you know",)) assert spans == [(2.0, 2.4)] @@ -96,7 +100,10 @@ class TestFindFillerSpans: assert spans == sorted(spans) def test_defaults_are_conservative(self): - # "like" and "so" are speech, not noise — must not be default-cut. + # "um", "uma", "like" and "so" are speech, not noise — must not be + # default-cut. + assert "um" not in DEFAULT_FILLERS + assert "uma" not in DEFAULT_FILLERS assert "like" not in DEFAULT_FILLERS assert "so" not in DEFAULT_FILLERS @@ -269,11 +276,12 @@ class TestRemoveFillerWordsHandler: assert "_defillered" in result[0].text segments = _spine_segments(str(tmp_path / "project_defillered.fcpxml")) - # um (0-0.5) trims the clip head, uh (3-3.5) splits -> 2 segments, ~7s total + # Only uh (3-3.5) is removed by default; Portuguese "um" is preserved + # because it is often grammatical speech ("de um jeito"). assert len(segments) == 2 total = sum(c.duration.seconds for c in segments) - assert total == pytest.approx(7.0, abs=0.1) - assert segments[0].source_start.seconds == pytest.approx(0.5, abs=0.05) + assert total == pytest.approx(7.5, abs=0.1) + assert segments[0].source_start.seconds == pytest.approx(0.0, abs=0.05) assert segments[1].source_start.seconds == pytest.approx(3.5, abs=0.05) async def test_no_fillers_found_saves_nothing(self, tmp_path): diff --git a/code/tests/test_voice_actions.py b/code/tests/test_voice_actions.py index 7ffc797..622b9d6 100644 --- a/code/tests/test_voice_actions.py +++ b/code/tests/test_voice_actions.py @@ -76,9 +76,14 @@ class TestParseActions: class TestZoomValidation: - def test_default_scale_when_absent(self): + def test_absent_scale_is_left_absent(self): + # The parser no longer stamps a default: an omitted scale must reach the + # applier untouched so it can fall back to the user's configured + # `zoom_scale` (see server_tools/_shared.py). Filling one in here would + # silently override that setting for every action the model sends + # without an explicit scale. actions, _ = parse_actions([{"kind": "zoom", "start": 1.0, "end": 2.0}]) - assert actions[0].params["scale"] == 1.3 + assert "scale" not in actions[0].params def test_rejects_scale_below_one(self): _, errors = parse_actions([ diff --git a/code/tests/test_voice_actions_tool.py b/code/tests/test_voice_actions_tool.py index f8517e6..908478f 100644 --- a/code/tests/test_voice_actions_tool.py +++ b/code/tests/test_voice_actions_tool.py @@ -94,6 +94,32 @@ class TestApplyVoiceActionsHandler: texts = [t.text for t in titles[0].iter() if t.text] assert any("SEGURANÇA" in t for t in texts) + async def test_text_callout_defaults_fit_the_frame(self, project): + from fcpxml.writer import FCPXMLModifier + from server import handle_apply_voice_actions + + await handle_apply_voice_actions({ + "filepath": str(project), + "actions": [{ + "kind": "text", "start": 3.0, "end": 4.0, + "params": {"content": "PRÓTESES DE SILICONE"}, + }], + }) + + modifier = FCPXMLModifier(str(_out(project))) + report = modifier.validate_subtitle_layout() + assert report["summary"]["outside_frame"] == 0 + + title = modifier.root.find(".//title") + style = title.find("text-style-def/text-style") + position = next( + p.get("value") + for p in title.findall("param") + if p.get("name") == "Position" + ) + assert float(style.get("fontSize")) < 530 + assert position != "0 0" + async def test_applies_marker(self, project): from server import handle_apply_voice_actions diff --git a/code/tests/test_voice_timeline.py b/code/tests/test_voice_timeline.py index d8f4dd8..3d6dedb 100644 --- a/code/tests/test_voice_timeline.py +++ b/code/tests/test_voice_timeline.py @@ -11,6 +11,7 @@ import pytest from fcpxml.voice_timeline import ( VOICE_TIMELINE_VERSION, + annotate_emotions, build_voice_timeline, enrich_words, load_voice_timeline, @@ -60,6 +61,14 @@ class TestEnrichWords: assert all(w["energy_norm"] == 0.0 for w in enriched) assert all(w["pitch_delta"] == 0.0 for w in enriched) + def test_emotion_labels_are_added_when_enabled(self): + enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY) + emotional = annotate_emotions(enriched, enabled=True, sensitivity=0.1) + loudest = max(emotional, key=lambda w: w["arousal"]) + assert loudest["word"] == "seguranca" + assert loudest["emotion"] in {"excited", "tense", "neutral"} + assert 0.0 <= loudest["emotion_confidence"] <= 1.0 + class TestBuildVoiceTimeline: @pytest.fixture @@ -74,6 +83,7 @@ class TestBuildVoiceTimeline: for key in ("version", "source", "language", "scales", "summary", "speakers", "segments"): assert key in timeline assert timeline["version"] == VOICE_TIMELINE_VERSION + assert "emotion" in timeline["layers"] def test_scales_document_every_word_metric(self, timeline): word = timeline["segments"][0]["words"][0] @@ -106,6 +116,8 @@ class TestBuildVoiceTimeline: quiet_segment = timeline["segments"][0] assert loud_segment["avg_energy"] > quiet_segment["avg_energy"] assert loud_segment["peak_emphasis"] >= max(w["emphasis"] for w in loud_segment["words"]) + assert "emotion" in loud_segment + assert "arousal" in loud_segment def test_peak_moments_are_sorted_by_emphasis(self, timeline): peaks = timeline["summary"]["peak_moments"] diff --git a/code/tests/test_voice_timeline_tool.py b/code/tests/test_voice_timeline_tool.py index c89ad31..42f5acd 100644 --- a/code/tests/test_voice_timeline_tool.py +++ b/code/tests/test_voice_timeline_tool.py @@ -94,6 +94,10 @@ class TestBuildVoiceTimelineHandler: }, "emotion_enabled": False, "emotion_sensitivity": 0.5, + "zoom_scale": 1.3, + "zoom_mode": "in_out", + "zoom_ease_in": 0.25, + "zoom_ease_out": 0.04, } monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(0.01))