feat: etapa 5 do assistente — revisão de ênfases com timeline

Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-19 21:29:27 -04:00
co-authored by Claude Opus 5
parent e7748c2c58
commit 1bebee4359
31 changed files with 4622 additions and 83 deletions
+265 -16
View File
@@ -51,6 +51,23 @@ Commands:
`refine_voice_timeline` never has to reopen the audio later. `refine_voice_timeline` never has to reopen the audio later.
-> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."} -> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."}
build_phrase_review {"voice_timeline": "..._voice_timeline.json",
"actions": {...}|[...]|null, "fresh": false}
The reviewable script for the wizard's emphasis step: every phrase with
the AI's decision already applied (active/emphasis/trim). A review saved
earlier for the same timeline is returned as-is unless `fresh` is true.
-> {"ok": true, "reused": bool, "source", "duration", "speakers",
"phrases": [{index, start, end, trim_start, trim_end, text, speaker,
active, emphasis (0-3), track, peak_emphasis,
take_boundary, gap_before, reason, words}],
"errors": [...]}
save_phrase_review {"voice_timeline": "...", "phrases": [...], "source": "...",
"duration": 0.0, "speakers": [...]}
Writes _phrase_review.json plus the _phrase_actions.json derived from it.
-> {"ok": true, "review_path", "actions_path", "emphasis_count",
"removed_count"}
dynamic_subtitle_config {} dynamic_subtitle_config {}
-> {"ok": true, "band_height", "block_center_y", "line_gap", "font", -> {"ok": true, "band_height", "block_center_y", "line_gap", "font",
"font_size", "emphasis_font", "emphasis_face", "emphasis_size", "font_size", "emphasis_font", "emphasis_face", "emphasis_size",
@@ -119,6 +136,10 @@ Commands:
-> {"ok": true, "diarization": bool, "diarization_message": "...", -> {"ok": true, "diarization": bool, "diarization_message": "...",
"num_speakers": "..."} "num_speakers": "..."}
acoustics_capability
Whether librosa (pitch/energy for voice analysis) is installed.
-> {"ok": true, "available": bool, "message": "..."}
voice_analysis voice_analysis
-> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85, -> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85,
"emphasis_weights": {...}, "emotion_enabled": false, "emphasis_weights": {...}, "emotion_enabled": false,
@@ -165,6 +186,7 @@ from fcpxml.model_manager import ( # noqa: E402
load_dynamic_subtitle_config, load_dynamic_subtitle_config,
load_hf_token, load_hf_token,
load_num_speakers, load_num_speakers,
load_plain_subtitle_config,
load_project_config, load_project_config,
load_selected_model, load_selected_model,
load_silence_config, load_silence_config,
@@ -175,6 +197,7 @@ from fcpxml.model_manager import ( # noqa: E402
save_hf_token, save_hf_token,
save_models_dir, save_models_dir,
save_num_speakers, save_num_speakers,
save_plain_subtitle_config,
save_project_config, save_project_config,
save_selected_model, save_selected_model,
save_silence_config, save_silence_config,
@@ -200,6 +223,28 @@ def _derived_output(path: str, suffix: str, args: dict) -> str:
from server import generate_output_path from server import generate_output_path
return generate_output_path(path, suffix) return generate_output_path(path, suffix)
def _is_no_change_message(message: str) -> bool:
"""Whether a tool completed cleanly without needing to save a new file."""
text = message.lower()
return any(
token in text
for token in (
"no cuts to make",
"no silence",
"file unchanged",
"nothing saved",
)
)
def _emit_no_change_or_error(path: str, message: str) -> int:
if _is_no_change_message(message):
_emit({"ok": True, "path": path, "unchanged": True, "message": message})
return 0
_emit({"ok": False, "error": message})
return 1
# Download cancellation events, keyed by model name. # Download cancellation events, keyed by model name.
_CANCEL: dict[str, threading.Event] = {} _CANCEL: dict[str, threading.Event] = {}
_LOCK = threading.Lock() _LOCK = threading.Lock()
@@ -243,6 +288,42 @@ def _save_json_atomic(path: Path, data: Any) -> None:
json.load(fh) json.load(fh)
def _project_media_paths(path: str) -> list[str]:
proj = parse_fcpxml(path)
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
media_paths: list[str] = []
if tl is not None:
for clip in getattr(tl, "clips", []):
mp = media_src_to_path(clip.media_path or "")
if mp and Path(mp).is_file() and mp not in media_paths:
media_paths.append(mp)
return media_paths
def _voice_timeline_json_path(media_path: str, output_dir: str = "") -> Path:
p = Path(media_path)
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
return directory / f"{p.stem}_voice_timeline.json"
return p.with_name(p.stem + "_voice_timeline.json")
def _load_cached_voice_timeline(json_path: Path, media_path: str) -> dict | None:
try:
with open(json_path, encoding="utf-8") as fh:
data = json.load(fh)
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
return None
if not isinstance(data, dict):
return None
if data.get("source") != Path(media_path).name:
return None
if not isinstance(data.get("segments"), list):
return None
return data
# ── commands ──────────────────────────────────────────────────────────────── # ── commands ────────────────────────────────────────────────────────────────
@@ -350,8 +431,7 @@ def cmd_remove_silences(args: dict) -> int:
contents = asyncio.run(handle_remove_media_silence({**args, "filepath": path, "output_path": output})) contents = asyncio.run(handle_remove_media_silence({**args, "filepath": path, "output_path": output}))
message = "\n".join(getattr(content, "text", str(content)) for content in contents) message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists(): if not Path(output).exists():
_emit({"ok": False, "error": message}) return _emit_no_change_or_error(path, message)
return 1
_emit({"ok": True, "path": output, "message": message}) _emit({"ok": True, "path": output, "message": message})
return 0 return 0
except Exception as exc: except Exception as exc:
@@ -398,8 +478,7 @@ def cmd_remove_filler_words(args: dict) -> int:
contents = asyncio.run(handle_remove_filler_words({**args, "filepath": path, "output_path": output})) contents = asyncio.run(handle_remove_filler_words({**args, "filepath": path, "output_path": output}))
message = "\n".join(getattr(content, "text", str(content)) for content in contents) message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists(): if not Path(output).exists():
_emit({"ok": False, "error": message}) return _emit_no_change_or_error(path, message)
return 1
_emit({"ok": True, "path": output, "message": message}) _emit({"ok": True, "path": output, "message": message})
return 0 return 0
except Exception as exc: except Exception as exc:
@@ -454,6 +533,29 @@ def cmd_generate_dynamic_subtitles(args: dict) -> int:
return 1 return 1
def cmd_generate_plain_subtitles(args: dict) -> int:
"""Generate simple static editable subtitle title clips."""
path = str(args.get("path", ""))
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
try:
from server import handle_generate_plain_subtitles
output = _derived_output(path, "_plain_subtitles", args)
contents = asyncio.run(
handle_generate_plain_subtitles({**args, "filepath": path, "output_path": output})
)
message = "\n".join(getattr(content, "text", str(content)) for content in contents)
if not Path(output).exists():
return _emit_no_change_or_error(path, message)
_emit({"ok": True, "path": output, "message": message})
return 0
except Exception as exc:
_emit({"ok": False, "error": str(exc)})
return 1
def cmd_add_zoom(args: dict) -> int: def cmd_add_zoom(args: dict) -> int:
"""Add an ease-in/ease-out punch-in zoom to one clip.""" """Add an ease-in/ease-out punch-in zoom to one clip."""
path = str(args.get("path", "")) path = str(args.get("path", ""))
@@ -720,17 +822,10 @@ def cmd_analyze_voice(args: dict) -> int:
num_speakers = str(args.get("num_speakers") or load_num_speakers() or "") num_speakers = str(args.get("num_speakers") or load_num_speakers() or "")
try: try:
proj = parse_fcpxml(path) media_paths = _project_media_paths(path)
except Exception as exc: except Exception as exc:
_emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"}) _emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
return 1 return 1
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
media_paths: list[str] = []
if tl is not None:
for clip in getattr(tl, "clips", []):
mp = media_src_to_path(clip.media_path or "")
if mp and Path(mp).is_file() and mp not in media_paths:
media_paths.append(mp)
if not media_paths: if not media_paths:
_emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."}) _emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."})
return 1 return 1
@@ -738,17 +833,41 @@ def cmd_analyze_voice(args: dict) -> int:
from server import handle_build_voice_timeline from server import handle_build_voice_timeline
messages: list[str] = [] messages: list[str] = []
output_dir = str(args.get("output_dir") or "").strip()
existing: list[Path] = []
for mp in media_paths: for mp in media_paths:
timeline_path = _voice_timeline_json_path(mp, output_dir)
if _load_cached_voice_timeline(timeline_path, mp) is not None:
existing.append(timeline_path)
if existing and len(existing) == len(media_paths) and not bool(args.get("force_reprocess", False)):
message = "# Voice Timeline Cache\n\n"
message += "Reaproveitando análise de voz existente. Nada foi reprocessado.\n\n"
for timeline_path in existing:
message += f"- **Timeline JSON**: {timeline_path}\n"
_emit({
"ok": True,
"path": path,
"reused": True,
"timelines": [str(p) for p in existing],
"message": message,
})
return 0
for mp in media_paths:
transcript_path = _transcript_json_path(mp, output_dir)
reused_prefix = ""
if _load_cached_transcript(transcript_path) is not None:
reused_prefix = f"# Cache\n\nReaproveitando transcrição existente: `{transcript_path}`\n\n"
try: try:
contents = asyncio.run(handle_build_voice_timeline({ contents = asyncio.run(handle_build_voice_timeline({
"media_path": mp, "model": model, "language": language, "media_path": mp, "model": model, "language": language,
"hf_token": token, "num_speakers": num_speakers, "hf_token": token, "num_speakers": num_speakers,
"output_dir": args.get("output_dir"), "output_dir": output_dir,
})) }))
except Exception as exc: except Exception as exc:
_emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"}) _emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"})
return 1 return 1
messages.append("\n".join(getattr(c, "text", str(c)) for c in contents)) messages.append(reused_prefix + "\n".join(getattr(c, "text", str(c)) for c in contents))
_emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)}) _emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)})
return 0 return 0
@@ -938,9 +1057,23 @@ def cmd_set_diarization(args: dict) -> int:
return 0 return 0
def cmd_acoustics_capability(args: dict) -> int:
"""Whether librosa (pitch/energy extraction) is installed in this venv.
Surfaces `features_capability()` — previously computed but never
exposed to the app, so `layers.acoustics: false` in a voice timeline
had no explanation the user could act on.
"""
from fcpxml.voice_features import features_capability
ok, msg = features_capability()
_emit({"ok": True, "available": ok, "message": msg})
return 0
def cmd_voice_analysis(args: dict) -> int: def cmd_voice_analysis(args: dict) -> int:
"""Read the persisted voice-analysis settings (energy/emphasis/emotion).""" """Read the persisted voice-analysis settings (energy/emphasis/emotion)."""
_emit({"ok": True, **load_voice_analysis_config()}) config = load_voice_analysis_config()
_emit({"ok": True, **config, "emphasis_threshold": config["emphasis_floor"]})
return 0 return 0
@@ -950,9 +1083,13 @@ def cmd_set_voice_analysis(args: dict) -> int:
config = save_voice_analysis_config( config = save_voice_analysis_config(
energy_threshold=args.get("energy_threshold"), energy_threshold=args.get("energy_threshold"),
emphasis_weights=weights if isinstance(weights, dict) else None, emphasis_weights=weights if isinstance(weights, dict) else None,
emphasis_threshold=args.get("emphasis_threshold"), emphasis_floor=args.get("emphasis_threshold"),
emotion_enabled=args.get("emotion_enabled"), emotion_enabled=args.get("emotion_enabled"),
emotion_sensitivity=args.get("emotion_sensitivity"), emotion_sensitivity=args.get("emotion_sensitivity"),
zoom_scale=args.get("zoom_scale"),
zoom_mode=args.get("zoom_mode"),
zoom_ease_in=args.get("zoom_ease_in"),
zoom_ease_out=args.get("zoom_ease_out"),
) )
_emit({"ok": True, **config}) _emit({"ok": True, **config})
return 0 return 0
@@ -977,6 +1114,24 @@ def cmd_set_dynamic_subtitle_config(args: dict) -> int:
return 0 return 0
def cmd_plain_subtitle_config(args: dict) -> int:
"""Read the persisted simple subtitle style."""
_emit({"ok": True, **load_plain_subtitle_config()})
return 0
def cmd_set_plain_subtitle_config(args: dict) -> int:
"""Persist simple subtitle style fields. Only the given fields change."""
config = save_plain_subtitle_config(**{
k: args.get(k) for k in (
"font", "font_size", "font_color", "max_words",
"position_y", "uppercase", "keep_punctuation", "text_scale",
)
})
_emit({"ok": True, **config})
return 0
def cmd_silence_config(args: dict) -> int: def cmd_silence_config(args: dict) -> int:
"""Read the persisted silence thresholds (noise floor, duration, padding).""" """Read the persisted silence thresholds (noise floor, duration, padding)."""
_emit({"ok": True, **load_silence_config()}) _emit({"ok": True, **load_silence_config()})
@@ -1026,6 +1181,14 @@ def cmd_apply_voice_actions(args: dict) -> int:
return 1 return 1
actions = loaded.get("actions") if isinstance(loaded, dict) else loaded actions = loaded.get("actions") if isinstance(loaded, dict) else loaded
# The documented output format is {"source": ..., "actions": [...]} —
# callers passing that whole object inline (e.g. the wizard pasting the
# skill's JSON verbatim) need the same unwrap the actions_path branch
# above already does, or a well-formed payload gets rejected as
# "malformed" for having one extra layer of nesting.
if isinstance(actions, dict):
actions = actions.get("actions")
if not isinstance(actions, list) or not actions: if not isinstance(actions, list) or not actions:
_emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."}) _emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."})
return 1 return 1
@@ -1054,6 +1217,86 @@ def cmd_apply_voice_actions(args: dict) -> int:
return 0 return 0
def cmd_build_phrase_review(args: dict) -> int:
"""Build the reviewable script (phrases + the AI's decisions) for the wizard.
`voice_timeline` points at the _voice_timeline.json; `actions` carries the
decision list the model returned (inline, in any of the shapes the skill
emits). The review is always rebuilt from the current analysis, then the
decisions saved on a previous visit are laid back over it — reopening the
step must show the edits the user left there without freezing the acoustics
as they were when they left.
"""
from fcpxml.phrase_review import (
build_phrase_review,
load_phrase_review,
merge_saved_decisions,
)
timeline_path = str(args.get("voice_timeline", ""))
if not timeline_path or not Path(timeline_path).exists():
_emit({"ok": False, "error": "Análise de voz (voice_timeline.json) não encontrada."})
return 1
try:
with open(timeline_path, encoding="utf-8") as fh:
timeline = json.load(fh)
except (OSError, ValueError) as exc:
_emit({"ok": False, "error": f"Erro ao ler a análise de voz: {exc}"})
return 1
extra = [d for d in (args.get("output_dir"), args.get("media_dir")) if d]
review = build_phrase_review(
timeline,
args.get("actions"),
voice_timeline_path=timeline_path,
extra_dirs=extra,
)
saved = None if args.get("fresh") else load_phrase_review(timeline_path)
review = merge_saved_decisions(review, saved)
_emit({"ok": True, "reused": saved is not None, **review})
return 0
def cmd_save_phrase_review(args: dict) -> int:
"""Persist the edited review and the actions derived from it."""
from fcpxml.phrase_review import save_phrase_review
timeline_path = str(args.get("voice_timeline", ""))
if not timeline_path:
_emit({"ok": False, "error": "Caminho da análise de voz não informado."})
return 1
phrases = args.get("phrases")
if not isinstance(phrases, list):
_emit({"ok": False, "error": "Nenhuma frase para salvar."})
return 1
review = {
"version": args.get("version", "1.0"),
"source": args.get("source", ""),
"duration": args.get("duration", 0.0),
"speakers": args.get("speakers", []),
"phrases": phrases,
"zooms": args.get("zooms", []),
}
try:
review_path, actions_path = save_phrase_review(timeline_path, review)
except OSError as exc:
_emit({"ok": False, "error": f"Erro ao salvar a revisão: {exc}"})
return 1
_emit({
"ok": True,
"review_path": str(review_path),
"actions_path": str(actions_path),
"emphasis_count": sum(1 for p in phrases if int(p.get("emphasis", 0) or 0) >= 1),
"removed_count": sum(1 for p in phrases if not p.get("active", True)),
})
return 0
def cmd_project_config(args: dict) -> int: def cmd_project_config(args: dict) -> int:
"""Read the last project folder/file the app was working on.""" """Read the last project folder/file the app was working on."""
_emit({"ok": True, **load_project_config()}) _emit({"ok": True, **load_project_config()})
@@ -1115,17 +1358,23 @@ def main() -> int:
"remove_filler_words": cmd_remove_filler_words, "remove_filler_words": cmd_remove_filler_words,
"transcript_markers": cmd_transcript_markers, "transcript_markers": cmd_transcript_markers,
"generate_dynamic_subtitles": cmd_generate_dynamic_subtitles, "generate_dynamic_subtitles": cmd_generate_dynamic_subtitles,
"generate_plain_subtitles": cmd_generate_plain_subtitles,
"add_zoom": cmd_add_zoom, "add_zoom": cmd_add_zoom,
"zoom_clips": cmd_zoom_clips, "zoom_clips": cmd_zoom_clips,
"zoom_segments": cmd_zoom_segments, "zoom_segments": cmd_zoom_segments,
"rename_speakers": cmd_rename_speakers, "rename_speakers": cmd_rename_speakers,
"set_diarization": cmd_set_diarization, "set_diarization": cmd_set_diarization,
"acoustics_capability": cmd_acoustics_capability,
"voice_analysis": cmd_voice_analysis, "voice_analysis": cmd_voice_analysis,
"set_voice_analysis": cmd_set_voice_analysis, "set_voice_analysis": cmd_set_voice_analysis,
"analyze_voice": cmd_analyze_voice, "analyze_voice": cmd_analyze_voice,
"dynamic_subtitle_config": cmd_dynamic_subtitle_config, "dynamic_subtitle_config": cmd_dynamic_subtitle_config,
"set_dynamic_subtitle_config": cmd_set_dynamic_subtitle_config, "set_dynamic_subtitle_config": cmd_set_dynamic_subtitle_config,
"plain_subtitle_config": cmd_plain_subtitle_config,
"set_plain_subtitle_config": cmd_set_plain_subtitle_config,
"apply_voice_actions": cmd_apply_voice_actions, "apply_voice_actions": cmd_apply_voice_actions,
"build_phrase_review": cmd_build_phrase_review,
"save_phrase_review": cmd_save_phrase_review,
"project_config": cmd_project_config, "project_config": cmd_project_config,
"set_project_config": cmd_set_project_config, "set_project_config": cmd_set_project_config,
"silence_config": cmd_silence_config, "silence_config": cmd_silence_config,
+47
View File
@@ -1183,6 +1183,51 @@ o outro; percentil entrega um punhado útil nos dois casos.
--- ---
## 21 — 2026-08-19 — Teste travado no default antigo de `zoom scale`
- **Sintoma:** `tests/test_voice_actions.py::test_default_scale_when_absent`
quebrando com `KeyError: 'scale'`, sem relação com a alteração em curso.
- **Causa raiz:** `parse_actions` deixou de carimbar `scale=1.3` quando o
parâmetro vem ausente, justamente para que
`server_tools/_shared.py` use o `zoom_scale` configurado pelo usuário. O
teste continuou afirmando o default antigo, então passou a acusar como erro
exatamente o comportamento desejado.
- **Solução adotada:** teste reescrito para o contrato novo — um `scale`
omitido tem que chegar ausente ao aplicador (`test_absent_scale_is_left_absent`).
- **Aprendizado:** quando um default sai do parser e vira configuração, o teste
que afirmava o valor antigo passa a defender o bug. Ao remover um default,
procure o teste que o fixava no mesmo commit — senão ele fica dizendo o
contrário do código, e a próxima pessoa perde tempo achando que quebrou algo.
- **Estado:** `resolvido`
---
## 22 — 2026-08-19 — `VideoPlayer` (AVKit) derruba o app compilado por `swiftc`
- **Sintoma:** "G-ART encerrou inesperadamente" (SIGABRT) toda vez que o
assistente entrava na etapa 5. Nada aparecia na tela antes do crash.
- **Causa raiz:** o app é montado invocando `swiftc` direto
(`MacApp/build_app.sh`), não pelo Xcode. Nesse modo o runtime não consegue
resolver a superclasse Objective-C de `VideoPlayer`:
`failed to demangle superclass of VideoPlayerView from mangled name
'So12AVPlayerViewC'` → `getSuperclassMetadata` chama `fatalError`. É erro de
runtime, então a compilação passa limpa e o problema só aparece ao abrir a
view.
- **Solução adotada:** trocar `VideoPlayer` por um `AVPlayerLayer` dentro de um
`NSViewRepresentable` (`PlayerSurface`/`PlayerLayerView` em
`PhraseReviewView.swift`). Só depende de AVFoundation, que linka normalmente.
Os controles de transporte já viviam na barra da timeline, então não se perde
nada com a chrome do AVKit.
- **Aprendizado:** compilar limpo não prova que um componente de framework
existe em runtime neste build. Ao usar uma view SwiftUI que embrulha uma
classe AppKit/ObjC (AVKit, WebKit, MapKit), abra a tela de fato antes de
concluir. Um harness pequeno (`swiftc` com os mesmos fontes + um `@main` que
monta só aquela view e sai) reproduz o crash em segundos, sem precisar
navegar o app inteiro até lá.
- **Estado:** `resolvido`
---
## Resumo rápido (índice) ## Resumo rápido (índice)
| # | Data | Problema | Estado | | # | Data | Problema | Estado |
@@ -1205,5 +1250,7 @@ o outro; percentil entrega um punhado útil nos dois casos.
| 18 | 2026-08-19 | Legendas dinâmicas geradas com `bold="0" fontFace="Bold"` não renderizam no FCP — negrito deve ser `bold="1"` (atributo) e itálico `fontFace`+`italic="1"` | `resolvido` | | 18 | 2026-08-19 | Legendas dinâmicas geradas com `bold="0" fontFace="Bold"` não renderizam no FCP — negrito deve ser `bold="1"` (atributo) e itálico `fontFace`+`italic="1"` | `resolvido` |
| 19 | 2026-08-19 | `output_dir` usado só como cerca de validação e nunca como destino — toda chamada entre pastas falhava acusando o caminho que ela mesma gerou | `resolvido` | | 19 | 2026-08-19 | `output_dir` usado só como cerca de validação e nunca como destino — toda chamada entre pastas falhava acusando o caminho que ela mesma gerou | `resolvido` |
| 20 | 2026-08-19 | `apply_voice_actions` ausente da ponte e do encadeamento do app — dava para analisar e legendar, não para cortar | `resolvido` | | 20 | 2026-08-19 | `apply_voice_actions` ausente da ponte e do encadeamento do app — dava para analisar e legendar, não para cortar | `resolvido` |
| 21 | 2026-08-19 | Teste ainda afirmava o default `zoom scale=1.3` removido do parser (agora vem do `zoom_scale` do usuário) | `resolvido` |
| 22 | 2026-08-19 | `VideoPlayer` (AVKit) aborta em runtime no app compilado por `swiftc` — etapa 5 fechava o app; trocado por `AVPlayerLayer` | `resolvido` |
> Mantenha o índice acima sempre sincronizado com as entradas mais recentes. > Mantenha o índice acima sempre sincronizado com as entradas mais recentes.
+19 -11
View File
@@ -12,6 +12,7 @@ struct GArtApp: App {
} }
enum ActiveTab: Hashable { enum ActiveTab: Hashable {
case wizard
case project case project
case captions case captions
case voiceAnalysis case voiceAnalysis
@@ -20,19 +21,23 @@ enum ActiveTab: Hashable {
} }
struct ContentView: View { struct ContentView: View {
@State private var activeTab: ActiveTab? = .project @State private var activeTab: ActiveTab? = .wizard
var body: some View { var body: some View {
NavigationSplitView { NavigationSplitView {
List(selection: $activeTab) { List(selection: $activeTab) {
Label("Projeto", systemImage: "film") Label("Assistente", systemImage: "wand.and.stars")
.tag(ActiveTab.project) .tag(ActiveTab.wizard)
Label("Legendas Dinâmicas", systemImage: "captions.bubble") Section("Avançado") {
.tag(ActiveTab.captions) Label("Projeto", systemImage: "film")
Label("Análise de Voz", systemImage: "waveform") .tag(ActiveTab.project)
.tag(ActiveTab.voiceAnalysis) Label("Legendas", systemImage: "captions.bubble")
Label("Modelos", systemImage: "tray.and.arrow.down") .tag(ActiveTab.captions)
.tag(ActiveTab.models) Label("Análise de Voz", systemImage: "waveform")
.tag(ActiveTab.voiceAnalysis)
Label("Modelos", systemImage: "tray.and.arrow.down")
.tag(ActiveTab.models)
}
Label("Sobre", systemImage: "info.circle") Label("Sobre", systemImage: "info.circle")
.tag(ActiveTab.about) .tag(ActiveTab.about)
} }
@@ -40,19 +45,22 @@ struct ContentView: View {
.navigationSplitViewColumnWidth(min: 180, ideal: 200) .navigationSplitViewColumnWidth(min: 180, ideal: 200)
} detail: { } detail: {
switch activeTab { switch activeTab {
case .wizard, nil:
WizardView().id(UUID())
.navigationTitle("Assistente")
case .project: case .project:
ProjectView().id(UUID()) ProjectView().id(UUID())
.navigationTitle("Projeto") .navigationTitle("Projeto")
case .captions: case .captions:
CaptionsView().id(UUID()) CaptionsView().id(UUID())
.navigationTitle("Legendas Dinâmicas") .navigationTitle("Legendas")
case .voiceAnalysis: case .voiceAnalysis:
VoiceAnalysisView().id(UUID()) VoiceAnalysisView().id(UUID())
.navigationTitle("Análise de Voz") .navigationTitle("Análise de Voz")
case .models: case .models:
ModelDownloadView().id(UUID()) ModelDownloadView().id(UUID())
.navigationTitle("Modelos") .navigationTitle("Modelos")
case .about, nil: case .about:
AboutView() AboutView()
.navigationTitle("Sobre") .navigationTitle("Sobre")
} }
+125 -2
View File
@@ -19,6 +19,7 @@ import UniformTypeIdentifiers
/// assunto. /// assunto.
struct CaptionsView: View { struct CaptionsView: View {
@State private var config = CaptionStyleConfig.defaults @State private var config = CaptionStyleConfig.defaults
@State private var plainConfig = PlainSubtitleConfig.defaults
@State private var isLoading = true @State private var isLoading = true
@State private var errorMessage: String? @State private var errorMessage: String?
@@ -53,6 +54,13 @@ struct CaptionsView: View {
) )
} }
private func plainBound<T>(_ keyPath: WritableKeyPath<PlainSubtitleConfig, T>) -> Binding<T> {
Binding(
get: { plainConfig[keyPath: keyPath] },
set: { plainConfig[keyPath: keyPath] = $0; savePlain() }
)
}
private func colorBound(_ keyPath: WritableKeyPath<CaptionStyleConfig, String>) -> Binding<Color> { private func colorBound(_ keyPath: WritableKeyPath<CaptionStyleConfig, String>) -> Binding<Color> {
Binding( Binding(
get: { Color(rgbaString: config[keyPath: keyPath]) }, get: { Color(rgbaString: config[keyPath: keyPath]) },
@@ -60,6 +68,13 @@ struct CaptionsView: View {
) )
} }
private func plainColorBound(_ keyPath: WritableKeyPath<PlainSubtitleConfig, String>) -> Binding<Color> {
Binding(
get: { Color(rgbaString: plainConfig[keyPath: keyPath]) },
set: { plainConfig[keyPath: keyPath] = $0.fcpxmlColorString; savePlain() }
)
}
var body: some View { var body: some View {
HSplitView { HSplitView {
previewColumn previewColumn
@@ -152,6 +167,7 @@ struct CaptionsView: View {
positionSection positionSection
bodySection bodySection
emphasisSection emphasisSection
plainSubtitleSection
calibrationSection calibrationSection
} }
if let errorMessage { if let errorMessage {
@@ -225,6 +241,35 @@ struct CaptionsView: View {
} }
} }
private var plainSubtitleSection: some View {
Section("Legenda comum") {
Picker("Fonte", selection: plainBound(\.font)) {
ForEach(fontChoices, id: \.self) { Text($0).tag($0) }
}
slider(
"Tamanho",
value: plainBound(\.fontSize), in: 28...300, step: 1,
readout: "\(Int(plainConfig.fontSize))pt",
help: "Tamanho da legenda comum editável no Final Cut."
)
slider(
"Máximo de palavras",
value: plainBound(\.maxWords), in: 1...14, step: 1,
readout: "\(Int(plainConfig.maxWords))",
help: "Quantidade máxima de palavras por bloco de legenda."
)
slider(
"Altura",
value: plainBound(\.positionY), in: -1200...300, step: 1,
readout: "\(Int(plainConfig.positionY))",
help: "Posição vertical da legenda comum no quadro; valores mais negativos descem."
)
ColorPicker("Cor", selection: plainColorBound(\.fontColor), supportsOpacity: true)
Toggle("Usar letra maiúscula", isOn: plainBound(\.uppercase))
Toggle("Manter vírgula e ponto", isOn: plainBound(\.keepPunctuation))
}
}
private var calibrationSection: some View { private var calibrationSection: some View {
Section { Section {
slider( slider(
@@ -305,8 +350,17 @@ struct CaptionsView: View {
} else if let error { } else if let error {
errorMessage = error errorMessage = error
} }
isLoading = false PythonBridge.call(command: "plain_subtitle_config") { plainResult, plainError in
continuation.resume() DispatchQueue.main.async {
if let plainResult {
plainConfig = PlainSubtitleConfig(from: plainResult)
} else if let plainError {
errorMessage = plainError
}
isLoading = false
continuation.resume()
}
}
} }
} }
} }
@@ -317,6 +371,12 @@ struct CaptionsView: View {
DispatchQueue.main.async { errorMessage = error } DispatchQueue.main.async { errorMessage = error }
} }
} }
private func savePlain() {
PythonBridge.call(command: "set_plain_subtitle_config", arguments: plainConfig.arguments()) { _, error in
DispatchQueue.main.async { errorMessage = error }
}
}
} }
/// O estilo das legendas dinâmicas, no formato que a tela edita e o bridge /// O estilo das legendas dinâmicas, no formato que a tela edita e o bridge
@@ -402,6 +462,69 @@ struct CaptionStyleConfig {
} }
} }
struct PlainSubtitleConfig {
var font: String
var fontSize: Double
var fontColor: String
var maxWords: Double
var positionY: Double
var uppercase: Bool
var keepPunctuation: Bool
var textScale: Double
static let defaults = PlainSubtitleConfig(
font: "Helvetica Neue",
fontSize: 82,
fontColor: "1 1 1 1",
maxWords: 7,
positionY: -820,
uppercase: false,
keepPunctuation: true,
textScale: 2.0
)
init(from json: [String: Any]) {
let d = PlainSubtitleConfig.defaults
self.init(
font: json["font"] as? String ?? d.font,
fontSize: (json["font_size"] as? NSNumber)?.doubleValue ?? d.fontSize,
fontColor: json["font_color"] as? String ?? d.fontColor,
maxWords: (json["max_words"] as? NSNumber)?.doubleValue ?? d.maxWords,
positionY: (json["position_y"] as? NSNumber)?.doubleValue ?? d.positionY,
uppercase: json["uppercase"] as? Bool ?? d.uppercase,
keepPunctuation: json["keep_punctuation"] as? Bool ?? d.keepPunctuation,
textScale: (json["text_scale"] as? NSNumber)?.doubleValue ?? d.textScale
)
}
init(
font: String, fontSize: Double, fontColor: String, maxWords: Double,
positionY: Double, uppercase: Bool, keepPunctuation: Bool, textScale: Double
) {
self.font = font
self.fontSize = fontSize
self.fontColor = fontColor
self.maxWords = maxWords
self.positionY = positionY
self.uppercase = uppercase
self.keepPunctuation = keepPunctuation
self.textScale = textScale
}
func arguments() -> [String: Any] {
[
"font": font,
"font_size": Int(fontSize),
"font_color": fontColor,
"max_words": Int(maxWords),
"position_y": positionY,
"uppercase": uppercase,
"keep_punctuation": keepPunctuation,
"text_scale": textScale,
]
}
}
extension Color { extension Color {
/// Parses an FCPXML "R G B A" space-separated 0-1 string into a Color. /// Parses an FCPXML "R G B A" space-separated 0-1 string into a Color.
init(rgbaString: String) { init(rgbaString: String) {
+99 -1
View File
@@ -15,6 +15,11 @@ struct ModelDownloadView: View {
@State private var hfTokenText: String = "" @State private var hfTokenText: String = ""
@State private var numSpeakersText: String = "" @State private var numSpeakersText: String = ""
@State private var language: String = "auto" @State private var language: String = "auto"
@State private var acousticsAvailable: Bool?
@State private var acousticsMessage: String = ""
@State private var isInstallingAcoustics = false
@State private var acousticsInstallLog: String = ""
@State private var acousticsInstallError: String?
private let languages: [(String, String)] = [ private let languages: [(String, String)] = [
("auto", "Detectar automaticamente"), ("auto", "Detectar automaticamente"),
@@ -33,6 +38,7 @@ struct ModelDownloadView: View {
var body: some View { var body: some View {
Form { Form {
storageSection storageSection
acousticsSection
diarizationSection diarizationSection
if let errorMessage { if let errorMessage {
Section { Section {
@@ -65,7 +71,7 @@ struct ModelDownloadView: View {
} }
} }
.formStyle(.grouped) .formStyle(.grouped)
.task { await refresh() } .task { await refresh(); checkAcoustics() }
} }
// MARK: - Transcription language // MARK: - Transcription language
@@ -95,6 +101,98 @@ struct ModelDownloadView: View {
PythonBridge.call(command: "set_language", arguments: ["language": code]) { _, _ in } PythonBridge.call(command: "set_language", arguments: ["language": code]) { _, _ in }
} }
// MARK: - Acoustic analysis (librosa)
/// A ênfase de voz (pitch/energia) precisa do `librosa`, que é uma
/// dependência opcional — sem ela `layers.acoustics` vem `false` na
/// análise e a decisão de zoom fica sem base real. Antes disso só dava
/// pra descobrir lendo o JSON exportado; agora o app já diz e resolve.
private var acousticsSection: some View {
Section {
VStack(alignment: .leading, spacing: 10) {
if let acousticsAvailable {
Label(
acousticsMessage.isEmpty
? (acousticsAvailable ? "Disponível" : "Indisponível")
: acousticsMessage,
systemImage: acousticsAvailable ? "checkmark.circle.fill" : "exclamationmark.triangle.fill"
)
.font(.caption)
.foregroundStyle(acousticsAvailable ? Color.green : Color.orange)
} else {
Label("Verificando…", systemImage: "hourglass")
.font(.caption).foregroundStyle(.secondary)
}
if acousticsAvailable == false {
Button {
installAcoustics()
} label: {
if isInstallingAcoustics {
HStack { ProgressView().controlSize(.small); Text("Instalando…") }
} else {
Label("Instalar (uv sync --all-extras)", systemImage: "arrow.down.circle")
}
}
.disabled(isInstallingAcoustics)
if !acousticsInstallLog.isEmpty {
ScrollView {
Text(acousticsInstallLog)
.font(.system(.caption2, design: .monospaced))
.foregroundStyle(.secondary)
.frame(maxWidth: .infinity, alignment: .leading)
}
.frame(height: 90)
.background(RoundedRectangle(cornerRadius: 6).fill(Color.secondary.opacity(0.06)))
}
if let acousticsInstallError {
Label(acousticsInstallError, systemImage: "xmark.circle.fill")
.font(.caption).foregroundStyle(.red)
}
}
}
} header: {
Text("Análise Acústica (zoom por voz)")
} footer: {
Text("Mede a energia e o tom de voz de verdade, para os candidatos a zoom da edição por voz. Sem isso, a análise ainda transcreve e decide cortes pelo texto — só o zoom fica sem base acústica.")
.font(.caption)
.foregroundStyle(.secondary)
}
}
private func checkAcoustics() {
PythonBridge.call(command: "acoustics_capability") { result, err in
DispatchQueue.main.async {
guard let result, result["ok"] as? Bool == true else { return }
acousticsAvailable = result["available"] as? Bool
acousticsMessage = result["message"] as? String ?? ""
}
}
}
private func installAcoustics() {
isInstallingAcoustics = true
acousticsInstallLog = ""
acousticsInstallError = nil
// --all-extras, não só "intelligence": `uv sync` substitui o
// ambiente pelos extras pedidos em vez de somar, então um sync
// parcial aqui derrubaria dev/transcribe/diarização já instalados.
PythonBridge.runUV(arguments: ["sync", "--all-extras"]) { line in
DispatchQueue.main.async {
acousticsInstallLog += (acousticsInstallLog.isEmpty ? "" : "\n") + line
}
} completion: { code, err in
DispatchQueue.main.async {
isInstallingAcoustics = false
if code != 0 {
acousticsInstallError = err ?? "Falha ao instalar."
}
checkAcoustics()
}
}
}
// MARK: - Diarization // MARK: - Diarization
private var diarizationSection: some View { private var diarizationSection: some View {
+139
View File
@@ -116,6 +116,145 @@ struct ZoomClip: Identifiable {
} }
} }
/// One word inside a phrase, with the acoustics that justify an emphasis.
struct ReviewWord: Identifiable {
let id: Int
let text: String
let start: Double
let end: Double
let energy: Double
let emphasis: Double
init(id: Int, json: [String: Any]) {
self.id = id
text = json["text"] as? String ?? ""
start = json["start"] as? Double ?? 0
end = json["end"] as? Double ?? 0
energy = json["energy"] as? Double ?? 0
emphasis = json["emphasis"] as? Double ?? 0
}
}
/// A phrase in the review step — one spoken line plus the decision made about
/// it. Mirrors `fcpxml/phrase_review.py`; `emphasis` is 0–3 and everything
/// mutable here is what the editor is allowed to change.
struct ReviewPhrase: Identifiable {
let id: Int
let start: Double
let end: Double
var trimStart: Double
var trimEnd: Double
var text: String
let speaker: String
var active: Bool
var emphasis: Int
var track: String
let peakEmphasis: Double
let emotion: String
let emotionConfidence: Double
let takeBoundary: Bool
let gapBefore: Double
let reason: String
let words: [ReviewWord]
static let trackScript = "roteiro"
static let trackBackstage = "bastidor"
/// Delivery emotion as the analysis names it, in the user's language plus a
/// glyph — the label alone is too easy to skim past in a dense list.
static func emotionLabel(_ emotion: String) -> (String, String) {
switch emotion {
case "excited": return ("Empolgado", "flame")
case "tense": return ("Tenso", "bolt")
case "calm": return ("Calmo", "leaf")
case "reflective": return ("Reflexivo", "moon")
default: return ("Neutro", "circle")
}
}
init(json: [String: Any]) {
id = json["index"] as? Int ?? 0
start = json["start"] as? Double ?? 0
end = json["end"] as? Double ?? 0
trimStart = json["trim_start"] as? Double ?? (json["start"] as? Double ?? 0)
trimEnd = json["trim_end"] as? Double ?? (json["end"] as? Double ?? 0)
text = json["text"] as? String ?? ""
speaker = json["speaker"] as? String ?? ""
active = json["active"] as? Bool ?? true
emphasis = json["emphasis"] as? Int ?? 0
track = json["track"] as? String ?? ReviewPhrase.trackScript
peakEmphasis = json["peak_emphasis"] as? Double ?? 0
emotion = json["emotion"] as? String ?? "neutral"
emotionConfidence = json["emotion_confidence"] as? Double ?? 0
takeBoundary = json["take_boundary"] as? Bool ?? false
gapBefore = json["gap_before"] as? Double ?? 0
reason = json["reason"] as? String ?? ""
words = (json["words"] as? [[String: Any]] ?? [])
.enumerated().map { ReviewWord(id: $0.offset, json: $0.element) }
}
var asJSON: [String: Any] {
[
"index": id,
"start": start,
"end": end,
"trim_start": trimStart,
"trim_end": trimEnd,
"text": text,
"speaker": speaker,
"active": active,
"emphasis": emphasis,
"track": track,
"reason": reason,
]
}
var isBackstage: Bool { track == ReviewPhrase.trackBackstage }
var isTrimmed: Bool { trimStart > start + 0.001 || trimEnd < end - 0.001 }
var timecode: String {
String(format: "%02d:%02d", Int(start) / 60, Int(start) % 60)
}
/// The word boundaries a trim handle is allowed to land on.
func snap(_ time: Double, edge: TrimEdge) -> Double {
let boundaries = words.map { edge == .start ? $0.start : $0.end }.filter { $0 > 0 }
guard let nearest = boundaries.min(by: { abs($0 - time) < abs($1 - time) }) else {
return time
}
return nearest
}
}
enum TrimEdge { case start, end }
/// A punch-in the editor placed by hand over an arbitrary range, next to the
/// whole-phrase zoom that an emphasis level produces. It stores only *when* —
/// the scale and the ramp come from the Voice Analysis settings at render time.
struct ManualZoom: Identifiable {
let id = UUID()
var start: Double
var end: Double
/// Below this a punch-in has no room to ramp in and back out; the writer
/// rejects the window, so offering it would place nothing.
static let minimumDuration: Double = 0.4
init(start: Double, end: Double) {
self.start = start
self.end = end
}
init?(json: [String: Any]) {
guard let start = json["start"] as? Double, let end = json["end"] as? Double,
end - start >= ManualZoom.minimumDuration
else { return nil }
self.start = start
self.end = end
}
var asJSON: [String: Any] { ["start": start, "end": end] }
}
struct ZoomSegment: Identifiable { struct ZoomSegment: Identifiable {
let id: Int let id: Int
let start: Double let start: Double
+429
View File
@@ -0,0 +1,429 @@
import AVFoundation
import Combine
import Foundation
/// State behind the wizard's emphasis-review step.
///
/// Holds the phrases, the selection, and the player — together, because they
/// are one thing to the user: clicking a phrase moves the playhead, playing
/// moves the selection, and skipping a removed line only works if whoever owns
/// playback also knows which lines are removed.
///
/// The preview deliberately plays the *original* media and jumps over whatever
/// the edit removes, instead of rendering a cut first. Rendering to check a
/// toggle would put minutes between a decision and its result; jumping gives
/// the same reading instantly, and the real cut is generated later from the
/// exact same phrase list.
@MainActor
final class PhraseReviewModel: ObservableObject {
@Published var phrases: [ReviewPhrase] = []
@Published var selection: Int?
@Published var isLoading = false
@Published var errorMessage: String?
@Published var currentTime: Double = 0
@Published var isPlaying = false
@Published var pixelsPerSecond: Double = 40
@Published var skipRemoved = true
@Published var zooms: [ManualZoom] = []
/// In/out the editor dragged on the timeline, in source seconds.
@Published var rangeStart: Double?
@Published var rangeEnd: Double?
/// Aspect ratio of the footage as recorded.
@Published var videoAspect: Double = 16.0 / 9.0
/// Aspect ratio the project delivers in, read from the .fcpxml. It is
/// routinely *not* the footage's: these takes are shot horizontal and
/// delivered vertical, so previewing the raw frame would show a crop the
/// audience never sees — and the emphasis decisions are about what lands on
/// screen. Nil until the project is known.
@Published var projectAspect: Double?
/// Whether the preview crops to the delivery frame. On by default whenever
/// the two aspects disagree.
@Published var matchProjectFraming = true
/// What the preview should actually draw.
var previewAspect: Double {
guard matchProjectFraming, let projectAspect else { return videoAspect }
return projectAspect
}
/// True when the delivery frame differs enough from the footage that the
/// preview is showing a crop rather than the whole take.
var isCropping: Bool {
guard matchProjectFraming, let projectAspect else { return false }
return abs(projectAspect - videoAspect) > 0.01
}
private(set) var source = ""
private(set) var sourcePath = ""
private(set) var duration: Double = 0
private(set) var speakers: [String] = []
private(set) var emotionAvailable = false
private(set) var player: AVPlayer?
private var voiceTimelinePath = ""
private var timeObserver: Any?
private var playbackLimit: Double?
let minPixelsPerSecond: Double = 8
let maxPixelsPerSecond: Double = 400
deinit {
if let timeObserver, let player {
player.removeTimeObserver(timeObserver)
}
}
// MARK: - Carregar
/// Builds the review from the voice timeline plus whatever the AI decided.
/// A review saved on a previous visit wins — see `cmd_build_phrase_review`.
/// Reads the delivery format from the project so the preview can frame the
/// take the way it will actually be seen.
func loadProjectFormat(projectPath: String) {
PythonBridge.call(command: "inspect", arguments: ["path": projectPath]) { [weak self] result, _ in
Task { @MainActor in
guard let self,
let timelines = result?["timelines"] as? [[String: Any]],
let first = timelines.first,
let width = first["width"] as? Int, let height = first["height"] as? Int,
width > 0, height > 0
else { return }
self.projectAspect = Double(width) / Double(height)
}
}
}
func load(voiceTimelinePath: String, decisionsJSON: String,
outputFolder: String? = nil, mediaFolder: String? = nil) {
self.voiceTimelinePath = voiceTimelinePath
isLoading = true
errorMessage = nil
var arguments: [String: Any] = ["voice_timeline": voiceTimelinePath]
if let outputFolder { arguments["output_dir"] = outputFolder }
if let mediaFolder { arguments["media_dir"] = mediaFolder }
if let data = decisionsJSON.data(using: .utf8),
let parsed = try? JSONSerialization.jsonObject(with: data) {
arguments["actions"] = parsed
}
PythonBridge.call(command: "build_phrase_review", arguments: arguments) { [weak self] result, error in
Task { @MainActor in
guard let self else { return }
self.isLoading = false
if let error {
self.errorMessage = error
return
}
guard let result, result["ok"] as? Bool == true else {
self.errorMessage = result?["error"] as? String ?? "Não foi possível montar a revisão."
return
}
self.apply(result)
}
}
}
private func apply(_ result: [String: Any]) {
source = result["source"] as? String ?? ""
// The timeline JSON stores only the media's file name; the bridge
// resolves it to something openable (see phrase_review.resolve_source).
sourcePath = result["source_path"] as? String ?? ""
duration = result["duration"] as? Double ?? 0
speakers = result["speakers"] as? [String] ?? []
emotionAvailable = result["emotion_available"] as? Bool ?? false
phrases = (result["phrases"] as? [[String: Any]] ?? []).map { ReviewPhrase(json: $0) }
zooms = (result["zooms"] as? [[String: Any]] ?? []).compactMap { ManualZoom(json: $0) }
selection = phrases.first?.id
if let errors = result["errors"] as? [String], !errors.isEmpty {
errorMessage = "A IA mandou \(errors.count) decisão(ões) que não deu para ler — o resto foi aplicado."
}
preparePlayer()
}
/// Point the preview at a media file the user chose by hand — the way out
/// when the footage moved somewhere the automatic lookup can't reach.
func useMedia(at path: String) {
sourcePath = path
preparePlayer()
}
private func preparePlayer() {
guard !sourcePath.isEmpty, FileManager.default.fileExists(atPath: sourcePath) else {
player = nil
return
}
if let timeObserver, let player {
player.removeTimeObserver(timeObserver)
self.timeObserver = nil
}
let asset = AVURLAsset(url: URL(fileURLWithPath: sourcePath))
let player = AVPlayer(playerItem: AVPlayerItem(asset: asset))
self.player = player
readAspect(from: asset)
// 60 Hz: the same observer drives the playhead *and* decides when to
// jump a removed stretch, so its period is the worst-case amount of cut
// material that can be heard before the skip lands. At 20 Hz that was an
// audible blip on every join.
let interval = CMTime(seconds: 1.0 / 60.0, preferredTimescale: 600)
timeObserver = player.addPeriodicTimeObserver(forInterval: interval, queue: .main) { [weak self] time in
Task { @MainActor in
self?.tick(time.seconds)
}
}
}
/// The displayed aspect ratio, honouring the rotation the camera recorded.
/// A phone take is stored 1920×1080 with a 90° transform: reading
/// `naturalSize` alone would call a vertical video horizontal.
private func readAspect(from asset: AVURLAsset) {
Task { [weak self] in
guard let track = try? await asset.loadTracks(withMediaType: .video).first,
let size = try? await track.load(.naturalSize),
let transform = try? await track.load(.preferredTransform)
else { return }
let displayed = size.applying(transform)
let width = abs(displayed.width), height = abs(displayed.height)
guard width > 0, height > 0 else { return }
await MainActor.run { self?.videoAspect = width / height }
}
}
// MARK: - Reprodução
private func tick(_ time: Double) {
currentTime = time
guard isPlaying else { return }
// Playing a single phrase or a marked range stops at its out point
// instead of running on into the rest of the take.
if let limit = playbackLimit, time >= limit {
pause()
seek(to: limit)
return
}
if skipRemoved, let jump = nextKeptTime(after: time), jump > time {
seek(to: jump)
}
if let phrase = phrase(at: time), selection != phrase.id {
selection = phrase.id
}
}
/// Where playback should resume when `time` lands on removed material.
/// Returns nil when the time is on material that survives.
func nextKeptTime(after time: Double) -> Double? {
for phrase in phrases where time >= phrase.start - 0.001 && time < phrase.end {
if !phrase.active { return phrase.end }
if time < phrase.trimStart { return phrase.trimStart }
if time >= phrase.trimEnd { return phrase.end }
return nil
}
return nil
}
func togglePlay() {
if isPlaying {
pause()
} else {
playbackLimit = nil
play()
}
}
private func play() {
guard let player else { return }
if skipRemoved, let jump = nextKeptTime(after: currentTime) { seek(to: jump) }
player.play()
isPlaying = true
}
func pause() {
player?.pause()
isPlaying = false
playbackLimit = nil
}
/// Play exactly one span and stop — how a cut is judged: in context, at
/// speed, without hunting for the out point by hand.
func playRange(from start: Double, to end: Double) {
guard end > start else { return }
seek(to: start)
playbackLimit = end
player?.play()
isPlaying = true
}
func playSelectedPhrase() {
guard let selection, let phrase = phrases.first(where: { $0.id == selection })
else { return }
playRange(from: phrase.active ? phrase.trimStart : phrase.start,
to: phrase.active ? phrase.trimEnd : phrase.end)
}
func seek(to time: Double) {
currentTime = max(0, time)
player?.seek(to: CMTime(seconds: max(0, time), preferredTimescale: 600),
toleranceBefore: .zero, toleranceAfter: .zero)
}
/// Move the playhead to a phrase and select it.
func goTo(phraseID: Int) {
guard let phrase = phrases.first(where: { $0.id == phraseID }) else { return }
selection = phraseID
seek(to: phrase.active ? phrase.trimStart : phrase.start)
}
func phrase(at time: Double) -> ReviewPhrase? {
phrases.first { time >= $0.start && time < $0.end }
}
func selectNeighbour(_ delta: Int) {
guard let selection, let index = phrases.firstIndex(where: { $0.id == selection }) else {
if let first = phrases.first { goTo(phraseID: first.id) }
return
}
let next = min(max(0, index + delta), phrases.count - 1)
goTo(phraseID: phrases[next].id)
}
// MARK: - Edições
private func update(_ id: Int, _ change: (inout ReviewPhrase) -> Void) {
guard let index = phrases.firstIndex(where: { $0.id == id }) else { return }
change(&phrases[index])
}
func setEmphasis(_ level: Int, for id: Int) {
update(id) { $0.emphasis = min(3, max(0, level)) }
}
func toggleActive(_ id: Int) {
update(id) { $0.active.toggle() }
}
func setTrack(_ track: String, for id: Int) {
update(id) { $0.track = track }
}
func setText(_ text: String, for id: Int) {
update(id) { $0.text = text }
}
/// Trim a phrase's head or tail, landing on a word boundary.
/// A trim that would swallow the whole line is refused — deactivating the
/// phrase is the way to remove it, and doing it by accident with a drag
/// would lose the emphasis decision along with the line.
func trim(_ id: Int, edge: TrimEdge, to time: Double) {
update(id) { phrase in
let snapped = phrase.snap(time, edge: edge)
switch edge {
case .start:
let value = min(max(phrase.start, snapped), phrase.trimEnd - 0.1)
if value < phrase.trimEnd { phrase.trimStart = value }
case .end:
let value = max(min(phrase.end, snapped), phrase.trimStart + 0.1)
if value > phrase.trimStart { phrase.trimEnd = value }
}
}
}
func resetTrim(_ id: Int) {
update(id) { $0.trimStart = $0.start; $0.trimEnd = $0.end }
}
/// Trim everything before/after a given word — the text-first way to cut,
/// since the editor reads the line and points at where it should begin.
func trimToWord(_ word: ReviewWord, edge: TrimEdge, in id: Int) {
trim(id, edge: edge, to: edge == .start ? word.start : word.end)
}
// MARK: - Trecho marcado e zooms
var hasRange: Bool {
guard let rangeStart, let rangeEnd else { return false }
return rangeEnd - rangeStart >= ManualZoom.minimumDuration
}
var rangeSpan: (start: Double, end: Double)? {
guard let rangeStart, let rangeEnd, rangeEnd > rangeStart else { return nil }
return (rangeStart, rangeEnd)
}
func setRange(from start: Double, to end: Double) {
rangeStart = min(start, end)
rangeEnd = max(start, end)
}
func clearRange() {
rangeStart = nil
rangeEnd = nil
}
/// Add a punch-in over the marked range. Scale and ramp are not stored:
/// they come from the "Análise de Voz" settings when the edit is rendered,
/// so changing the look there restyles every zoom at once.
func addZoomForRange() {
guard let span = rangeSpan, span.end - span.start >= ManualZoom.minimumDuration
else { return }
zooms.append(ManualZoom(start: span.start, end: span.end))
zooms.sort { $0.start < $1.start }
clearRange()
}
func addZoomForPhrase(_ id: Int) {
guard let phrase = phrases.first(where: { $0.id == id }) else { return }
zooms.append(ManualZoom(start: phrase.trimStart, end: phrase.trimEnd))
zooms.sort { $0.start < $1.start }
}
func removeZoom(_ id: UUID) {
zooms.removeAll { $0.id == id }
}
func zoom(at time: Double) -> ManualZoom? {
zooms.first { time >= $0.start && time <= $0.end }
}
func setEmphasisForAll(_ level: Int) {
for index in phrases.indices where phrases[index].active {
phrases[index].emphasis = level
}
}
// MARK: - Resumo e gravação
var emphasisCount: Int { phrases.filter { $0.active && $0.emphasis >= 1 }.count }
var removedCount: Int { phrases.filter { !$0.active }.count }
var keptDuration: Double {
phrases.filter { $0.active }.reduce(0) { $0 + ($1.trimEnd - $1.trimStart) }
}
/// Persists the edited review plus the actions derived from it. Called when
/// the wizard advances — the render itself happens in the next step.
func save(completion: @escaping (String?) -> Void) {
guard !voiceTimelinePath.isEmpty, !phrases.isEmpty else {
completion(nil)
return
}
let arguments: [String: Any] = [
"voice_timeline": voiceTimelinePath,
"source": source,
"duration": duration,
"speakers": speakers,
"phrases": phrases.map { $0.asJSON },
"zooms": zooms.map { $0.asJSON },
]
PythonBridge.call(command: "save_phrase_review", arguments: arguments) { result, error in
Task { @MainActor in
if let error {
completion(nil)
_ = error
return
}
completion(result?["review_path"] as? String)
}
}
}
}
+413
View File
@@ -0,0 +1,413 @@
import AVFoundation
import SwiftUI
/// The video surface, as a plain `AVPlayerLayer` in an `NSView`.
///
/// AVKit's `VideoPlayer` would be the obvious choice and is a trap here: this
/// app is built by invoking `swiftc` directly (see `MacApp/build_app.sh`), and
/// `_AVKit_SwiftUI` aborts at launch instantiating its generic metadata under
/// that build. A player layer needs only AVFoundation, which links cleanly —
/// and the transport controls live in the timeline's own toolbar anyway, so
/// nothing is lost by dropping AVKit's chrome.
private struct PlayerSurface: NSViewRepresentable {
let player: AVPlayer
/// When true the frame is filled and cropped instead of letterboxed — used
/// to preview horizontal footage inside a vertical delivery frame.
var fills: Bool
func makeNSView(context: Context) -> PlayerLayerView {
let view = PlayerLayerView()
view.player = player
view.fills = fills
return view
}
func updateNSView(_ view: PlayerLayerView, context: Context) {
if view.player !== player { view.player = player }
view.fills = fills
}
}
final class PlayerLayerView: NSView {
private let playerLayer = AVPlayerLayer()
var player: AVPlayer? {
get { playerLayer.player }
set { playerLayer.player = newValue }
}
var fills: Bool = false {
didSet { playerLayer.videoGravity = fills ? .resizeAspectFill : .resizeAspect }
}
override init(frame frameRect: NSRect) {
super.init(frame: frameRect)
wantsLayer = true
layer = CALayer()
layer?.backgroundColor = NSColor.black.cgColor
playerLayer.videoGravity = .resizeAspect
layer?.addSublayer(playerLayer)
}
required init?(coder: NSCoder) {
super.init(coder: coder)
wantsLayer = true
layer = CALayer()
playerLayer.videoGravity = .resizeAspect
layer?.addSublayer(playerLayer)
}
override func layout() {
super.layout()
playerLayer.frame = bounds
}
}
/// The wizard's emphasis-review step, laid out like an editing room: preview on
/// top, timeline across the bottom, and the script as an inspector down the
/// right side.
///
/// The arrangement is the point. Every decision here is about a *sentence*, so
/// the same phrase has to be legible in all three places at once — a block on
/// the timeline, a line of text in the inspector, and a moment in the preview.
/// Selecting in any one of them selects in the other two.
struct PhraseReviewView: View {
@ObservedObject var model: PhraseReviewModel
var body: some View {
VSplitView {
HSplitView {
previewPane
.frame(minWidth: 320, idealWidth: 640)
inspectorPane
.frame(minWidth: 300, idealWidth: 360, maxWidth: 520)
}
.frame(minHeight: 240)
TimelineTracksView(model: model)
.frame(minHeight: 190, idealHeight: 210)
}
.overlay { if model.isLoading { loadingOverlay } }
.focusable()
.onKeyPress(.space) { model.togglePlay(); return .handled }
.onKeyPress(.return) { model.playSelectedPhrase(); return .handled }
.onKeyPress(.leftArrow) { model.selectNeighbour(-1); return .handled }
.onKeyPress(.rightArrow) { model.selectNeighbour(1); return .handled }
.onKeyPress(characters: .decimalDigits) { press in
guard let level = Int(press.characters), (0...3).contains(level),
let selection = model.selection else { return .ignored }
model.setEmphasis(level, for: selection)
return .handled
}
}
private var loadingOverlay: some View {
ZStack {
Color(nsColor: .windowBackgroundColor).opacity(0.85)
VStack(spacing: 10) {
ProgressView()
Text("Montando a revisão…").font(.callout).foregroundStyle(.secondary)
}
}
}
// MARK: - Preview
private var previewPane: some View {
VStack(spacing: 0) {
if let player = model.player {
// The footage here is usually vertical. Sizing the surface to
// the take's own aspect keeps a 9:16 frame as tall as the pane
// allows instead of shrinking it to fit a horizontal box.
// Framed to what the project delivers, not to what the camera
// recorded: these takes are shot horizontal and cut vertical,
// so the raw frame would show material the audience never sees.
ZStack {
Color.black
PlayerSurface(player: player, fills: model.isCropping)
.aspectRatio(model.previewAspect, contentMode: .fit)
.clipped()
}
.overlay(alignment: .topTrailing) { framingBadge }
} else {
ZStack {
Color.black.opacity(0.85)
VStack(spacing: 10) {
Image(systemName: "film.stack")
.font(.system(size: 28)).foregroundStyle(.secondary)
Text(model.source.isEmpty
? "A análise de voz não registrou qual mídia foi usada."
: "Não achei \(model.source) na pasta do projeto.")
.font(.callout).foregroundStyle(.secondary)
Text("A revisão funciona igual sem o preview — ele só ajuda a conferir o corte.")
.font(.caption).foregroundStyle(.tertiary)
Button("Localizar a mídia…") { pickMedia() }
.buttonStyle(.bordered)
}
.multilineTextAlignment(.center)
.padding(.horizontal, 24)
}
}
Divider()
summaryBar
}
}
private var summaryBar: some View {
HStack(spacing: 16) {
summaryItem("text.quote", "\(model.phrases.count) frases")
summaryItem("sparkles", "\(model.emphasisCount) com ênfase")
summaryItem("scissors", "\(model.removedCount) fora do corte")
summaryItem("clock", durationLabel(model.keptDuration))
if !model.zooms.isEmpty {
summaryItem("plus.magnifyingglass", "\(model.zooms.count) zooms")
}
Spacer()
if let phrase = selectedPhrase, !phrase.reason.isEmpty {
Label(phrase.reason, systemImage: "brain")
.font(.caption).foregroundStyle(.secondary)
.lineLimit(1).truncationMode(.tail)
}
}
.padding(.horizontal, 14)
.padding(.vertical, 8)
}
private func summaryItem(_ icon: String, _ text: String) -> some View {
Label(text, systemImage: icon).font(.caption).foregroundStyle(.secondary)
}
private func durationLabel(_ seconds: Double) -> String {
String(format: "%02d:%02d finais", Int(seconds) / 60, Int(seconds) % 60)
}
/// Says which frame is on screen, and lets the editor flip to the raw take.
/// Without it a centred crop looks like the footage itself, and someone
/// would judge framing on an approximation without knowing it.
@ViewBuilder
private var framingBadge: some View {
if model.projectAspect != nil, abs((model.projectAspect ?? 0) - model.videoAspect) > 0.01 {
Button {
model.matchProjectFraming.toggle()
} label: {
Label(model.matchProjectFraming ? "Enquadramento do projeto" : "Mídia original",
systemImage: model.matchProjectFraming ? "crop" : "rectangle.expand.vertical")
.font(.caption2)
}
.buttonStyle(.borderless)
.padding(6)
.background(Capsule().fill(.black.opacity(0.45)))
.foregroundStyle(.white)
.padding(8)
.help("A fonte é horizontal e o projeto é vertical — o preview mostra o corte central aproximado. O enquadramento real de cada clipe vem do Final Cut.")
}
}
private func pickMedia() {
let panel = NSOpenPanel()
panel.canChooseFiles = true
panel.canChooseDirectories = false
panel.allowsMultipleSelection = false
panel.prompt = "Usar esta mídia"
panel.message = model.source.isEmpty
? "Escolha o arquivo de vídeo desta gravação."
: "Escolha onde está \(model.source)."
if panel.runModal() == .OK, let url = panel.url {
model.useMedia(at: url.path)
}
}
private var selectedPhrase: ReviewPhrase? {
guard let selection = model.selection else { return nil }
return model.phrases.first { $0.id == selection }
}
// MARK: - Inspector de frases
private var inspectorPane: some View {
VStack(spacing: 0) {
inspectorHeader
Divider()
List(selection: $model.selection) {
ForEach($model.phrases) { $phrase in
PhraseRow(phrase: $phrase, model: model)
.tag(phrase.id)
}
}
.listStyle(.inset)
.onChange(of: model.selection) { _, newValue in
if let newValue { model.goTo(phraseID: newValue) }
}
}
}
private var inspectorHeader: some View {
VStack(alignment: .leading, spacing: 6) {
Text("Frases").font(.headline)
Text("Só as frases com ênfase recebem zoom e legenda dinâmica. O resto fica com legenda comum.")
.font(.caption).foregroundStyle(.secondary)
if !model.emotionAvailable {
Label("Emoção da fala não foi detectada nesta análise — ligue em Avançado → Análise de Voz e refaça o passo 3.",
systemImage: "waveform.path.ecg")
.font(.caption2).foregroundStyle(.secondary)
}
HStack(spacing: 8) {
Button("Limpar ênfases") { model.setEmphasisForAll(0) }
.buttonStyle(.link).font(.caption)
Spacer()
Text("0–3 no teclado · ← → navega")
.font(.caption2).foregroundStyle(.secondary)
}
}
.padding(12)
}
}
/// One phrase in the inspector: the line as it will be said, plus every
/// decision attached to it. Kept in one row on purpose — jumping to a separate
/// detail pane to set a toggle would double the clicks on the most repeated
/// action in the screen.
private struct PhraseRow: View {
@Binding var phrase: ReviewPhrase
@ObservedObject var model: PhraseReviewModel
@State private var isEditing = false
var body: some View {
VStack(alignment: .leading, spacing: 6) {
HStack(spacing: 6) {
Text(phrase.timecode)
.font(.system(.caption2, design: .monospaced))
.foregroundStyle(.secondary)
if phrase.takeBoundary {
Image(systemName: "scissors.badge.ellipsis")
.font(.caption2).foregroundStyle(.orange)
.help("Nova tomada começa aqui")
}
if phrase.isTrimmed {
Image(systemName: "arrow.left.and.right.square")
.font(.caption2).foregroundStyle(.blue)
.help("Frase cortada nas pontas")
}
if model.emotionAvailable {
emotionChip
}
Spacer()
Toggle("", isOn: $phrase.active)
.toggleStyle(.switch)
.controlSize(.mini)
.labelsHidden()
.help(phrase.active ? "No corte" : "Fora do corte")
}
if isEditing {
TextField("Texto da frase", text: $phrase.text, axis: .vertical)
.textFieldStyle(.roundedBorder)
.font(.callout)
.onSubmit { isEditing = false }
} else {
Text(phrase.text.isEmpty ? "(sem texto)" : phrase.text)
.font(.callout)
.foregroundStyle(phrase.active ? .primary : .secondary)
.strikethrough(!phrase.active)
.onTapGesture(count: 2) { isEditing = true }
}
HStack(spacing: 8) {
Picker("", selection: $phrase.emphasis) {
ForEach(0..<4, id: \.self) { level in
Text(EmphasisPalette.label(level)).tag(level)
}
}
.pickerStyle(.segmented)
.controlSize(.mini)
.labelsHidden()
.disabled(!phrase.active)
Picker("", selection: $phrase.track) {
Text("Roteiro").tag(ReviewPhrase.trackScript)
Text("Bastidor").tag(ReviewPhrase.trackBackstage)
}
.pickerStyle(.menu)
.controlSize(.mini)
.labelsHidden()
.frame(width: 92)
}
if model.selection == phrase.id && !phrase.words.isEmpty {
wordTrimmer
}
}
.padding(.vertical, 4)
.opacity(phrase.active ? 1 : 0.55)
}
/// The delivery emotion the acoustics suggest. Shown faded below its own
/// confidence: a guess the analysis is unsure about should not compete for
/// attention with the emphasis decision, which is the point of the row.
private var emotionChip: some View {
let (label, icon) = ReviewPhrase.emotionLabel(phrase.emotion)
return Label(label, systemImage: icon)
.font(.caption2)
.padding(.horizontal, 5)
.padding(.vertical, 1)
.background(
Capsule().fill(Color.secondary.opacity(0.12))
)
.foregroundStyle(phrase.emotionConfidence >= 0.5 ? .secondary : .tertiary)
.help("Emoção da entrega: \(label) — confiança \(Int(phrase.emotionConfidence * 100))%")
}
/// Trimming by pointing at the transcript: click a word to start the phrase
/// there, option-click to end it there. Same edit as dragging the block's
/// edge on the timeline, but reachable while reading the line.
private var wordTrimmer: some View {
VStack(alignment: .leading, spacing: 4) {
HStack(spacing: 4) {
Text("Cortar pelas palavras").font(.caption2).foregroundStyle(.secondary)
Spacer()
if phrase.isTrimmed {
Button("Inteira") { model.resetTrim(phrase.id) }
.buttonStyle(.link).font(.caption2)
}
}
FlowWords(words: phrase.words, phrase: phrase) { word, edge in
model.trimToWord(word, edge: edge, in: phrase.id)
}
Text("Clique = começa aqui · ⌥clique = termina aqui")
.font(.caption2).foregroundStyle(.tertiary)
}
.padding(.top, 2)
}
}
/// The phrase's words as wrapping chips, dimmed where they fall outside the trim.
private struct FlowWords: View {
let words: [ReviewWord]
let phrase: ReviewPhrase
let onTrim: (ReviewWord, TrimEdge) -> Void
var body: some View {
// A LazyVGrid with adaptive columns wraps chips without a custom layout;
// phrases are short enough that the slight raggedness beats the cost of
// hand-rolling a flow layout here.
LazyVGrid(columns: [GridItem(.adaptive(minimum: 44), spacing: 3)],
alignment: .leading, spacing: 3) {
ForEach(words) { word in
let kept = word.start >= phrase.trimStart - 0.001 && word.end <= phrase.trimEnd + 0.001
Text(word.text)
.font(.caption2)
.padding(.horizontal, 4)
.padding(.vertical, 2)
.background(
RoundedRectangle(cornerRadius: 3)
.fill(kept ? Color.accentColor.opacity(0.12) : Color.secondary.opacity(0.08))
)
.foregroundStyle(kept ? .primary : .secondary)
.strikethrough(!kept)
.onTapGesture {
onTrim(word, NSEvent.modifierFlags.contains(.option) ? .end : .start)
}
}
}
}
}
+69 -2
View File
@@ -44,12 +44,26 @@ enum PythonBridge {
return ["python3", scriptURL.path] return ["python3", scriptURL.path]
} }
/// `admin/models_api.py` lives outside `code/`, but its dependencies
/// (`pyproject.toml`, `.venv`) live inside it. `uv run` picks the
/// environment from the process's cwd, not from the script path — so
/// running with cwd at the repo root made `uv` create/use a second,
/// empty `.venv` there, silently ignoring everything installed into
/// `code/.venv` (this cost a real debugging session: librosa/pyannote
/// installed successfully but the app kept reporting them missing).
/// Every `uv run` must share the same cwd as `uv sync` to see the same
/// environment.
static var workingDirectory: URL { static var workingDirectory: URL {
projectRoot codeDirectory
}
/// Directory containing `pyproject.toml` — where `uv sync` must run from.
static var codeDirectory: URL {
projectRoot.appendingPathComponent("code")
} }
/// Locate `uv` on PATH or in common install locations. /// Locate `uv` on PATH or in common install locations.
private static func findUV() -> String? { static func findUV() -> String? {
if let onPath = which("uv") { return onPath } if let onPath = which("uv") { return onPath }
let candidates = [ let candidates = [
"/usr/local/bin/uv", "/usr/local/bin/uv",
@@ -148,6 +162,59 @@ enum PythonBridge {
} }
} }
// MARK: - uv sync (installing optional extras, e.g. acoustic analysis)
/// Runs `uv <arguments>` from `codeDirectory` (where `pyproject.toml`
/// lives), streaming each output line as plain text — used for
/// `sync --extra intelligence` so "Modelos" can install the librosa
/// extra without the user opening a terminal.
static func runUV(arguments: [String],
onLine: @escaping (String) -> Void,
completion: @escaping (Int, String?) -> Void) {
guard let uv = findUV() else {
completion(1, "uv não encontrado. Instale com: curl -LsSf https://astral.sh/uv/install.sh | sh")
return
}
let process = Process()
process.executableURL = URL(fileURLWithPath: "/usr/bin/env")
process.arguments = [uv] + arguments
process.currentDirectoryURL = codeDirectory
let pipe = Pipe()
process.standardOutput = pipe
process.standardError = pipe
var buffer = ""
let lock = NSLock()
pipe.fileHandleForReading.readabilityHandler = { handle in
let data = handle.availableData
guard !data.isEmpty, let s = String(data: data, encoding: .utf8) else { return }
lock.lock()
buffer += s
let parts = buffer.split(separator: "\n", omittingEmptySubsequences: false)
buffer = String(parts.last ?? "")
let lines = parts.dropLast()
lock.unlock()
for line in lines where !line.isEmpty { onLine(String(line)) }
}
process.terminationHandler = { p in
pipe.fileHandleForReading.readabilityHandler = nil
lock.lock()
let last = buffer.trimmingCharacters(in: .whitespacesAndNewlines)
buffer = ""
lock.unlock()
if !last.isEmpty { onLine(last) }
completion(Int(p.terminationStatus), p.terminationStatus == 0 ? nil : "uv sync terminou com erro (código \(p.terminationStatus)).")
}
do {
try process.run()
} catch {
completion(1, error.localizedDescription)
}
}
// MARK: - Convenience: single JSON result // MARK: - Convenience: single JSON result
/// Runs a command and delivers the first parsed JSON document as the result. /// Runs a command and delivers the first parsed JSON document as the result.
@@ -0,0 +1,506 @@
import SwiftUI
/// Colors shared by the timeline and the inspector, so a block and its row in
/// the list always read as the same thing.
enum EmphasisPalette {
static func color(_ level: Int) -> Color {
switch level {
case 1: return Color.blue
case 2: return Color.orange
case 3: return Color.pink
default: return Color.secondary
}
}
static func label(_ level: Int) -> String {
switch level {
case 1: return "Leve"
case 2: return "Média"
case 3: return "Forte"
default: return "Sem"
}
}
static func speakerColor(_ speaker: String, among speakers: [String]) -> Color {
let palette: [Color] = [.teal, .purple, .green, .indigo, .brown, .cyan]
guard let index = speakers.firstIndex(of: speaker) else { return .gray }
return palette[index % palette.count]
}
}
/// The timeline strip: four stacked tracks over one shared time axis.
///
/// Phrases are laid out as real views rather than drawn into a Canvas, because
/// every one of them is a target — click to select, drag its edge to trim,
/// right-click to change emphasis. The dense per-word energy track *is* a
/// Canvas: it has thousands of bars and nothing to hit.
struct TimelineTracksView: View {
@ObservedObject var model: PhraseReviewModel
private let rulerHeight: CGFloat = 18
private let phraseHeight: CGFloat = 46
private let energyHeight: CGFloat = 34
private let stripHeight: CGFloat = 12
private let handleWidth: CGFloat = 8
private let gutterWidth: CGFloat = 92
private let trackSpacing: CGFloat = 4
private var pps: CGFloat { CGFloat(model.pixelsPerSecond) }
private var contentWidth: CGFloat { max(320, CGFloat(model.duration) * pps) }
/// Name, icon and height of each lane, in the order they stack. The gutter
/// and the tracks are built from this one list so a label can never drift
/// off the lane it names.
private var lanes: [(label: String, icon: String, height: CGFloat)] {
[
("", "", rulerHeight),
("Zooms", "plus.magnifyingglass", stripHeight + 6),
("Frases", "text.quote", phraseHeight),
("Energia", "waveform", energyHeight),
("Emoção", "face.smiling", stripHeight),
("Locutor", "person.wave.2", stripHeight),
("Roteiro", "list.bullet.rectangle", stripHeight),
]
}
var body: some View {
VStack(spacing: 0) {
toolbar
Divider()
HStack(alignment: .top, spacing: 0) {
gutter
Divider()
timelineScroller
}
}
.background(Color(nsColor: .underPageBackgroundColor))
}
/// Fixed column naming each lane. Without it the stripes are six colours
/// with no way to tell which one is emotion and which one is the speaker.
private var gutter: some View {
VStack(alignment: .leading, spacing: trackSpacing) {
ForEach(lanes.indices, id: \.self) { index in
let lane = lanes[index]
HStack(spacing: 4) {
if !lane.icon.isEmpty {
Image(systemName: lane.icon).font(.system(size: 9))
}
Text(lane.label).font(.system(size: 10))
Spacer(minLength: 0)
}
.foregroundStyle(.secondary)
.frame(height: lane.height, alignment: .center)
}
}
.padding(.horizontal, 8)
.padding(.vertical, 8)
.frame(width: gutterWidth, alignment: .leading)
}
private var timelineScroller: some View {
ScrollViewReader { proxy in
ScrollView([.horizontal]) {
ZStack(alignment: .topLeading) {
VStack(alignment: .leading, spacing: trackSpacing) {
ruler
zoomTrack
phraseTrack
energyTrack
emotionTrack
speakerTrack
scriptTrack
}
.frame(width: contentWidth, alignment: .leading)
rangeOverlay
playhead
// Anchors the auto-scroll: one invisible marker per
// phrase, so selecting a line off-screen brings it in.
ForEach(model.phrases) { phrase in
Color.clear
.frame(width: 1, height: 1)
.offset(x: x(phrase.start))
.id(phrase.id)
}
}
.padding(.vertical, 8)
.contentShape(Rectangle())
.gesture(scrubGesture)
.contextMenu { timelineMenu }
}
.onChange(of: model.selection) { _, newValue in
guard let newValue else { return }
withAnimation(.easeOut(duration: 0.2)) {
proxy.scrollTo(newValue, anchor: .center)
}
}
}
}
// MARK: - Barra de controles
private var toolbar: some View {
HStack(spacing: 12) {
Button {
model.togglePlay()
} label: {
Image(systemName: model.isPlaying ? "pause.fill" : "play.fill")
}
.buttonStyle(.borderless)
.help("Reproduzir (espaço)")
.disabled(model.player == nil)
Text(timecode(model.currentTime))
.font(.system(.caption, design: .monospaced))
.foregroundStyle(.secondary)
Button {
model.playSelectedPhrase()
} label: {
Image(systemName: "play.rectangle")
}
.buttonStyle(.borderless)
.help("Tocar só a frase selecionada (⏎)")
.disabled(model.player == nil || model.selection == nil)
Toggle("Pular removidos", isOn: $model.skipRemoved)
.toggleStyle(.checkbox)
.font(.caption)
.help("Durante a reprodução, salta os trechos desativados — mostra como o corte ficou.")
Button {
model.addZoomForRange()
} label: {
Label("Zoom no trecho", systemImage: "plus.magnifyingglass")
}
.buttonStyle(.borderless)
.font(.caption)
.disabled(!model.hasRange)
.help("Arraste na timeline para marcar um trecho e crie um zoom nele. A escala vem de Análise de Voz.")
Spacer()
legend
Spacer()
Image(systemName: "minus.magnifyingglass").foregroundStyle(.secondary)
Slider(value: $model.pixelsPerSecond,
in: model.minPixelsPerSecond...model.maxPixelsPerSecond)
.frame(width: 130)
Image(systemName: "plus.magnifyingglass").foregroundStyle(.secondary)
}
.padding(.horizontal, 12)
.padding(.vertical, 8)
}
private var legend: some View {
HStack(spacing: 10) {
ForEach(0..<4, id: \.self) { level in
HStack(spacing: 4) {
RoundedRectangle(cornerRadius: 2)
.fill(EmphasisPalette.color(level))
.frame(width: 10, height: 10)
Text(EmphasisPalette.label(level)).font(.caption2)
}
}
}
.foregroundStyle(.secondary)
}
// MARK: - Trilhas
private var ruler: some View {
Canvas { context, size in
let step = tickStep()
var time = 0.0
while time <= model.duration {
let position = x(time)
context.stroke(
Path { $0.move(to: CGPoint(x: position, y: size.height - 6))
$0.addLine(to: CGPoint(x: position, y: size.height)) },
with: .color(.secondary.opacity(0.5))
)
context.draw(
Text(timecode(time)).font(.system(size: 9, design: .monospaced))
.foregroundColor(.secondary),
at: CGPoint(x: position + 18, y: 6)
)
time += step
}
}
.frame(width: contentWidth, height: rulerHeight)
}
private var phraseTrack: some View {
ZStack(alignment: .topLeading) {
RoundedRectangle(cornerRadius: 4)
.fill(Color.secondary.opacity(0.06))
.frame(width: contentWidth, height: phraseHeight)
ForEach(model.phrases) { phrase in
phraseBlock(phrase)
}
}
.frame(width: contentWidth, height: phraseHeight, alignment: .topLeading)
}
@ViewBuilder
private func phraseBlock(_ phrase: ReviewPhrase) -> some View {
let isSelected = model.selection == phrase.id
let color = EmphasisPalette.color(phrase.emphasis)
let fullWidth = max(2, width(from: phrase.start, to: phrase.end))
let keptWidth = max(1, width(from: phrase.trimStart, to: phrase.trimEnd))
ZStack(alignment: .topLeading) {
// The whole line, dim — what is there before the edit.
RoundedRectangle(cornerRadius: 4)
.fill(color.opacity(phrase.active ? 0.15 : 0.10))
.frame(width: fullWidth, height: phraseHeight)
// What survives: the kept span, drawn solid over it.
RoundedRectangle(cornerRadius: 4)
.fill(color.opacity(phrase.active ? 0.55 : 0.12))
.frame(width: keptWidth, height: phraseHeight)
.offset(x: width(from: phrase.start, to: phrase.trimStart))
Text(phrase.text)
.font(.system(size: 10))
.lineLimit(2)
.padding(.horizontal, 4)
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
.foregroundStyle(phrase.active ? .primary : .secondary)
.strikethrough(!phrase.active)
RoundedRectangle(cornerRadius: 4)
.stroke(isSelected ? Color.accentColor : color.opacity(0.4),
lineWidth: isSelected ? 2 : 1)
.frame(width: fullWidth, height: phraseHeight)
if isSelected && phrase.active {
trimHandle(phrase, edge: .start)
trimHandle(phrase, edge: .end)
}
}
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
.offset(x: x(phrase.start))
.contentShape(Rectangle())
.onTapGesture { model.goTo(phraseID: phrase.id) }
.contextMenu { phraseMenu(phrase) }
.help(phrase.reason.isEmpty ? phrase.text : "\(phrase.text)\n— \(phrase.reason)")
}
private func trimHandle(_ phrase: ReviewPhrase, edge: TrimEdge) -> some View {
let offset = edge == .start
? width(from: phrase.start, to: phrase.trimStart)
: width(from: phrase.start, to: phrase.trimEnd) - handleWidth
return RoundedRectangle(cornerRadius: 2)
.fill(Color.accentColor)
.frame(width: handleWidth, height: phraseHeight)
.offset(x: offset)
.gesture(
DragGesture(minimumDistance: 1)
.onChanged { value in
let time = phrase.start + Double((value.location.x) / pps)
model.trim(phrase.id, edge: edge, to: time)
}
)
.help(edge == .start ? "Arraste para cortar o começo (pula de palavra em palavra)"
: "Arraste para cortar o fim (pula de palavra em palavra)")
}
@ViewBuilder
private func phraseMenu(_ phrase: ReviewPhrase) -> some View {
Button("Tocar esta frase") {
model.goTo(phraseID: phrase.id)
model.playSelectedPhrase()
}
Button(phrase.active ? "Remover do corte" : "Trazer de volta") {
model.toggleActive(phrase.id)
}
Button("Adicionar zoom nesta frase") { model.addZoomForPhrase(phrase.id) }
Divider()
ForEach(0..<4, id: \.self) { level in
Button("Ênfase: \(EmphasisPalette.label(level))") {
model.setEmphasis(level, for: phrase.id)
}
}
Divider()
Button(phrase.isBackstage ? "Marcar como roteiro" : "Marcar como bastidor") {
model.setTrack(phrase.isBackstage ? ReviewPhrase.trackScript : ReviewPhrase.trackBackstage,
for: phrase.id)
}
if phrase.isTrimmed {
Divider()
Button("Desfazer corte da frase") { model.resetTrim(phrase.id) }
}
}
/// Per-word energy/emphasis, straight from the voice timeline — the closest
/// thing to a waveform without opening the audio again.
private var energyTrack: some View {
Canvas { context, size in
for phrase in model.phrases {
for word in phrase.words {
let start = x(word.start)
let barWidth = max(1, width(from: word.start, to: word.end) - 1)
let height = size.height * CGFloat(max(0.04, word.energy))
let rect = CGRect(x: start, y: size.height - height,
width: barWidth, height: height)
let color = word.emphasis >= 0.65 ? Color.pink
: word.emphasis >= 0.45 ? Color.orange
: Color.secondary
context.fill(Path(rect),
with: .color(color.opacity(phrase.active ? 0.6 : 0.2)))
}
}
}
.frame(width: contentWidth, height: energyHeight)
.background(RoundedRectangle(cornerRadius: 4).fill(Color.secondary.opacity(0.06)))
}
private var speakerTrack: some View {
stripTrack { phrase in
EmphasisPalette.speakerColor(phrase.speaker, among: model.speakers)
}
}
private var scriptTrack: some View {
stripTrack { phrase in phrase.isBackstage ? Color.gray : Color.mint }
}
private func stripTrack(_ color: @escaping (ReviewPhrase) -> Color) -> some View {
Canvas { context, size in
for phrase in model.phrases {
let rect = CGRect(x: x(phrase.start), y: 0,
width: max(1, width(from: phrase.start, to: phrase.end)),
height: size.height)
context.fill(Path(roundedRect: rect, cornerRadius: 2),
with: .color(color(phrase).opacity(phrase.active ? 0.7 : 0.2)))
}
}
.frame(width: contentWidth, height: stripHeight)
}
private var playhead: some View {
Rectangle()
.fill(Color.red)
.frame(width: 1.5)
.offset(x: x(model.currentTime))
.allowsHitTesting(false)
}
/// One gesture, two meanings, decided by whether the mouse moved: a click
/// parks the playhead, a drag marks in/out. Splitting them across separate
/// controls would mean choosing a tool before every action, which is
/// exactly the ceremony this screen is meant to avoid.
private var scrubGesture: some Gesture {
DragGesture(minimumDistance: 0)
.onChanged { value in
let from = Double(value.startLocation.x / pps)
let to = Double(value.location.x / pps)
if abs(value.translation.width) > 3 {
model.setRange(from: from, to: to)
model.seek(to: min(from, to))
} else {
model.clearRange()
model.seek(to: to)
}
}
}
/// The marked in/out, drawn over every track so the span reads against the
/// phrases and the energy at once.
private var rangeOverlay: some View {
Group {
if let span = model.rangeSpan {
Rectangle()
.fill(Color.accentColor.opacity(0.18))
.overlay(Rectangle().stroke(Color.accentColor.opacity(0.6), lineWidth: 1))
.frame(width: max(1, width(from: span.start, to: span.end)))
.offset(x: x(span.start))
.allowsHitTesting(false)
}
}
}
@ViewBuilder
private var timelineMenu: some View {
if model.hasRange, let span = model.rangeSpan {
Button("Adicionar zoom no trecho (\(secondsLabel(span.end - span.start)))") {
model.addZoomForRange()
}
Button("Tocar o trecho") { model.playRange(from: span.start, to: span.end) }
Button("Limpar seleção") { model.clearRange() }
} else {
Text("Arraste na timeline para marcar um trecho")
}
if let zoom = model.zoom(at: model.currentTime) {
Divider()
Button("Remover o zoom daqui") { model.removeZoom(zoom.id) }
}
}
private func secondsLabel(_ seconds: Double) -> String {
String(format: "%.1fs", seconds)
}
/// Punch-ins, on their own lane above the script: they are a second layer
/// over the same time, not a property of a phrase.
private var zoomTrack: some View {
ZStack(alignment: .topLeading) {
RoundedRectangle(cornerRadius: 3)
.fill(Color.secondary.opacity(0.06))
.frame(width: contentWidth, height: stripHeight + 6)
ForEach(model.zooms) { zoom in
RoundedRectangle(cornerRadius: 3)
.fill(Color.yellow.opacity(0.55))
.overlay(
Image(systemName: "plus.magnifyingglass")
.font(.system(size: 8)).foregroundStyle(.black.opacity(0.6))
)
.frame(width: max(6, width(from: zoom.start, to: zoom.end)),
height: stripHeight + 6)
.offset(x: x(zoom.start))
.help("Zoom marcado — \(secondsLabel(zoom.end - zoom.start)). A escala vem de Análise de Voz.")
.contextMenu {
Button("Remover este zoom") { model.removeZoom(zoom.id) }
}
}
}
.frame(width: contentWidth, height: stripHeight + 6, alignment: .topLeading)
}
/// Delivery emotion per phrase — the fourth signal to read against the text.
private var emotionTrack: some View {
stripTrack { phrase in
switch phrase.emotion {
case "excited": return .orange
case "tense": return .red
case "calm": return .blue
case "reflective": return .purple
default: return .secondary
}
}
}
// MARK: - Escala
private func x(_ time: Double) -> CGFloat { CGFloat(time) * pps }
private func width(from: Double, to: Double) -> CGFloat {
max(0, CGFloat(to - from) * pps)
}
/// Ruler spacing that keeps labels ~80pt apart at any zoom.
private func tickStep() -> Double {
let candidates: [Double] = [1, 2, 5, 10, 15, 30, 60, 120, 300, 600]
let wanted = 80 / Double(pps)
return candidates.first { $0 >= wanted } ?? 600
}
private func timecode(_ seconds: Double) -> String {
let total = Int(seconds.rounded(.down))
return String(format: "%02d:%02d", total / 60, total % 60)
}
}
+3 -3
View File
@@ -271,7 +271,7 @@ struct TranscriptionView: View {
Toggle("Marcar o que foi dito na timeline", isOn: $batchMarkers) Toggle("Marcar o que foi dito na timeline", isOn: $batchMarkers)
Divider() Divider()
Toggle("Exportar legendas SRT", isOn: $batchSubtitles) Toggle("Gerar legenda comum (texto editável no FCP)", isOn: $batchSubtitles)
Divider() Divider()
batchOptionRow( batchOptionRow(
@@ -793,7 +793,7 @@ struct TranscriptionView: View {
if batchFillers { operations.append("remove_filler_words") } if batchFillers { operations.append("remove_filler_words") }
if batchPhrases { operations.append("edit_by_transcript") } if batchPhrases { operations.append("edit_by_transcript") }
if batchMarkers { operations.append("transcript_markers") } if batchMarkers { operations.append("transcript_markers") }
if batchSubtitles { operations.append("export_srt") } if batchSubtitles { operations.append("generate_plain_subtitles") }
// Runs last, on the timing already cut by any earlier steps (see the // Runs last, on the timing already cut by any earlier steps (see the
// "Abrir no Final Cut Pro" fallback chain and exportSubtitles()'s own // "Abrir no Final Cut Pro" fallback chain and exportSubtitles()'s own
// preference for `processedPath` — same reasoning). // preference for `processedPath` — same reasoning).
@@ -834,7 +834,7 @@ struct TranscriptionView: View {
} }
let nextPath = result?["path"] as? String ?? currentPath let nextPath = result?["path"] as? String ?? currentPath
if operation == "remove_silences" { processedPath = nextPath } if operation == "remove_silences" { processedPath = nextPath }
if operation == "export_srt" { subtitlePaths = result?["paths"] as? [String] ?? [] } if operation == "generate_plain_subtitles" { subtitlePaths = [nextPath] }
if operation == "generate_dynamic_subtitles" { dynamicSubtitlesPath = nextPath } if operation == "generate_dynamic_subtitles" { dynamicSubtitlesPath = nextPath }
processBatchStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder) processBatchStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder)
} }
+68 -4
View File
@@ -23,6 +23,7 @@ struct VoiceAnalysisView: View {
} else { } else {
energySection energySection
emphasisSection emphasisSection
zoomSection
weightsSection weightsSection
emotionSection emotionSection
resetSection resetSection
@@ -90,6 +91,44 @@ struct VoiceAnalysisView: View {
} }
} }
private var zoomSection: some View {
Section {
sliderRow(
title: "Zoom na ênfase",
value: $config.zoomScale,
range: 1.0...3.0,
readout: "\(Int(config.zoomScale * 100))%",
help: "Fator aplicado nos punch-ins de ênfase. 130% equivale a escala 1,30 no Final Cut."
)
Picker("Movimento", selection: $config.zoomMode) {
Text("Zoom in e out").tag("in_out")
Text("Só zoom in").tag("in")
Text("Só zoom out").tag("out")
}
.onChange(of: config.zoomMode) { _, _ in save() }
sliderRow(
title: "Velocidade do zoom in",
value: $config.zoomEaseIn,
range: 0.05...2.0,
readout: String(format: "%.2fs", config.zoomEaseIn),
help: "Duração da entrada do zoom. Menor é mais rápido."
)
sliderRow(
title: "Velocidade do zoom out",
value: $config.zoomEaseOut,
range: 0.01...2.0,
readout: String(format: "%.2fs", config.zoomEaseOut),
help: "Duração da saída do zoom. Menor é mais seco."
)
} header: {
Text("Zoom de Ênfase")
} footer: {
Text("Esses valores viram o padrão para ações de zoom que não trouxerem scale/ease/ease_out no JSON da edição por voz.")
.font(.caption)
.foregroundStyle(.secondary)
}
}
// MARK: - Emoção // MARK: - Emoção
private var emotionSection: some View { private var emotionSection: some View {
@@ -129,13 +168,14 @@ struct VoiceAnalysisView: View {
title: String, title: String,
value: Binding<Double>, value: Binding<Double>,
range: ClosedRange<Double> = 0...1, range: ClosedRange<Double> = 0...1,
readout: String? = nil,
help: String? = nil help: String? = nil
) -> some View { ) -> some View {
VStack(alignment: .leading, spacing: 2) { VStack(alignment: .leading, spacing: 2) {
HStack { HStack {
Text(title) Text(title)
Spacer() Spacer()
Text(String(format: "%.2f", value.wrappedValue)) Text(readout ?? String(format: "%.2f", value.wrappedValue))
.monospacedDigit() .monospacedDigit()
.foregroundStyle(.secondary) .foregroundStyle(.secondary)
} }
@@ -188,6 +228,10 @@ struct VoiceAnalysisConfig {
var weightDuration: Double var weightDuration: Double
var emotionEnabled: Bool var emotionEnabled: Bool
var emotionSensitivity: Double var emotionSensitivity: Double
var zoomScale: Double
var zoomMode: String
var zoomEaseIn: Double
var zoomEaseOut: Double
static let defaults = VoiceAnalysisConfig( static let defaults = VoiceAnalysisConfig(
energyThreshold: 0.5, energyThreshold: 0.5,
@@ -198,7 +242,11 @@ struct VoiceAnalysisConfig {
weightPause: 0.15, weightPause: 0.15,
weightDuration: 0.10, weightDuration: 0.10,
emotionEnabled: false, emotionEnabled: false,
emotionSensitivity: 0.5 emotionSensitivity: 0.5,
zoomScale: 1.30,
zoomMode: "in_out",
zoomEaseIn: 0.25,
zoomEaseOut: 0.04
) )
init( init(
@@ -210,7 +258,11 @@ struct VoiceAnalysisConfig {
weightPause: Double, weightPause: Double,
weightDuration: Double, weightDuration: Double,
emotionEnabled: Bool, emotionEnabled: Bool,
emotionSensitivity: Double emotionSensitivity: Double,
zoomScale: Double,
zoomMode: String,
zoomEaseIn: Double,
zoomEaseOut: Double
) { ) {
self.energyThreshold = energyThreshold self.energyThreshold = energyThreshold
self.emphasisThreshold = emphasisThreshold self.emphasisThreshold = emphasisThreshold
@@ -221,6 +273,10 @@ struct VoiceAnalysisConfig {
self.weightDuration = weightDuration self.weightDuration = weightDuration
self.emotionEnabled = emotionEnabled self.emotionEnabled = emotionEnabled
self.emotionSensitivity = emotionSensitivity self.emotionSensitivity = emotionSensitivity
self.zoomScale = zoomScale
self.zoomMode = zoomMode
self.zoomEaseIn = zoomEaseIn
self.zoomEaseOut = zoomEaseOut
} }
/// Lê a resposta do bridge, caindo no padrão para qualquer campo ausente. /// Lê a resposta do bridge, caindo no padrão para qualquer campo ausente.
@@ -236,7 +292,11 @@ struct VoiceAnalysisConfig {
weightPause: weights["pause_before"] as? Double ?? defaults.weightPause, weightPause: weights["pause_before"] as? Double ?? defaults.weightPause,
weightDuration: weights["duration"] as? Double ?? defaults.weightDuration, weightDuration: weights["duration"] as? Double ?? defaults.weightDuration,
emotionEnabled: json["emotion_enabled"] as? Bool ?? defaults.emotionEnabled, emotionEnabled: json["emotion_enabled"] as? Bool ?? defaults.emotionEnabled,
emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity,
zoomScale: json["zoom_scale"] as? Double ?? defaults.zoomScale,
zoomMode: json["zoom_mode"] as? String ?? defaults.zoomMode,
zoomEaseIn: json["zoom_ease_in"] as? Double ?? defaults.zoomEaseIn,
zoomEaseOut: json["zoom_ease_out"] as? Double ?? defaults.zoomEaseOut
) )
} }
@@ -253,6 +313,10 @@ struct VoiceAnalysisConfig {
], ],
"emotion_enabled": emotionEnabled, "emotion_enabled": emotionEnabled,
"emotion_sensitivity": emotionSensitivity, "emotion_sensitivity": emotionSensitivity,
"zoom_scale": zoomScale,
"zoom_mode": zoomMode,
"zoom_ease_in": zoomEaseIn,
"zoom_ease_out": zoomEaseOut,
] ]
} }
} }
+808
View File
@@ -0,0 +1,808 @@
import SwiftUI
import AppKit
/// Guia passo a passo do fluxo completo: projeto → transcrição → análise de
/// voz → copiar para o chat e trazer as decisões → revisar as ênfases →
/// processamento final. Existe para que o usuário não precise entender a ordem
/// certa de botões espalhados em várias abas — cada etapa só libera a próxima
/// quando o passo anterior terminou, e a "ponte" com o chat (que hoje exigia
/// sair do app e escolher um arquivo na mão) vira copiar/colar assistido
/// dentro da própria tela.
enum WizardStep: Int, CaseIterable, Identifiable {
case projeto, transcricao, analise, exportarChat, revisar, finalizar, concluido
var id: Int { rawValue }
var titulo: String {
switch self {
case .projeto: return "Projeto"
case .transcricao: return "Transcrever"
case .analise: return "Analisar voz"
case .exportarChat: return "Decisões da IA"
case .revisar: return "Revisar ênfases"
case .finalizar: return "Processar"
case .concluido: return "Concluído"
}
}
}
struct WizardView: View {
@State private var step: WizardStep = .projeto
// Passo 1 — projeto
@State private var outputFolder: String?
@State private var projectPath: String?
@State private var catalog: Catalog?
// Passo 2 — transcrição
@State private var isTranscribing = false
@State private var transcribeProgress: Double = 0
@State private var transcribeStage = ""
@State private var transcribeResults: [TranscriptResult] = []
// Passo 3 — análise de voz
@State private var isAnalyzing = false
@State private var voiceTimelinePath: String?
@State private var voiceAnalysisMessage = ""
@State private var acousticsAvailable: Bool?
@State private var showVoiceTimelineReuseAlert = false
@State private var existingVoiceTimelinePath: String?
// Passo 4 — enviar ao chat e trazer as decisões de volta
@State private var copiedFeedback = ""
@State private var decisionsText = ""
@State private var isApplyingDecisions = false
@State private var appliedPath: String?
@State private var skippedVoiceEdit = false
// Passo 5 — revisar ênfases
@StateObject private var reviewModel = PhraseReviewModel()
@State private var reviewLoadedFor: String?
@State private var phraseReviewPath: String?
// Passo 6 — processamento final
@State private var finalSilences = true
@State private var finalFillers = false
@State private var finalSubtitles = true
@State private var finalDynamicSubtitles = false
@State private var isFinalizing = false
@State private var finalStatus = ""
@State private var finalPath: String?
@State private var errorMessage: String?
var body: some View {
VStack(spacing: 0) {
stepperHeader
.padding(.horizontal, 24)
.padding(.top, 20)
.padding(.bottom, 16)
Divider()
// A revisão é uma sala de edição, não um formulário: ela precisa da
// largura toda e rola por conta própria (timeline horizontal, lista
// vertical). As demais etapas continuam na coluna estreita, que é o
// que mantém um passo a passo legível.
if step == .revisar {
revisarStep
} else {
ScrollView {
VStack(alignment: .leading, spacing: 18) {
if let errorMessage, !errorMessage.isEmpty {
Label(errorMessage, systemImage: "exclamationmark.triangle.fill")
.foregroundStyle(.red)
.padding(.top, 4)
}
content
}
.padding(24)
.frame(maxWidth: 640, alignment: .leading)
.frame(maxWidth: .infinity)
}
}
Divider()
navFooter
.padding(.horizontal, 24)
.padding(.vertical, 16)
}
.task {
loadProjectConfig()
await loadCatalog()
}
.alert("Análise de voz já existe", isPresented: $showVoiceTimelineReuseAlert) {
Button("Usar existente") {
if let existingVoiceTimelinePath {
voiceTimelinePath = existingVoiceTimelinePath
voiceAnalysisMessage = "Reaproveitando análise existente: \(existingVoiceTimelinePath)"
}
}
Button("Reprocessar") {
analyzeVoice(forceReprocess: true)
}
Button("Cancelar", role: .cancel) {}
} message: {
Text("Já existe um arquivo voice_timeline para este projeto. Quer manter o processamento anterior para ganhar tempo?")
}
}
// MARK: - Cabeçalho com os passos
private var stepperHeader: some View {
HStack(spacing: 6) {
ForEach(WizardStep.allCases) { s in
HStack(spacing: 6) {
ZStack {
Circle()
.fill(colorFor(s))
.frame(width: 24, height: 24)
if s.rawValue < step.rawValue {
Image(systemName: "checkmark")
.font(.caption2.weight(.bold))
.foregroundStyle(.white)
} else {
Text("\(s.rawValue + 1)")
.font(.caption2.weight(.bold))
.foregroundStyle(s == step ? .white : .secondary)
}
}
Text(s.titulo)
.font(.caption)
.foregroundStyle(s == step ? .primary : .secondary)
.fontWeight(s == step ? .semibold : .regular)
}
if s != WizardStep.allCases.last {
Rectangle()
.fill(s.rawValue < step.rawValue ? Color.accentColor : Color.secondary.opacity(0.25))
.frame(height: 2)
.frame(maxWidth: .infinity)
}
}
}
}
private func colorFor(_ s: WizardStep) -> Color {
if s.rawValue < step.rawValue { return .accentColor }
if s == step { return .accentColor }
return Color.secondary.opacity(0.25)
}
// MARK: - Conteúdo por etapa
@ViewBuilder
private var content: some View {
switch step {
case .projeto: projetoStep
case .transcricao: transcricaoStep
case .analise: analiseStep
case .exportarChat: exportarChatStep
case .revisar: revisarStep
case .finalizar: finalizarStep
case .concluido: concluidoStep
}
}
private var projetoStep: some View {
VStack(alignment: .leading, spacing: 16) {
Text("1. Escolha o projeto").font(.title3.weight(.semibold))
Text("A pasta é onde tudo o que for gerado nesse fluxo fica salvo. O arquivo é o .fcpxml exportado do Final Cut Pro.")
.font(.callout).foregroundStyle(.secondary)
fieldRow(icon: "folder", label: outputFolder ?? "Nenhuma pasta selecionada", isSet: outputFolder != nil) {
pickOutputFolder()
}
fieldRow(icon: "doc.text", label: projectPath.map { URL(fileURLWithPath: $0).lastPathComponent } ?? "Nenhum arquivo selecionado", isSet: projectPath != nil) {
pickProjectFile()
}
if looksLikeGeneratedFile(projectPath) {
Label("Esse arquivo parece já ter sido processado por este fluxo (o nome tem um sufixo como \"_voice_edit\" ou \"_silence_removed\"). Rodar o wizard de novo em cima dele reaplica os cortes por cima de cortes já feitos. Selecione o .fcpxml original do Final Cut, a menos que a intenção seja mesmo reprocessar.",
systemImage: "exclamationmark.triangle.fill")
.font(.caption).foregroundStyle(.orange)
}
if (catalog?.installedCount ?? 0) == 0 {
Label("Nenhum modelo de transcrição instalado. Baixe um na aba \"Modelos\" antes de continuar.",
systemImage: "exclamationmark.triangle.fill")
.font(.caption).foregroundStyle(.orange)
}
}
}
private var transcricaoStep: some View {
VStack(alignment: .leading, spacing: 16) {
Text("2. Transcreva o áudio").font(.title3.weight(.semibold))
Text("Roda localmente com o modelo escolhido na aba Modelos. Vira a base de tudo que vem depois — o corte por voz, as legendas, os marcadores.")
.font(.callout).foregroundStyle(.secondary)
Button {
startTranscription()
} label: {
if isTranscribing {
HStack { ProgressView().controlSize(.small); Text(transcribeStage.isEmpty ? "Transcrevendo…" : transcribeStage) }
.frame(maxWidth: .infinity)
} else {
Label(transcribeResults.isEmpty ? "Transcrever" : "Transcrever novamente", systemImage: "waveform")
.frame(maxWidth: .infinity)
}
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
.disabled(isTranscribing || projectPath == nil || outputFolder == nil)
if isTranscribing {
VStack(alignment: .leading, spacing: 6) {
ProgressView(value: transcribeProgress)
Text("\(Int(transcribeProgress * 100))%").font(.caption).foregroundStyle(.secondary).monospacedDigit()
}
}
if !transcribeResults.isEmpty {
ForEach(transcribeResults, id: \.media) { r in
VStack(alignment: .leading, spacing: 4) {
HStack {
Image(systemName: "checkmark.circle.fill").foregroundStyle(.green)
Text(r.media).font(.body.weight(.medium))
Spacer()
Text("\(r.language) · \(r.words) palavras").font(.caption).foregroundStyle(.secondary)
}
Text(r.preview).font(.caption).foregroundStyle(.secondary).lineLimit(2)
}
.padding(12)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
}
}
}
}
private var analiseStep: some View {
VStack(alignment: .leading, spacing: 16) {
Text("3. Analise a voz").font(.title3.weight(.semibold))
Text("Gera o JSON com transcrição, locutor e intensidade (pitch/energia/ritmo) por palavra — é esse arquivo que o chat lê para decidir o que cortar. Não corta nada sozinho.")
.font(.callout).foregroundStyle(.secondary)
Button {
analyzeVoice()
} label: {
if isAnalyzing {
HStack { ProgressView().controlSize(.small); Text("Analisando…") }.frame(maxWidth: .infinity)
} else {
Label(voiceTimelinePath == nil ? "Analisar voz" : "Analisar novamente", systemImage: "waveform.badge.magnifyingglass")
.frame(maxWidth: .infinity)
}
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
.disabled(isAnalyzing || projectPath == nil || outputFolder == nil)
if let voiceTimelinePath {
VStack(alignment: .leading, spacing: 6) {
Label("Análise pronta", systemImage: "checkmark.circle.fill").foregroundStyle(.green)
Text(voiceTimelinePath).font(.caption).foregroundStyle(.secondary).lineLimit(1).truncationMode(.middle)
}
.padding(12)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
if acousticsAvailable == false {
VStack(alignment: .leading, spacing: 4) {
Label("Sem análise acústica real", systemImage: "exclamationmark.triangle.fill")
.font(.caption.weight(.semibold)).foregroundStyle(.orange)
Text("Falta o componente \"librosa\" — os cortes ainda são decididos pelo texto, mas o chat não vai propor zoom com confiança. Instale em Avançado → Modelos → \"Análise Acústica\", e refaça esta etapa depois.")
.font(.caption).foregroundStyle(.secondary)
}
.padding(12)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.orange.opacity(0.08)))
}
}
}
}
private var exportarChatStep: some View {
VStack(alignment: .leading, spacing: 16) {
Text("4. Envie para o chat decidir os cortes").font(.title3.weight(.semibold))
Text("Esta é a única etapa manual que sobra: o julgamento de qual tomada usar, onde dar zoom e o que escrever na tela é feito pela IA numa conversa, não por um botão. Copie abaixo, cole numa sessão do Claude e peça pra rodar a skill \"editar-por-voz\".")
.font(.callout).foregroundStyle(.secondary)
if let voiceTimelinePath {
Button {
copyForChat(path: voiceTimelinePath)
} label: {
Label("Copiar para colar no chat", systemImage: "doc.on.clipboard")
.frame(maxWidth: .infinity)
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
if !copiedFeedback.isEmpty {
Label(copiedFeedback, systemImage: "checkmark.circle.fill")
.font(.caption).foregroundStyle(.green)
}
VStack(alignment: .leading, spacing: 8) {
Text("O que é copiado").font(.caption.weight(.semibold)).foregroundStyle(.secondary)
Text("Um pedido pronto + o conteúdo de \(URL(fileURLWithPath: voiceTimelinePath).lastPathComponent), já formatado. É só colar (⌘V) numa conversa com o Claude.")
.font(.caption).foregroundStyle(.secondary)
}
.padding(12)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
Divider().padding(.vertical, 4)
Text("Cole aqui o que o chat devolveu").font(.callout.weight(.semibold))
Text("Na próxima etapa essas decisões aparecem já marcadas na timeline, frase por frase, para você lapidar.")
.font(.caption).foregroundStyle(.secondary)
HStack {
Button {
if let s = NSPasteboard.general.string(forType: .string) {
decisionsText = s
}
} label: {
Label("Colar da área de transferência", systemImage: "list.clipboard")
}
Spacer()
if !decisionsText.isEmpty {
Label(jsonIsValid ? "JSON válido" : "JSON inválido",
systemImage: jsonIsValid ? "checkmark.circle.fill" : "xmark.circle.fill")
.font(.caption)
.foregroundStyle(jsonIsValid ? .green : .red)
}
}
TextEditor(text: $decisionsText)
.font(.system(.caption, design: .monospaced))
.frame(minHeight: 140)
.padding(8)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
.overlay(RoundedRectangle(cornerRadius: 8).stroke(Color.secondary.opacity(0.2)))
Button {
applyDecisions()
} label: {
if isApplyingDecisions {
HStack { ProgressView().controlSize(.small); Text("Aplicando…") }
.frame(maxWidth: .infinity)
} else {
Label("Aplicar decisões", systemImage: "checkmark.seal")
.frame(maxWidth: .infinity)
}
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
.disabled(isApplyingDecisions || !jsonIsValid)
if let appliedPath {
Label("Decisões aplicadas — \(URL(fileURLWithPath: appliedPath).lastPathComponent)",
systemImage: "checkmark.circle.fill")
.font(.caption).foregroundStyle(.green)
}
Divider()
Button("Pular esta etapa (revisar as ênfases direto, sem passar pela IA)") {
skippedVoiceEdit = true
appliedPath = nil
decisionsText = ""
}
.buttonStyle(.plain)
.font(.caption)
.foregroundStyle(.secondary)
} else {
Label("Volte ao passo anterior e rode a análise de voz primeiro.", systemImage: "exclamationmark.triangle.fill")
.font(.caption).foregroundStyle(.orange)
}
}
}
/// Etapa 5 — a sala de edição. Diferente das outras, não é um formulário
/// dentro da coluna do assistente: ocupa a janela toda e se carrega sozinha
/// na primeira vez que aparece para aquela análise de voz.
private var revisarStep: some View {
Group {
if voiceTimelinePath != nil {
PhraseReviewView(model: reviewModel)
} else {
VStack(spacing: 8) {
Label("Volte ao passo 3 e rode a análise de voz primeiro.",
systemImage: "exclamationmark.triangle.fill")
.foregroundStyle(.orange)
}
.frame(maxWidth: .infinity, maxHeight: .infinity)
}
}
.onAppear { loadReviewIfNeeded() }
}
private var finalizarStep: some View {
VStack(alignment: .leading, spacing: 16) {
Text("6. Finalize o corte").font(.title3.weight(.semibold))
Text("Últimos passos automáticos, sem decisão envolvida — rodam com os parâmetros já configurados na aba \"Análise de Voz\" / \"Legendas Dinâmicas\".")
.font(.callout).foregroundStyle(.secondary)
Toggle("Remover silêncios do áudio", isOn: $finalSilences)
Toggle("Remover palavras de preenchimento", isOn: $finalFillers)
Toggle("Gerar legenda comum (texto editável no FCP)", isOn: $finalSubtitles)
Toggle("Gerar legendas dinâmicas (estilo configurado na aba própria)", isOn: $finalDynamicSubtitles)
Button {
finalizeProcessing()
} label: {
if isFinalizing {
HStack { ProgressView().controlSize(.small); Text(finalStatus.isEmpty ? "Processando…" : finalStatus) }
.frame(maxWidth: .infinity)
} else {
Label("Processar", systemImage: "play.fill").frame(maxWidth: .infinity)
}
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
.disabled(isFinalizing || (!finalSilences && !finalFillers && !finalSubtitles && !finalDynamicSubtitles))
if !finalStatus.isEmpty && !isFinalizing {
Text(finalStatus).font(.caption).foregroundStyle(.secondary)
}
}
}
private var concluidoStep: some View {
VStack(alignment: .leading, spacing: 16) {
Label("Concluído", systemImage: "checkmark.seal.fill")
.font(.title3.weight(.semibold))
.foregroundStyle(.green)
if let finalPath {
Text(finalPath).font(.caption).foregroundStyle(.secondary).lineLimit(1).truncationMode(.middle)
HStack {
Button("Abrir no Final Cut Pro") { NSWorkspace.shared.open(URL(fileURLWithPath: finalPath)) }
.buttonStyle(.borderedProminent)
Button("Mostrar no Finder") {
NSWorkspace.shared.activateFileViewerSelecting([URL(fileURLWithPath: finalPath)])
}
}
}
Divider().padding(.vertical, 8)
Button("Começar outro projeto") { resetWizard() }
}
}
// MARK: - Navegação
private var navFooter: some View {
HStack {
if step != .projeto && step != .concluido {
Button("Voltar") { goBack() }
}
Spacer()
if step != .concluido {
Button(step == .finalizar ? "Concluir" : "Continuar") { goNext() }
.buttonStyle(.borderedProminent)
.disabled(!canAdvance)
}
}
}
private var canAdvance: Bool {
switch step {
case .projeto: return outputFolder != nil && projectPath != nil
case .transcricao: return !transcribeResults.isEmpty
case .analise: return voiceTimelinePath != nil
case .exportarChat: return appliedPath != nil || skippedVoiceEdit
// Revisar é opcional: a sugestão da IA já é utilizável como veio, então
// o botão nunca trava aqui — o passo existe para lapidar, não para
// exigir mais uma confirmação.
case .revisar: return true
case .finalizar: return finalPath != nil && !isFinalizing
case .concluido: return false
}
}
private func goNext() {
guard let next = WizardStep(rawValue: step.rawValue + 1) else { return }
// Sair da revisão grava o que foi decidido (e as ações derivadas dela)
// ao lado da análise de voz. Nada é renderizado aqui: a etapa 6 é que
// lê esse arquivo para dar zoom e legenda dinâmica só nas ênfases.
if step == .revisar {
reviewModel.save { path in
phraseReviewPath = path
}
}
step = next
}
private func goBack() {
guard let prev = WizardStep(rawValue: step.rawValue - 1) else { return }
step = prev
}
private func resetWizard() {
step = .projeto
transcribeResults = []
voiceTimelinePath = nil
voiceAnalysisMessage = ""
decisionsText = ""
appliedPath = nil
skippedVoiceEdit = false
reviewLoadedFor = nil
phraseReviewPath = nil
finalStatus = ""
finalPath = nil
errorMessage = nil
}
// MARK: - Componentes auxiliares
@ViewBuilder
private func fieldRow(icon: String, label: String, isSet: Bool, action: @escaping () -> Void) -> some View {
HStack {
Image(systemName: icon).foregroundStyle(isSet ? .primary : .secondary)
Text(label).lineLimit(1).truncationMode(.middle).foregroundStyle(isSet ? .primary : .secondary)
Spacer()
Button("Escolher…", action: action)
}
.padding(12)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
}
/// Todo output do fluxo carrega um destes sufixos no nome (ver
/// `_derived_output` / suffixes usados por `apply_voice_actions`,
/// `remove_silences`, `generate_dynamic_subtitles` em
/// `admin/models_api.py`). Selecionar um deles como "o projeto" no passo
/// 1 é o erro que gerou arquivos como `_voice_edit_voice_edit_...`: os
/// cortes de voz assumem timestamps da mídia ORIGINAL, então reaplicá-los
/// sobre um arquivo já cortado desloca tudo silenciosamente.
private static let generatedSuffixes = [
"_voice_edit", "_silence_removed", "_dynamic_subtitles",
"_transcript_edit", "_fillers_removed", "_markers",
]
private func looksLikeGeneratedFile(_ path: String?) -> Bool {
guard let path else { return false }
let stem = URL(fileURLWithPath: path).deletingPathExtension().lastPathComponent
return Self.generatedSuffixes.contains { stem.contains($0) }
}
private var jsonIsValid: Bool {
guard let data = decisionsText.data(using: .utf8), !decisionsText.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { return false }
return (try? JSONSerialization.jsonObject(with: data)) != nil
}
// MARK: - Ações — Python bridge
private func loadProjectConfig() {
PythonBridge.call(command: "project_config") { result, _ in
DispatchQueue.main.async {
guard let result, result["ok"] as? Bool == true else { return }
if let folder = result["folder"] as? String, !folder.isEmpty { outputFolder = folder }
if let file = result["file"] as? String, !file.isEmpty { projectPath = file }
}
}
}
private func loadCatalog() async {
PythonBridge.call(command: "catalog") { result, _ in
DispatchQueue.main.async {
if let result { catalog = Catalog(json: result) }
}
}
}
private func pickOutputFolder() {
let panel = NSOpenPanel()
panel.canChooseFiles = false
panel.canChooseDirectories = true
panel.allowsMultipleSelection = false
panel.prompt = "Usar esta pasta"
panel.message = "Escolha a pasta onde os resultados serão salvos."
if panel.runModal() == .OK, let url = panel.url {
outputFolder = url.path
PythonBridge.call(command: "set_project_config", arguments: ["folder": url.path]) { _, _ in }
}
}
private func pickProjectFile() {
let panel = NSOpenPanel()
panel.canChooseFiles = true
panel.canChooseDirectories = false
panel.allowsMultipleSelection = false
panel.prompt = "Selecionar"
panel.message = "Selecione o arquivo (.fcpxml) ou o bundle (.fcpxmld) exportado pelo Final Cut Pro."
if panel.runModal() == .OK, let url = panel.url {
let ext = url.pathExtension.lowercased()
if ext == "fcpxml" || ext == "fcpxmld" || ext == "xml" {
projectPath = url.path
PythonBridge.call(command: "set_project_config", arguments: ["file": url.path]) { _, _ in }
} else {
errorMessage = "Selecione um arquivo .fcpxml, .fcpxmld ou .xml do Final Cut Pro."
}
}
}
private func startTranscription() {
guard let projectPath, let outputFolder else { return }
isTranscribing = true
errorMessage = nil
transcribeResults = []
transcribeProgress = 0
PythonBridge.run(command: "transcribe", arguments: ["path": projectPath, "output_dir": outputFolder]) { obj in
DispatchQueue.main.async {
let type = obj["type"] as? String
if type == "progress" {
transcribeProgress = (obj["fraction"] as? NSNumber)?.doubleValue ?? 0
transcribeStage = obj["stage"] as? String ?? ""
} else if type == "error" {
errorMessage = obj["message"] as? String ?? "Erro na transcrição."
} else if type == "result", let arr = obj["transcripts"] as? [[String: Any]] {
transcribeResults = arr.map(TranscriptResult.init)
}
}
} completion: { code, err in
DispatchQueue.main.async {
isTranscribing = false
transcribeProgress = 1
if code != 0 && transcribeResults.isEmpty {
errorMessage = err ?? "A transcrição falhou."
}
}
}
}
private func analyzeVoice(forceReprocess: Bool = false) {
guard let projectPath, let outputFolder else { return }
isAnalyzing = true
errorMessage = nil
PythonBridge.call(command: "analyze_voice", arguments: [
"path": projectPath,
"output_dir": outputFolder,
"force_reprocess": forceReprocess,
]) { result, err in
DispatchQueue.main.async {
isAnalyzing = false
guard result?["ok"] as? Bool == true else {
errorMessage = result?["error"] as? String ?? err ?? "Falha ao analisar a voz."
return
}
if result?["reused"] as? Bool == true, !forceReprocess {
let timelines = result?["timelines"] as? [String] ?? []
existingVoiceTimelinePath = timelines.first ?? extractPath(from: result?["message"] as? String ?? "", marker: "**Timeline JSON**:")
showVoiceTimelineReuseAlert = true
return
}
let message = result?["message"] as? String ?? ""
voiceAnalysisMessage = message
if let path = extractPath(from: message, marker: "**Timeline JSON**:") {
voiceTimelinePath = path
} else {
voiceTimelinePath = nil
// ok:true não garante que a análise gerou timeline — se
// não houver fala detectável no áudio, o Python volta com
// sucesso mas sem "Timeline JSON" na mensagem. Sem isso
// aqui, a etapa parecia não fazer nada.
errorMessage = "A análise terminou mas não encontrou fala reconhecível no áudio. Mensagem do motor: " + (message.isEmpty ? "(vazia)" : message)
}
checkAcoustics()
}
}
}
/// A ênfase de voz (energia/tom) depende do `librosa`, dependência
/// opcional. Sem ela, a análise ainda transcreve e corta pelo texto,
/// mas nunca deveria propor zoom — por isso avisamos aqui, no ponto
/// onde o usuário sentiria falta, em vez de só na aba Modelos.
private func checkAcoustics() {
PythonBridge.call(command: "acoustics_capability") { result, _ in
DispatchQueue.main.async {
guard let result, result["ok"] as? Bool == true else { return }
acousticsAvailable = result["available"] as? Bool
}
}
}
/// Localiza uma linha markdown do tipo "- **Marker**: valor" (usado nas
/// mensagens do bridge Python) e devolve o valor. Aceita o marcador de
/// lista "- " opcional antes dos asteriscos.
private func extractPath(from message: String, marker: String) -> String? {
for line in message.split(separator: "\n") {
var trimmed = Substring(line.trimmingCharacters(in: .whitespaces))
if trimmed.hasPrefix("- ") { trimmed = trimmed.dropFirst(2) }
if trimmed.hasPrefix(marker) {
return trimmed.dropFirst(marker.count).trimmingCharacters(in: .whitespaces)
}
}
return nil
}
private func copyForChat(path: String) {
guard let content = try? String(contentsOfFile: path, encoding: .utf8) else {
errorMessage = "Não foi possível ler \(path)."
return
}
let prompt = """
Use a skill "editar-por-voz" para decidir os cortes deste projeto a partir da timeline de voz abaixo. Devolva só o JSON de decisões (cortes, zooms, textos, marcadores) pronto para eu colar de volta no app.
```json
\(content)
```
"""
let pasteboard = NSPasteboard.general
pasteboard.clearContents()
pasteboard.setString(prompt, forType: .string)
copiedFeedback = "Copiado — cole (⌘V) numa conversa com o Claude."
}
/// Monta a revisão uma vez por análise de voz. Voltar e avançar de novo não
/// recarrega: isso jogaria fora as edições manuais em silêncio, que é
/// exatamente o que esta tela existe para preservar.
private func loadReviewIfNeeded() {
guard let voiceTimelinePath, reviewLoadedFor != voiceTimelinePath else { return }
reviewLoadedFor = voiceTimelinePath
// A pasta do projeto e a do .fcpxml entram como onde procurar a mídia:
// a análise de voz guarda só o nome do arquivo, não o caminho.
reviewModel.load(
voiceTimelinePath: voiceTimelinePath,
decisionsJSON: decisionsText,
outputFolder: outputFolder,
mediaFolder: projectPath.map { URL(fileURLWithPath: $0).deletingLastPathComponent().path }
)
if let projectPath { reviewModel.loadProjectFormat(projectPath: projectPath) }
}
private func applyDecisions() {
guard let projectPath, let outputFolder,
let data = decisionsText.data(using: .utf8),
let parsed = try? JSONSerialization.jsonObject(with: data) else { return }
isApplyingDecisions = true
errorMessage = nil
PythonBridge.call(command: "apply_voice_actions", arguments: [
"path": projectPath,
"output_dir": outputFolder,
"actions": parsed,
]) { result, err in
DispatchQueue.main.async {
isApplyingDecisions = false
guard result?["ok"] as? Bool == true else {
errorMessage = result?["error"] as? String ?? err ?? "Falha ao aplicar as decisões."
return
}
appliedPath = result?["path"] as? String ?? projectPath
skippedVoiceEdit = false
}
}
}
private func finalizeProcessing() {
guard let outputFolder else { return }
let startPath = appliedPath ?? projectPath
guard let startPath else { return }
var operations: [String] = []
if finalSilences { operations.append("remove_silences") }
if finalFillers { operations.append("remove_filler_words") }
if finalSubtitles { operations.append("generate_plain_subtitles") }
if finalDynamicSubtitles { operations.append("generate_dynamic_subtitles") }
guard !operations.isEmpty else { return }
isFinalizing = true
errorMessage = nil
finalStatus = "Iniciando…"
finalizeStep(operations, index: 0, currentPath: startPath, outputFolder: outputFolder)
}
private func finalizeStep(_ operations: [String], index: Int, currentPath: String, outputFolder: String) {
guard index < operations.count else {
isFinalizing = false
finalStatus = "Processamento concluído."
finalPath = currentPath
return
}
let operation = operations[index]
finalStatus = "Processando: \(operation)…"
PythonBridge.call(command: operation, arguments: ["path": currentPath, "output_dir": outputFolder]) { result, err in
DispatchQueue.main.async {
guard result?["ok"] as? Bool == true else {
isFinalizing = false
errorMessage = result?["error"] as? String ?? err ?? "Falha em \(operation)."
finalStatus = "Processamento interrompido."
return
}
let nextPath = result?["path"] as? String ?? currentPath
finalizeStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder)
}
}
}
}
+101 -2
View File
@@ -382,6 +382,10 @@ DEFAULT_VOICE_ANALYSIS_CONFIG: dict = {
"emphasis_floor": 0.25, "emphasis_floor": 0.25,
"emotion_enabled": False, "emotion_enabled": False,
"emotion_sensitivity": 0.5, "emotion_sensitivity": 0.5,
"zoom_scale": 1.30,
"zoom_mode": "in_out",
"zoom_ease_in": 0.25,
"zoom_ease_out": 0.04,
} }
@@ -402,12 +406,28 @@ def load_voice_analysis_config() -> dict:
stored = _load_config().get("voice_analysis") stored = _load_config().get("voice_analysis")
if not isinstance(stored, dict): if not isinstance(stored, dict):
return cfg return cfg
for key in ("energy_threshold", "peak_percentile", "emphasis_floor", "emotion_sensitivity"): for key in (
"energy_threshold", "peak_percentile", "emphasis_floor",
"emotion_sensitivity", "zoom_scale", "zoom_ease_in", "zoom_ease_out",
):
if key in stored: if key in stored:
try: try:
cfg[key] = max(0.0, min(1.0, float(stored[key]))) value = float(stored[key])
if key == "zoom_scale":
cfg[key] = max(1.0, min(3.0, value))
elif key.startswith("zoom_ease"):
cfg[key] = max(0.01, min(5.0, value))
else:
cfg[key] = max(0.0, min(1.0, value))
except (TypeError, ValueError): except (TypeError, ValueError):
pass pass
if "emphasis_threshold" in stored and "emphasis_floor" not in stored:
try:
cfg["emphasis_floor"] = max(0.0, min(1.0, float(stored["emphasis_threshold"])))
except (TypeError, ValueError):
pass
if stored.get("zoom_mode") in ("in_out", "in", "out"):
cfg["zoom_mode"] = stored["zoom_mode"]
if "emotion_enabled" in stored: if "emotion_enabled" in stored:
cfg["emotion_enabled"] = bool(stored["emotion_enabled"]) cfg["emotion_enabled"] = bool(stored["emotion_enabled"])
weights = stored.get("emphasis_weights") weights = stored.get("emphasis_weights")
@@ -428,6 +448,10 @@ def save_voice_analysis_config(
emphasis_floor: float | None = None, emphasis_floor: float | None = None,
emotion_enabled: bool | None = None, emotion_enabled: bool | None = None,
emotion_sensitivity: float | None = None, emotion_sensitivity: float | None = None,
zoom_scale: float | None = None,
zoom_mode: str | None = None,
zoom_ease_in: float | None = None,
zoom_ease_out: float | None = None,
) -> dict: ) -> dict:
"""Persist voice-analysis thresholds/weights. Only given fields change. """Persist voice-analysis thresholds/weights. Only given fields change.
@@ -446,6 +470,14 @@ def save_voice_analysis_config(
cfg["emotion_enabled"] = bool(emotion_enabled) cfg["emotion_enabled"] = bool(emotion_enabled)
if emotion_sensitivity is not None: if emotion_sensitivity is not None:
cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity))) cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity)))
if zoom_scale is not None:
cfg["zoom_scale"] = max(1.0, min(3.0, float(zoom_scale)))
if zoom_mode in ("in_out", "in", "out"):
cfg["zoom_mode"] = zoom_mode
if zoom_ease_in is not None:
cfg["zoom_ease_in"] = max(0.01, min(5.0, float(zoom_ease_in)))
if zoom_ease_out is not None:
cfg["zoom_ease_out"] = max(0.01, min(5.0, float(zoom_ease_out)))
if emphasis_weights is not None: if emphasis_weights is not None:
for key, value in emphasis_weights.items(): for key, value in emphasis_weights.items():
if key in cfg["emphasis_weights"] and value is not None: if key in cfg["emphasis_weights"] and value is not None:
@@ -536,6 +568,73 @@ def save_dynamic_subtitle_config(**fields) -> dict:
return cfg return cfg
DEFAULT_PLAIN_SUBTITLE_CONFIG: dict = {
"font": "Helvetica Neue",
"font_size": 82,
"font_color": "1 1 1 1",
"max_words": 7,
"position_y": -820.0,
"uppercase": False,
"keep_punctuation": True,
"text_scale": 2.0,
}
def load_plain_subtitle_config() -> dict:
"""Persisted style for simple editable FCPXML title subtitles."""
cfg = dict(DEFAULT_PLAIN_SUBTITLE_CONFIG)
stored = _load_config().get("plain_subtitles")
if not isinstance(stored, dict):
return cfg
for key in ("position_y", "text_scale"):
if key in stored:
try:
cfg[key] = float(stored[key])
except (TypeError, ValueError):
pass
for key in ("font_size", "max_words"):
if key in stored:
try:
cfg[key] = int(stored[key])
except (TypeError, ValueError):
pass
for key in ("font", "font_color"):
if key in stored and isinstance(stored[key], str) and stored[key]:
cfg[key] = stored[key]
for key in ("uppercase", "keep_punctuation"):
if key in stored:
cfg[key] = bool(stored[key])
cfg["max_words"] = max(1, int(cfg["max_words"]))
return cfg
def save_plain_subtitle_config(**fields) -> dict:
"""Persist simple subtitle style fields. Only given fields change."""
cfg = load_plain_subtitle_config()
for key, value in fields.items():
if key not in DEFAULT_PLAIN_SUBTITLE_CONFIG or value is None:
continue
if isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], bool):
cfg[key] = bool(value)
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], float):
try:
cfg[key] = float(value)
except (TypeError, ValueError):
continue
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], int):
try:
cfg[key] = int(value)
except (TypeError, ValueError):
continue
else:
cfg[key] = str(value)
cfg["max_words"] = max(1, int(cfg["max_words"]))
data = _load_config()
data["plain_subtitles"] = cfg
_write_config(data)
return cfg
# Mirrors the silence thresholds the detection/removal handlers use when no # Mirrors the silence thresholds the detection/removal handlers use when no
# argument is passed (server_tools/qc.py). Persisted so the app's slider and # argument is passed (server_tools/qc.py). Persisted so the app's slider and
# any later run agree without threading three fields through every call. # any later run agree without threading three fields through every call.
+547
View File
@@ -0,0 +1,547 @@
"""Phrase review — the human pass between the AI's decisions and the render.
A voice timeline says *how* every line was spoken; a list of voice actions says
what the model decided to do about it. Neither is reviewable on its own: the
timeline has no editorial intent, and the action list is a set of timecodes with
no text attached. This module joins them into the one view an editor can
actually judge — the script, phrase by phrase, each carrying the decision that
was made about it.
The phrase is the unit on purpose. Emphasis, in this pipeline, is not a property
of a word but of a line: an emphasized phrase gets a punch-in and a dynamic
caption, everything else gets a plain caption. Keeping the same granularity in
the review, the JSON, and the render means a toggle in the UI maps to exactly
one editorial outcome, with nothing to reconcile in between.
Trimming stays inside the phrase for the same reason. A line is rarely wrong as
a whole — it has a false start, or a trailing "né" — so each phrase carries a
``trim_start``/``trim_end`` pair that rides on word boundaries. Editing a cut
therefore means picking a word, never hunting for a frame, and a partial cut
from the model arrives as a trim instead of being rounded away.
Round-tripping is the other half of the contract. :func:`build_phrase_review`
derives the review from actions, :func:`phrase_review_to_actions` derives
actions back from the edited review, and everything the editor touched wins over
what was inferred — so re-opening the screen shows what was left there, not a
re-derivation that quietly discards the edits.
"""
import json
from pathlib import Path
from typing import Any, Dict, List, Optional, Sequence, Tuple
from .voice_actions import VoiceAction, merge_cut_ranges, parse_actions
PHRASE_REVIEW_VERSION = "1.0"
# Emphasis is stored 0-3 rather than as a float so the UI, the JSON and the
# render agree on the same discrete decision. The thresholds map the continuous
# `peak_emphasis` of the voice timeline onto those levels when the model gave no
# explicit direction for a phrase.
EMPHASIS_LEVELS = (0, 1, 2, 3)
EMPHASIS_THRESHOLDS = (0.25, 0.45, 0.65)
# Zoom scale applied per emphasis level when the review is turned back into
# actions. Level 0 never produces a zoom. The values stay inside
# voice_actions.MIN_ZOOM_SCALE..MAX_ZOOM_SCALE.
ZOOM_SCALE_BY_LEVEL = {1: 1.15, 2: 1.3, 3: 1.5}
# A phrase only survives if most of it does. Speech boundaries from a transcript
# are approximate, so a cut clipping a fraction of a second off the tail is a
# trim, not a removal — treating that as "phrase deleted" would grey out lines
# that are still fully audible.
CUT_COVERAGE_TO_DEACTIVATE = 0.6
# A punch-in shorter than this has no time to ramp in and back out — the writer
# rejects the window anyway (see the zoom ease-in/ease-out shape), so refusing
# it here turns a silent drop at render time into nothing being placed at all.
MIN_ZOOM_DURATION = 0.4
TRACK_SCRIPT = "roteiro"
TRACK_BACKSTAGE = "bastidor"
TRACKS = (TRACK_SCRIPT, TRACK_BACKSTAGE)
def resolve_source(
source: str, voice_timeline_path: str, extra_dirs: Sequence[str] = ()
) -> str:
"""The playable path for a timeline's ``source``, or "" when it's gone.
The voice timeline stores only the media's *file name* — it is written to be
read by a model, where a machine-specific absolute path is noise. That makes
it useless for opening a preview, so the file is looked up where it can
actually be: beside its own timeline JSON first (that is where
``analyze_voice`` writes it), then in whatever project folders the caller
knows about.
"""
if not source:
return ""
candidate = Path(source)
if candidate.is_absolute() and candidate.is_file():
return str(candidate)
directories = [Path(voice_timeline_path).parent] if voice_timeline_path else []
directories += [Path(d) for d in extra_dirs if d]
for directory in directories:
found = directory / candidate.name
if found.is_file():
return str(found)
return ""
def _overlap(a_start: float, a_end: float, b_start: float, b_end: float) -> float:
"""Seconds shared by two spans (0.0 when they don't touch)."""
return max(0.0, min(a_end, b_end) - max(a_start, b_start))
def _cut_coverage(
start: float, end: float, cuts: Sequence[Tuple[float, float]]
) -> float:
"""Fraction of ``start``-``end`` that falls inside ``cuts`` (0-1)."""
span = end - start
if span <= 0:
return 0.0
removed = sum(_overlap(start, end, c_start, c_end) for c_start, c_end in cuts)
return min(1.0, removed / span)
def snap_to_words(
time: float, words: Sequence[dict], fallback: float, edge: str
) -> float:
"""Move ``time`` onto the nearest word boundary of this phrase.
Trims are expressed by pointing at a word, so a trim handle that landed
mid-word would cut a syllable in half. ``edge`` is ``"in"`` (snap to word
starts) or ``"out"`` (snap to word ends); with no word timings available the
time is left as-is.
"""
boundaries = [
float(word.get("start" if edge == "in" else "end", 0.0)) for word in words
]
boundaries = [b for b in boundaries if b > 0]
if not boundaries:
return fallback
return min(boundaries, key=lambda b: abs(b - time))
def _trim_from_cuts(
start: float,
end: float,
words: Sequence[dict],
cuts: Sequence[Tuple[float, float]],
) -> Tuple[float, float]:
"""Read a partial cut over this phrase as a head/tail trim.
Only cuts that touch an edge become trims: a cut carved out of the middle of
a line has no representation here (the phrase is the unit), so it is left
for the whole-phrase coverage rule to decide.
"""
trim_start, trim_end = start, end
for cut_start, cut_end in cuts:
if _overlap(start, end, cut_start, cut_end) <= 0:
continue
if cut_start <= trim_start < cut_end < end:
trim_start = snap_to_words(cut_end, words, cut_end, "in")
if start < cut_start < trim_end <= cut_end:
trim_end = snap_to_words(cut_start, words, cut_start, "out")
if trim_end <= trim_start:
return start, end
return trim_start, trim_end
def _level_from_peak(peak: float) -> int:
"""Map a 0-1 ``peak_emphasis`` onto a 0-3 level."""
for level, threshold in enumerate(EMPHASIS_THRESHOLDS):
if peak < threshold:
return level
return 3
def _level_from_scale(scale: Optional[float]) -> int:
"""Map a zoom's scale factor back onto a 0-3 level.
The model is free to send any scale inside the allowed range, so this picks
the nearest level rather than requiring one of our own three values.
"""
if scale is None:
return 2
best = 1
smallest = None
for level, level_scale in ZOOM_SCALE_BY_LEVEL.items():
distance = abs(level_scale - float(scale))
if smallest is None or distance < smallest:
smallest, best = distance, level
return best
def _emphasis_from_actions(
start: float,
end: float,
actions: Sequence[VoiceAction],
) -> Tuple[Optional[int], str]:
"""The level the model asked for on this phrase, and why.
A ``zoom`` or ``text`` action anywhere inside the phrase is read as "this
line is the emphasis" — the model places them on the word that carries the
point, not on the whole line, so requiring a full-span match would find
nothing. Returns ``(None, "")`` when no action touches the phrase.
"""
level: Optional[int] = None
reason = ""
for action in actions:
if action.kind not in ("zoom", "text"):
continue
if _overlap(start, end, action.start, action.end) <= 0:
continue
if action.kind == "zoom":
candidate = _level_from_scale(action.params.get("scale"))
else:
candidate = 2
if level is None or candidate > level:
level = candidate
reason = action.reason
return level, reason
def _cut_reason(
start: float, end: float, actions: Sequence[VoiceAction]
) -> str:
"""The reason given for the cut that removes this phrase."""
for action in actions:
if action.kind != "cut":
continue
if _overlap(start, end, action.start, action.end) > 0 and action.reason:
return action.reason
return ""
def build_phrase_review(
timeline: dict,
actions: Any = None,
voice_timeline_path: str = "",
extra_dirs: Sequence[str] = (),
) -> dict:
"""Join a voice timeline with the AI's actions into a reviewable script.
``actions`` accepts whatever :func:`~.voice_actions.parse_actions` accepts —
a bare list, ``{"actions": [...]}``, or ``None`` when there is no AI pass and
the review starts from the acoustics alone. Malformed rows are skipped and
reported in ``errors`` rather than raising, matching the rest of the
decision pipeline.
"""
parsed, errors = parse_actions(actions) if actions else ([], [])
cuts = merge_cut_ranges(parsed)
phrases: List[dict] = []
for index, segment in enumerate(timeline.get("segments", [])):
start = float(segment.get("start", 0.0))
end = float(segment.get("end", 0.0))
peak = float(segment.get("peak_emphasis", 0.0))
take_boundary = bool(segment.get("take_boundary", False))
words = list(segment.get("words", []))
coverage = _cut_coverage(start, end, cuts)
active = coverage < CUT_COVERAGE_TO_DEACTIVATE
trim_start, trim_end = (
_trim_from_cuts(start, end, words, cuts) if active else (start, end)
)
asked_level, asked_reason = _emphasis_from_actions(start, end, parsed)
if asked_level is not None:
emphasis, reason = asked_level, asked_reason
else:
emphasis = _level_from_peak(peak)
reason = f"ênfase {peak:.2f}" if emphasis else ""
if not active:
# A removed line carries the reason it was removed; the emphasis it
# would have had is kept so re-activating it restores the decision.
reason = _cut_reason(start, end, parsed) or reason
phrases.append(
{
"index": index,
"start": round(start, 3),
"end": round(end, 3),
"trim_start": round(trim_start, 3),
"trim_end": round(trim_end, 3),
"text": str(segment.get("text", "")).strip(),
"speaker": str(segment.get("speaker", "")),
"active": active,
"emphasis": emphasis,
"track": TRACK_BACKSTAGE if (not active and take_boundary) else TRACK_SCRIPT,
"peak_emphasis": round(peak, 3),
# Delivery emotion is a heuristic over the acoustics (see
# voice_timeline._emotion_for_word) and only means anything when
# the analysis actually ran — `emotion_available` below is what
# separates "spoken flat" from "never measured".
"emotion": str(segment.get("emotion", "neutral")),
"emotion_confidence": round(
float(segment.get("emotion_confidence", 0.0)), 3
),
"take_boundary": take_boundary,
"gap_before": round(float(segment.get("gap_before", 0.0)), 3),
"reason": reason,
"words": [
{
"text": str(word.get("text", "")),
"start": round(float(word.get("start", 0.0)), 3),
"end": round(float(word.get("end", 0.0)), 3),
"energy": round(float(word.get("energy", 0.0)), 3),
"emphasis": round(float(word.get("emphasis", 0.0)), 3),
}
for word in words
],
}
)
source = timeline.get("source", "")
layers = timeline.get("layers", {}) if isinstance(timeline.get("layers"), dict) else {}
return {
"version": PHRASE_REVIEW_VERSION,
"source": source,
"source_path": resolve_source(source, voice_timeline_path, extra_dirs),
"duration": round(phrases[-1]["end"], 3) if phrases else 0.0,
"speakers": timeline.get("speakers", []),
"emotion_available": bool(layers.get("emotion", False)),
"phrases": phrases,
# Punch-ins the editor places by hand on an arbitrary range, alongside
# the whole-phrase zoom that an emphasis level produces. Both end up as
# zoom actions; this one exists because the moment worth punching into
# is not always a whole sentence.
"zooms": [],
"errors": errors,
}
def _coerce_zoom(raw: Any) -> Optional[Dict[str, float]]:
"""Normalize one manually placed zoom range."""
if not isinstance(raw, dict):
return None
try:
start = float(raw.get("start"))
end = float(raw.get("end"))
except (TypeError, ValueError):
return None
if end - start < MIN_ZOOM_DURATION:
return None
return {"start": start, "end": end}
def _coerce_phrase(raw: Any, index: int) -> Optional[Dict[str, Any]]:
"""Normalize one edited phrase row coming back from the UI."""
if not isinstance(raw, dict):
return None
try:
start = float(raw.get("start"))
end = float(raw.get("end"))
except (TypeError, ValueError):
return None
if end <= start:
return None
try:
emphasis = int(raw.get("emphasis", 0))
except (TypeError, ValueError):
emphasis = 0
try:
trim_start = float(raw.get("trim_start", start))
trim_end = float(raw.get("trim_end", end))
except (TypeError, ValueError):
trim_start, trim_end = start, end
# A trim that escaped the phrase, or inverted, is treated as no trim at all:
# the UI is the only thing that writes these, and silently discarding a bad
# pair keeps a rounding slip from deleting material the editor kept.
if not (start <= trim_start < trim_end <= end):
trim_start, trim_end = start, end
track = str(raw.get("track", TRACK_SCRIPT))
return {
"index": int(raw.get("index", index)),
"start": start,
"end": end,
"trim_start": trim_start,
"trim_end": trim_end,
"text": str(raw.get("text", "")).strip(),
"speaker": str(raw.get("speaker", "")),
"active": bool(raw.get("active", True)),
"emphasis": min(3, max(0, emphasis)),
"track": track if track in TRACKS else TRACK_SCRIPT,
"reason": str(raw.get("reason", "")),
}
def phrase_review_to_actions(review: dict) -> dict:
"""Turn an edited review back into the action list the applier consumes.
Every deactivated phrase becomes a ``cut``, a trimmed one becomes a cut over
the head and/or tail it lost, and every emphasized one becomes a ``zoom``
scaled by its level. The emphasis flags ride along in ``emphasis_spans`` so
the caption step can give those lines the dynamic treatment and everything
else the plain one, without re-deriving the decision from the acoustics.
"""
phrases = [
coerced
for index, raw in enumerate(review.get("phrases", []))
if (coerced := _coerce_phrase(raw, index)) is not None
]
actions: List[dict] = []
emphasis_spans: List[dict] = []
for phrase in phrases:
if not phrase["active"]:
actions.append(
VoiceAction(
kind="cut",
start=phrase["start"],
end=phrase["end"],
reason=phrase["reason"] or "desativada na revisão",
speaker=phrase["speaker"],
).as_dict()
)
continue
# Head and tail the editor trimmed off — each becomes its own cut, so a
# false start disappears without taking the line with it.
for trim_start, trim_end, where in (
(phrase["start"], phrase["trim_start"], "início"),
(phrase["trim_end"], phrase["end"], "fim"),
):
if trim_end - trim_start <= 0:
continue
actions.append(
VoiceAction(
kind="cut",
start=trim_start,
end=trim_end,
reason=f"trecho do {where} da frase removido na revisão",
speaker=phrase["speaker"],
).as_dict()
)
if phrase["emphasis"] >= 1:
actions.append(
VoiceAction(
kind="zoom",
start=phrase["trim_start"],
end=phrase["trim_end"],
params={"scale": ZOOM_SCALE_BY_LEVEL[phrase["emphasis"]]},
reason=phrase["reason"] or f"ênfase nível {phrase['emphasis']}",
speaker=phrase["speaker"],
).as_dict()
)
emphasis_spans.append(
{
"start": phrase["trim_start"],
"end": phrase["trim_end"],
"level": phrase["emphasis"],
"text": phrase["text"],
}
)
# Hand-placed punch-ins carry no scale on purpose: an omitted scale lets the
# applier use the shape configured in "Análise de Voz" (zoom_scale, ease in
# and out), so changing that setting restyles every manual zoom instead of
# leaving a scale frozen into each one at the moment it was drawn.
for raw in review.get("zooms", []):
zoom = _coerce_zoom(raw)
if zoom is None:
continue
actions.append(
VoiceAction(
kind="zoom",
start=zoom["start"],
end=zoom["end"],
reason="zoom marcado na revisão",
).as_dict()
)
return {
"source": review.get("source", ""),
"actions": actions,
"emphasis_spans": emphasis_spans,
}
def merge_saved_decisions(review: dict, saved: Optional[dict]) -> dict:
"""Lay a previously saved review's decisions over a freshly built one.
Only the editorial fields travel — active, emphasis, track, text, trims.
Everything else (words, emotion, energy) is re-derived from the current
analysis, so re-running the voice pass with better settings improves the
screen instead of being masked by a stale copy of itself, and the saved file
never has to carry a duplicate of data it does not own.
Phrases are matched by index *and* start time: if the analysis changed
enough to move a line, the old decision for that slot is dropped rather than
applied to a different sentence.
"""
if not saved:
return review
review["zooms"] = [
zoom for raw in saved.get("zooms", []) if (zoom := _coerce_zoom(raw)) is not None
]
by_index = {}
for raw in saved.get("phrases", []):
if isinstance(raw, dict) and "index" in raw:
by_index[raw["index"]] = raw
for phrase in review["phrases"]:
previous = by_index.get(phrase["index"])
if previous is None:
continue
if abs(float(previous.get("start", -1)) - phrase["start"]) > 0.25:
continue
phrase["active"] = bool(previous.get("active", phrase["active"]))
phrase["emphasis"] = min(3, max(0, int(previous.get("emphasis", phrase["emphasis"]))))
track = str(previous.get("track", phrase["track"]))
phrase["track"] = track if track in TRACKS else phrase["track"]
if previous.get("text"):
phrase["text"] = str(previous["text"])
trim_start = float(previous.get("trim_start", phrase["trim_start"]))
trim_end = float(previous.get("trim_end", phrase["trim_end"]))
if phrase["start"] <= trim_start < trim_end <= phrase["end"]:
phrase["trim_start"], phrase["trim_end"] = trim_start, trim_end
return review
def review_paths(voice_timeline_path: str) -> Tuple[Path, Path]:
"""Where the review and its derived actions live, next to the timeline.
Both files sit beside the ``_voice_timeline.json`` they came from and are
named after it, so a project folder stays readable and re-running the wizard
on the same take overwrites its own files instead of accumulating copies.
"""
base = Path(voice_timeline_path)
stem = base.stem
if stem.endswith("_voice_timeline"):
stem = stem[: -len("_voice_timeline")]
return (
base.with_name(f"{stem}_phrase_review.json"),
base.with_name(f"{stem}_phrase_actions.json"),
)
def save_phrase_review(voice_timeline_path: str, review: dict) -> Tuple[Path, Path]:
"""Write the edited review and the actions derived from it. Returns both paths."""
review_path, actions_path = review_paths(voice_timeline_path)
review_path.write_text(
json.dumps(review, ensure_ascii=False, indent=2), encoding="utf-8"
)
actions_path.write_text(
json.dumps(phrase_review_to_actions(review), ensure_ascii=False, indent=2),
encoding="utf-8",
)
return review_path, actions_path
def load_phrase_review(voice_timeline_path: str) -> Optional[dict]:
"""The review saved earlier for this timeline, or ``None`` if there is none."""
review_path, _ = review_paths(voice_timeline_path)
if not review_path.is_file():
return None
try:
data = json.loads(review_path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
return None
return data if isinstance(data, dict) else None
+4 -1
View File
@@ -28,8 +28,11 @@ ALLOWED_MODELS = (
) )
# Conservative by default: interjections that are near-universally filler. # Conservative by default: interjections that are near-universally filler.
# Portuguese "um"/"uma" are usually articles/numerals inside real phrases
# ("de um jeito") rather than discardable hesitations, so only cut them when
# the caller explicitly opts in through the fillers argument.
# "like" / "so" / "actually" are speech, not noise, unless the user opts in. # "like" / "so" / "actually" are speech, not noise, unless the user opts in.
DEFAULT_FILLERS = ("um", "uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm") DEFAULT_FILLERS = ("uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm")
_NORM_RE = re.compile(r"[^\w']+") _NORM_RE = re.compile(r"[^\w']+")
+24 -9
View File
@@ -82,21 +82,36 @@ def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]:
params = dict(params) if isinstance(params, dict) else {} params = dict(params) if isinstance(params, dict) else {}
if kind == "zoom": if kind == "zoom":
try: if "scale" in params and params.get("scale") is not None:
scale = float(params.get("scale", 1.3)) try:
except (TypeError, ValueError): scale = float(params["scale"])
return None, f"{where}: zoom scale must be a number" except (TypeError, ValueError):
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE): return None, f"{where}: zoom scale must be a number"
return None, ( if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}" return None, (
) f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}"
params["scale"] = scale )
params["scale"] = scale
if kind == "text": if kind == "text":
content = str(params.get("content", "")).strip() content = str(params.get("content", "")).strip()
if not content: if not content:
return None, f"{where}: text action needs params.content" return None, f"{where}: text action needs params.content"
params["content"] = content[:MAX_TEXT_LENGTH] params["content"] = content[:MAX_TEXT_LENGTH]
# Style is optional — omitted fields fall back to the "Legendas
# Dinâmicas" emphasis style at apply time (see _apply_placed_action),
# so a callout matches the captions' look without the caller having
# to know or repeat that configuration. Anything given here wins.
for key in ("font", "font_color", "face"):
if key in params and not isinstance(params[key], str):
del params[key]
if "font_size" in params:
try:
params["font_size"] = int(params["font_size"])
except (TypeError, ValueError):
del params["font_size"]
if "bold" in params:
params["bold"] = bool(params["bold"])
return ( return (
VoiceAction( VoiceAction(
+89
View File
@@ -56,12 +56,20 @@ VALUE_SCALES = {
"rate_delta": "0-1, how much the local speaking rate departs from the average", "rate_delta": "0-1, how much the local speaking rate departs from the average",
"pause_before": "seconds of silence immediately before the word", "pause_before": "seconds of silence immediately before the word",
"emphasis": "0-1 combined index; high values are punch-in/highlight candidates", "emphasis": "0-1 combined index; high values are punch-in/highlight candidates",
"emotion": "heuristic label from delivery: neutral, excited, tense, calm, reflective",
"emotion_confidence": "0-1 confidence in the heuristic emotion label",
"arousal": "0-1 vocal activation from energy/rate/pitch movement",
"valence": "0-1 rough positive tone; lower values suggest tension/weight",
}, },
"segment": { "segment": {
"gap_before": "seconds of silence before this line", "gap_before": "seconds of silence before this line",
"take_boundary": "true when the gap is long enough that the take likely restarted here", "take_boundary": "true when the gap is long enough that the take likely restarted here",
"avg_energy": "0-1 mean loudness across the line", "avg_energy": "0-1 mean loudness across the line",
"peak_emphasis": "0-1 highest emphasis of any word in the line", "peak_emphasis": "0-1 highest emphasis of any word in the line",
"emotion": "dominant delivery emotion across the line",
"emotion_confidence": "0-1 confidence in the dominant segment emotion",
"arousal": "0-1 mean vocal activation across the line",
"valence": "0-1 mean rough positive tone across the line",
}, },
} }
@@ -92,11 +100,74 @@ def _round_word(word: dict) -> dict:
"rate_delta": round(word.get("rate_delta", 0.0), 3), "rate_delta": round(word.get("rate_delta", 0.0), 3),
"pause_before": round(word.get("pause_before", 0.0), 3), "pause_before": round(word.get("pause_before", 0.0), 3),
"emphasis": round(word.get("emphasis", 0.0), 3), "emphasis": round(word.get("emphasis", 0.0), 3),
"emotion": word.get("emotion", "neutral"),
"emotion_confidence": round(word.get("emotion_confidence", 0.0), 3),
"arousal": round(word.get("arousal", 0.0), 3),
"valence": round(word.get("valence", 0.5), 3),
"energy_raw": word.get("energy"), "energy_raw": word.get("energy"),
"pitch_hz": word.get("pitch_hz"), "pitch_hz": word.get("pitch_hz"),
} }
def _emotion_for_word(word: dict, enabled: bool, sensitivity: float) -> dict:
"""Classify delivery emotion from normalized acoustic features.
This is deliberately a local heuristic rather than a claimed clinical
emotion model. It gives the editor a useful signal about delivery shape
while degrading predictably when acoustic extraction is unavailable.
"""
if not enabled:
return {
"emotion": "neutral",
"emotion_confidence": 0.0,
"arousal": 0.0,
"valence": 0.5,
}
energy = float(word.get("energy_norm", 0.0))
pitch = float(word.get("pitch_delta", 0.0))
rate = float(word.get("rate_delta", 0.0))
pause = min(float(word.get("pause_before", 0.0)) / 2.0, 1.0)
emphasis = float(word.get("emphasis", 0.0))
arousal = max(0.0, min(1.0, energy * 0.45 + pitch * 0.25 + rate * 0.20 + emphasis * 0.10))
valence = max(0.0, min(1.0, 0.55 + energy * 0.15 - pause * 0.20 - rate * 0.10))
if arousal >= 0.68 and valence >= 0.50:
label = "excited"
confidence = arousal
elif arousal >= 0.58 and valence < 0.50:
label = "tense"
confidence = max(arousal, 1.0 - valence)
elif arousal <= 0.28 and pause >= 0.25:
label = "reflective"
confidence = max(1.0 - arousal, pause)
elif arousal <= 0.35:
label = "calm"
confidence = 1.0 - arousal
else:
label = "neutral"
confidence = 1.0 - abs(arousal - 0.5) * 2.0
confidence = max(0.0, min(1.0, confidence))
if confidence < sensitivity:
label = "neutral"
return {
"emotion": label,
"emotion_confidence": confidence,
"arousal": arousal,
"valence": valence,
}
def annotate_emotions(words: Sequence[dict], enabled: bool, sensitivity: float) -> List[dict]:
"""Attach heuristic emotion labels to enriched word rows."""
return [
{**w, **_emotion_for_word(w, enabled, sensitivity)}
for w in words
]
def enrich_words( def enrich_words(
words: Sequence[dict], words: Sequence[dict],
pitch_track: Optional[Sequence] = None, pitch_track: Optional[Sequence] = None,
@@ -166,6 +237,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]
in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end] in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end]
energies = [w["energy_norm"] for w in in_seg] energies = [w["energy_norm"] for w in in_seg]
emphases = [w["emphasis"] for w in in_seg] emphases = [w["emphasis"] for w in in_seg]
arousals = [w.get("arousal", 0.0) for w in in_seg]
valences = [w.get("valence", 0.5) for w in in_seg]
emotions = [w.get("emotion", "neutral") for w in in_seg]
dominant = max(set(emotions), key=emotions.count) if emotions else "neutral"
emotion_confidences = [
w.get("emotion_confidence", 0.0) for w in in_seg if w.get("emotion") == dominant
]
gap = max(0.0, start - previous_end) gap = max(0.0, start - previous_end)
rows.append( rows.append(
{ {
@@ -181,6 +259,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]
"take_boundary": gap >= TAKE_BOUNDARY_GAP, "take_boundary": gap >= TAKE_BOUNDARY_GAP,
"avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0, "avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0,
"peak_emphasis": round(max(emphases), 3) if emphases else 0.0, "peak_emphasis": round(max(emphases), 3) if emphases else 0.0,
"emotion": dominant,
"emotion_confidence": (
round(sum(emotion_confidences) / len(emotion_confidences), 3)
if emotion_confidences else 0.0
),
"arousal": round(sum(arousals) / len(arousals), 3) if arousals else 0.0,
"valence": round(sum(valences) / len(valences), 3) if valences else 0.5,
"words": [_round_word(w) for w in in_seg], "words": [_round_word(w) for w in in_seg],
} }
) )
@@ -425,6 +510,8 @@ def build_voice_timeline(
weights: EmphasisWeights = EmphasisWeights(), weights: EmphasisWeights = EmphasisWeights(),
peak_percentile: float = 0.02, peak_percentile: float = 0.02,
emphasis_floor: float = 0.25, emphasis_floor: float = 0.25,
emotion_enabled: bool = False,
emotion_sensitivity: float = 0.5,
progress_cb: Optional[Callable[[float, str], None]] = None, progress_cb: Optional[Callable[[float, str], None]] = None,
) -> dict: ) -> dict:
"""Build the consolidated voice timeline for one media file. """Build the consolidated voice timeline for one media file.
@@ -445,6 +532,7 @@ def build_voice_timeline(
report(0.5, "Calculando ênfase...") report(0.5, "Calculando ênfase...")
words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights) words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights)
words = annotate_emotions(words, emotion_enabled, emotion_sensitivity)
report(0.7, "Identificando participantes...") report(0.7, "Identificando participantes...")
tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None
@@ -465,6 +553,7 @@ def build_voice_timeline(
"transcript": bool(transcript.get("words")), "transcript": bool(transcript.get("words")),
"acoustics": pitch_track is not None or energy_track is not None, "acoustics": pitch_track is not None or energy_track is not None,
"speakers": tracks is not None, "speakers": tracks is not None,
"emotion": bool(emotion_enabled),
}, },
"scales": VALUE_SCALES, "scales": VALUE_SCALES,
"summary": _summary( "summary": _summary(
+63 -2
View File
@@ -2218,13 +2218,23 @@ class FCPXMLModifier:
seg_start: 'TimeValue', seg_start: 'TimeValue',
seg_duration: 'TimeValue', seg_duration: 'TimeValue',
) -> None: ) -> None:
"""Remove markers/keywords from *clip* that fall outside the segment range. """Remove markers/keywords/titles from *clip* that fall outside the segment range.
After ``split_clip`` deepcopy's the original clip into each segment, every After ``split_clip`` deepcopy's the original clip into each segment, every
segment inherits all child elements. Markers whose ``start`` falls outside segment inherits all child elements. Markers whose ``start`` falls outside
``[seg_start, seg_start + seg_duration)`` are phantom duplicates and must be ``[seg_start, seg_start + seg_duration)`` are phantom duplicates and must be
removed. Keywords that partially overlap get their ``start``/``duration`` removed. Keywords that partially overlap get their ``start``/``duration``
clamped to the segment boundaries. clamped to the segment boundaries.
A lane-nested ``<title>`` (a "text" voice action's on-screen callout,
or a caption from an earlier `generate_dynamic_subtitles` pass) is
the same kind of phantom duplicate, just keyed on ``offset`` instead
of ``start`` — its offset lives in the same source-media coordinate
space as a marker's ``start`` (see ``add_text_title``/``add_marker``,
both anchored at ``parent.start``). Left unfiltered, every further
cut (silence removal, filler removal) duplicates it into every
resulting piece, so the same word shows up several times across the
edited timeline instead of once where it was placed.
""" """
seg_end = seg_start + seg_duration seg_end = seg_start + seg_duration
to_remove = [] to_remove = []
@@ -2234,6 +2244,10 @@ class FCPXMLModifier:
child_start = TimeValue.from_timecode(child.get('start', '0s')) child_start = TimeValue.from_timecode(child.get('start', '0s'))
if child_start < seg_start or child_start >= seg_end: if child_start < seg_start or child_start >= seg_end:
to_remove.append(child) to_remove.append(child)
elif tag == 'title':
title_offset = TimeValue.from_timecode(child.get('offset', '0s'))
if title_offset < seg_start or title_offset >= seg_end:
to_remove.append(child)
elif tag == 'keyword': elif tag == 'keyword':
kw_start = TimeValue.from_timecode(child.get('start', '0s')) kw_start = TimeValue.from_timecode(child.get('start', '0s'))
kw_dur = TimeValue.from_timecode(child.get('duration', '0s')) kw_dur = TimeValue.from_timecode(child.get('duration', '0s'))
@@ -2304,6 +2318,7 @@ class FCPXMLModifier:
self._filter_children_for_segment( self._filter_children_for_segment(
new_clip, current_start, segment_duration new_clip, current_start, segment_duration
) )
self._reassign_text_style_ids(new_clip)
spine.insert(clip_index + len(new_clips), new_clip) spine.insert(clip_index + len(new_clips), new_clip)
new_clips.append(new_clip) new_clips.append(new_clip)
@@ -2404,6 +2419,7 @@ class FCPXMLModifier:
new_clip.set('start', seg_start.to_fcpxml()) new_clip.set('start', seg_start.to_fcpxml())
new_clip.set('duration', seg_duration.to_fcpxml()) new_clip.set('duration', seg_duration.to_fcpxml())
self._filter_children_for_segment(new_clip, seg_start, seg_duration) self._filter_children_for_segment(new_clip, seg_start, seg_duration)
self._reassign_text_style_ids(new_clip)
spine.insert(clip_index + len(new_clips), new_clip) spine.insert(clip_index + len(new_clips), new_clip)
new_clips.append(new_clip) new_clips.append(new_clip)
current_offset = current_offset + seg_duration current_offset = current_offset + seg_duration
@@ -2946,6 +2962,7 @@ class FCPXMLModifier:
('-469658744/1000000000s', '0'), ('-469658744/1000000000s', '0'),
('12328542033/1000000000s', '1'), ('12328542033/1000000000s', '1'),
) )
_TEXT_SIZE_KEY = '9999/10003/13260/3296672360/5/3296672362/3'
def _ensure_text_title_effect(self, resources: ET.Element) -> str: def _ensure_text_title_effect(self, resources: ET.Element) -> str:
"""Return the resource id of the "Text" (Basic Text) effect, creating it if absent.""" """Return the resource id of the "Text" (Basic Text) effect, creating it if absent."""
@@ -2995,6 +3012,34 @@ class FCPXMLModifier:
self._text_style_ids.add(candidate) self._text_style_ids.add(candidate)
return candidate return candidate
def _reassign_text_style_ids(self, clip: ET.Element) -> None:
"""Give every ``<text-style-def>`` inside a just-deepcopy'd *clip* a
fresh document-unique id, repointing any ``<text-style ref="...">``
in the same subtree that pointed at the old one.
``split_clip``/``cut_clip_ranges`` deepcopy the clip once per
resulting segment, so a clip carrying a ``<title>`` (from a "text"
voice action) keeps the exact same ``text-style-def id`` in every
copy. A single cut is harmless — but the batch chain re-cuts the
same clip at each step (silence removal, filler removal, dynamic
subtitles), and every pass multiplies the duplicate, so the DTD
validator eventually rejects the file with "ID ... already
defined". Regenerating here, at the only place copies are made,
fixes it for every caller instead of each one having to remember to.
"""
for style_def in clip.findall('.//text-style-def'):
old_id = style_def.get('id')
if not old_id:
continue
slug = old_id[3:] if old_id.startswith('ts_') else old_id
slug = re.sub(r'_\d+$', '', slug) # drop a prior _<N> counter
new_id = self._unique_text_style_id(slug)
if new_id == old_id:
continue
style_def.set('id', new_id)
for ref_el in clip.findall(f".//text-style[@ref='{old_id}']"):
ref_el.set('ref', new_id)
def _make_text_title_clip( def _make_text_title_clip(
self, self,
effect_id: str, effect_id: str,
@@ -3012,6 +3057,8 @@ class FCPXMLModifier:
face: Optional[str] = None, face: Optional[str] = None,
kerning: Optional[float] = None, kerning: Optional[float] = None,
font_scale: float = TEXT_TEMPLATE_FONT_SCALE, font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
animated: bool = True,
size_param: Optional[float] = None,
) -> ET.Element: ) -> ET.Element:
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template. """Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
@@ -3042,9 +3089,12 @@ class FCPXMLModifier:
param.set('key', key) param.set('key', key)
param.set('value', value) param.set('value', value)
animation_params = {'Opacity', 'Speed', 'Apply Speed'}
for param_name, param_key, param_value in self._TEXT_TITLE_PARAMS: for param_name, param_key, param_value in self._TEXT_TITLE_PARAMS:
if not animated and param_name in animation_params:
continue
_add_param(param_name, param_key, param_value) _add_param(param_name, param_key, param_value)
if param_name == 'Speed': if animated and param_name == 'Speed':
# "Custom Speed" lands between "Speed" and "Apply Speed" and # "Custom Speed" lands between "Speed" and "Apply Speed" and
# carries a <keyframeAnimation> child instead of a value. # carries a <keyframeAnimation> child instead of a value.
cs = ET.SubElement(elem, 'param') cs = ET.SubElement(elem, 'param')
@@ -3056,6 +3106,9 @@ class FCPXMLModifier:
kf.set('time', kf_time) kf.set('time', kf_time)
kf.set('value', kf_value) kf.set('value', kf_value)
if size_param is not None:
_add_param('Size', self._TEXT_SIZE_KEY, f"{float(size_param):g}")
text_el = ET.SubElement(elem, 'text') text_el = ET.SubElement(elem, 'text')
ts_id = self._unique_text_style_id(name) ts_id = self._unique_text_style_id(name)
run = ET.SubElement(text_el, 'text-style') run = ET.SubElement(text_el, 'text-style')
@@ -3109,6 +3162,10 @@ class FCPXMLModifier:
font_size: int = 196, font_size: int = 196,
font_color: str = '1 1 1 1', font_color: str = '1 1 1 1',
bold: bool = True, bold: bool = True,
face: Optional[str] = None,
animated: bool = True,
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
size_param: Optional[float] = None,
) -> ET.Element: ) -> ET.Element:
"""Add a single static "Text" (Basic Text) title over *parent_clip*. """Add a single static "Text" (Basic Text) title over *parent_clip*.
@@ -3142,6 +3199,10 @@ class FCPXMLModifier:
font_size=font_size, font_size=font_size,
font_color=font_color, font_color=font_color,
bold=bold, bold=bold,
face=face,
animated=animated,
font_scale=font_scale,
size_param=size_param,
) )
_dtd_insert(parent, title) _dtd_insert(parent, title)
return title return title
+2
View File
@@ -154,6 +154,7 @@ from server_tools.roles import (
) )
from server_tools.subtitles import ( from server_tools.subtitles import (
handle_generate_dynamic_subtitles, handle_generate_dynamic_subtitles,
handle_generate_plain_subtitles,
handle_validate_subtitle_layout, handle_validate_subtitle_layout,
) )
from server_tools.timeline import ( from server_tools.timeline import (
@@ -320,6 +321,7 @@ __all__ = [
"handle_save_voice_analysis_config", "handle_save_voice_analysis_config",
"handle_validate_subtitle_layout", "handle_validate_subtitle_layout",
"handle_generate_dynamic_subtitles", "handle_generate_dynamic_subtitles",
"handle_generate_plain_subtitles",
"handle_push_to_fcp", "handle_push_to_fcp",
"handle_list_fcp_libraries", "handle_list_fcp_libraries",
] ]
+60 -7
View File
@@ -15,6 +15,7 @@ from typing import Any, Sequence
from mcp.types import TextContent from mcp.types import TextContent
from fcpxml.media_intel import media_src_to_path from fcpxml.media_intel import media_src_to_path
from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config
from fcpxml.models import ( from fcpxml.models import (
DuplicateGroup, DuplicateGroup,
FlashFrame, FlashFrame,
@@ -25,6 +26,7 @@ from fcpxml.models import (
) )
from fcpxml.parser import FCPXMLParser from fcpxml.parser import FCPXMLParser
from fcpxml.rough_cut import RoughCutGenerator from fcpxml.rough_cut import RoughCutGenerator
from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text
from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe
from fcpxml.writer import FCPXMLModifier from fcpxml.writer import FCPXMLModifier
@@ -639,28 +641,79 @@ def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str:
rel_end = action.end - clip_start rel_end = action.end - clip_start
if action.kind == "zoom": if action.kind == "zoom":
config = load_voice_analysis_config()
# Only forward an explicit ease — otherwise add_zoom's own default # Only forward an explicit ease — otherwise add_zoom's own default
# (a fast ramp in, instant snap back out) is what should apply. # (a fast ramp in, instant snap back out) is what should apply.
zoom_args = {} zoom_args = {
if action.params.get("ease") is not None: "ease": float(action.params.get("ease", config["zoom_ease_in"])),
zoom_args["ease"] = float(action.params["ease"]) "ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])),
if action.params.get("ease_out") is not None: }
zoom_args["ease_out"] = float(action.params["ease_out"]) mode = str(action.params.get("mode", config["zoom_mode"]))
if mode == "in":
zoom_args["hold_at_end"] = True
zoom_args["start_at_peak"] = False
elif mode == "out":
zoom_args["hold_at_end"] = False
zoom_args["start_at_peak"] = True
elif mode == "in_out":
zoom_args["hold_at_end"] = False
zoom_args["start_at_peak"] = False
modifier.add_zoom( modifier.add_zoom(
clip_id=clip_el, clip_id=clip_el,
start=rel_start, start=rel_start,
end=rel_end, end=rel_end,
scale=float(action.params.get("scale", 1.3)), scale=float(action.params.get("scale", config["zoom_scale"])),
**zoom_args, **zoom_args,
) )
return f"zoom {action.params.get('scale', 1.3):.2f}x" return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x"
if action.kind == "text": if action.kind == "text":
# Default to the "Legendas Dinâmicas" emphasis style (the font used
# to highlight a word in the captions) rather than a hardcoded
# Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of
# the video's on-screen text instead of looking like a stray default
# title. Any of these the action itself specifies still wins.
subtitle_cfg = load_dynamic_subtitle_config()
font = action.params.get("font", subtitle_cfg["emphasis_font"])
face = action.params.get("face", subtitle_cfg["emphasis_face"])
font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0)
requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"]))
requested_kerning = float(action.params.get("kerning", 0.0) or 0.0)
# Voice-action callouts are not part of the dynamic subtitle block.
# When omitted, put them above the subtitle band and shrink wide
# phrases to the title-safe width. The previous default (Position 0 0,
# full emphasis size) made long callouts like "PRÓTESES DE SILICONE"
# collide with captions and run off both sides of a vertical frame.
emitted_size = requested_size * font_scale
emitted_kerning = requested_kerning * font_scale
safe_width = modifier.frame_width() * 0.90
width = measure_text(
action.params["content"],
emitted_size,
bold=bool(action.params.get("bold", False)),
kerning=emitted_kerning,
font=font,
face=face,
)
font_size = requested_size
if width > safe_width and width > 0:
font_size = max(32, int(requested_size * safe_width / width))
position = action.params.get("position")
if not position:
position = f"0 {modifier.frame_height() * 0.23:g}"
modifier.add_text_title( modifier.add_text_title(
clip_el, clip_el,
action.params["content"], action.params["content"],
offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(), offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(), duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(),
position=position,
font=font,
font_size=font_size,
font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]),
face=face,
bold=action.params.get("bold", False),
) )
return f"text \"{action.params['content'][:24]}\"" return f"text \"{action.params['content'][:24]}\""
+176 -10
View File
@@ -6,13 +6,14 @@ Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalo
from __future__ import annotations from __future__ import annotations
import json import json
import re
from pathlib import Path from pathlib import Path
from typing import Sequence from typing import Sequence
from mcp.types import TextContent, Tool from mcp.types import TextContent, Tool
from fcpxml.media_intel import media_src_to_path from fcpxml.media_intel import media_src_to_path
from fcpxml.model_manager import load_dynamic_subtitle_config from fcpxml.model_manager import load_dynamic_subtitle_config, load_plain_subtitle_config
from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle from fcpxml.models import DynamicSubtitleConfig, WordLook, WordStyle
from fcpxml.writer import FCPXMLModifier from fcpxml.writer import FCPXMLModifier
from server_tools._shared import ( from server_tools._shared import (
@@ -69,9 +70,76 @@ TOOLS = [
"required": ["filepath"] "required": ["filepath"]
} }
), ),
Tool(
name="generate_plain_subtitles",
description="Generate simple editable FCPXML text-title subtitles, synchronized to transcript words but without visual build-in/build-out effects. Words are grouped into short blocks, placed at a configurable vertical position, and written as static Text titles rather than SRT captions.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"language": {"type": "string", "description": "ISO language code hint (e.g. 'pt'); auto-detected if omitted"},
"font": {"type": "string", "description": "Text font family. Falls back to saved plain-subtitle config."},
"font_size": {"type": "integer", "description": "Font size in canvas points. Falls back to saved plain-subtitle config."},
"font_color": {"type": "string", "description": "RGBA (0-1, space-separated). Falls back to saved plain-subtitle config."},
"max_words": {"type": "integer", "description": "Maximum words per subtitle block. Falls back to saved plain-subtitle config."},
"position_y": {"type": "number", "description": "Vertical title position in canvas points; negative sits lower in frame."},
"uppercase": {"type": "boolean", "description": "Render text in uppercase."},
"keep_punctuation": {"type": "boolean", "description": "Keep punctuation such as comma and period."},
"text_scale": {"type": "number", "description": "Template font-size scale. Falls back to saved plain-subtitle config."},
"output_path": {"type": "string", "description": "Output path (default: adds _plain_subtitles suffix)"},
},
"required": ["filepath"]
}
),
] ]
_PUNCT_RE = re.compile(r"[^\w\sÀ-ÖØ-öø-ÿ]", re.UNICODE)
def _words_overlapping_clip(words: Sequence[dict], start: float, end: float) -> list[dict]:
"""Return transcript words that overlap a source window, rebased to it."""
clip_words: list[dict] = []
for w in words:
word_start = float(w.get("start", 0.0))
word_end = float(w.get("end", word_start))
if word_end <= start or word_start >= end:
continue
clip_words.append(
{
"word": w.get("word", ""),
"start": max(0.0, word_start - start),
"end": max(0.0, min(word_end, end) - start),
}
)
return clip_words
def _plain_word_text(word: str, *, uppercase: bool, keep_punctuation: bool) -> str:
text = str(word or "").strip()
if not keep_punctuation:
text = _PUNCT_RE.sub("", text)
text = re.sub(r"\s+", " ", text).strip()
return text.upper() if uppercase else text
def _plain_subtitle_blocks(words: Sequence[dict], max_words: int) -> list[list[dict]]:
blocks: list[list[dict]] = []
pending: list[dict] = []
for word in words:
if not str(word.get("word", "")).strip():
continue
pending.append(word)
if len(pending) >= max(1, max_words):
blocks.append(pending)
pending = []
if pending:
blocks.append(pending)
return blocks
async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]: async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]:
"""Validate title/subtitle layout for spatial collisions and safe-area """Validate title/subtitle layout for spatial collisions and safe-area
containment (collision.validate_titles over every <title> in the file).""" containment (collision.validate_titles over every <title> in the file)."""
@@ -210,15 +278,7 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
window_end = clip_source_start + clip_duration window_end = clip_source_start + clip_duration
clip_words = [ clip_words = _words_overlapping_clip(data.get("words", []), clip_source_start, window_end)
{
"word": w.get("word", ""),
"start": float(w.get("start", 0.0)) - clip_source_start,
"end": float(w.get("end", 0.0)) - clip_source_start,
}
for w in data.get("words", [])
if clip_source_start <= float(w.get("start", 0.0)) < window_end
]
if not clip_words: if not clip_words:
skipped.append((name, "no words in clip's source range")) skipped.append((name, "no words in clip's source range"))
continue continue
@@ -277,7 +337,113 @@ async def handle_generate_dynamic_subtitles(arguments: dict) -> Sequence[TextCon
return _text_result(result) return _text_result(result)
async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextContent]:
"""Generate static, editable title subtitles from word-level transcripts."""
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
clip_filter = arguments.get("clip_name")
saved = load_plain_subtitle_config()
font = arguments.get("font") or saved["font"]
font_size = int(arguments.get("font_size", saved["font_size"]))
font_color = arguments.get("font_color") or saved["font_color"]
max_words = max(1, int(arguments.get("max_words", saved["max_words"])))
position_y = float(arguments.get("position_y", saved["position_y"]))
uppercase = bool(arguments.get("uppercase", saved["uppercase"]))
keep_punctuation = bool(arguments.get("keep_punctuation", saved["keep_punctuation"]))
filepath, output_path, modifier = _setup_modifier(arguments, "_plain_subtitles")
added: list[tuple[str, int, int]] = []
skipped: list[tuple[str, str]] = []
spine_clips = [el for _, el in modifier._iter_spine_clips()]
for el in spine_clips:
name = el.get("name", "")
if clip_filter and name != clip_filter:
continue
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
media_path = media_src_to_path(src)
if not media_path or not Path(media_path).is_file():
skipped.append((name, "media file missing"))
continue
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
if data is None:
skipped.append((name, reason))
continue
clip_source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
clip_words = _words_overlapping_clip(
data.get("words", []), clip_source_start, clip_source_start + clip_duration
)
if not clip_words:
skipped.append((name, "no words in clip's source range"))
continue
blocks = _plain_subtitle_blocks(clip_words, max_words)
created = 0
for block in blocks:
parts = [
_plain_word_text(w.get("word", ""), uppercase=uppercase, keep_punctuation=keep_punctuation)
for w in block
]
text = " ".join(p for p in parts if p).strip()
if not text:
continue
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
end = max(float(w.get("end", start)) for w in block)
duration = max(end - start, modifier.frame_duration_fraction())
modifier.add_text_title(
el,
text,
offset=f"{start:.6f}s",
duration=f"{duration:.6f}s",
lane=20,
position=f"0 {position_y:g}",
font=font,
font_size=font_size,
font_color=font_color,
bold=True,
face=None,
font_scale=1.0,
size_param=font_size,
)
created += 1
if created:
added.append((name, created, len(clip_words)))
if not added:
text = "# Plain Subtitles\n\nNo subtitles generated — file unchanged (nothing saved)."
if skipped:
text += "\n\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
return _text_result(text)
modifier.save(output_path)
total_titles = sum(lines for _, lines, _ in added)
total_words = sum(words for _, _, words in added)
result = "# Plain Subtitles Generated\n\n## Summary\n"
result += (
f"- **Clips Captioned**: {len(added)}\n"
f"- **Title Clips**: {total_titles}\n"
f"- **Total Words**: {total_words}\n\n"
)
result += _markdown_table(
["Clip", "Title Clips", "Words"],
[[n, str(lines), str(words)] for n, lines, words in added],
)
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json.*"
return _text_result(result)
HANDLERS = { HANDLERS = {
"validate_subtitle_layout": handle_validate_subtitle_layout, "validate_subtitle_layout": handle_validate_subtitle_layout,
"generate_dynamic_subtitles": handle_generate_dynamic_subtitles, "generate_dynamic_subtitles": handle_generate_dynamic_subtitles,
"generate_plain_subtitles": handle_generate_plain_subtitles,
} }
+2 -2
View File
@@ -67,12 +67,12 @@ TOOLS = [
), ),
Tool( Tool(
name="remove_filler_words", name="remove_filler_words",
description="Cut filler words (um, uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.", description="Cut filler interjections (uh, erm...) out of the timeline with ripple, using word-level transcripts of the real source audio. Conservative default filler list — words like 'um', 'uma', 'like' and 'so' are only cut if you pass them explicitly. Uses each media file's _transcript.json (auto-transcribes if missing). Non-destructive: writes a _defillered copy.",
inputSchema={ inputSchema={
"type": "object", "type": "object",
"properties": { "properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"}, "filepath": {"type": "string", "description": "Path to FCPXML file"},
"fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: um, uh, uhh, umm, erm, ehm, mmm, hmm, mhm)"}, "fillers": {"type": "array", "items": {"type": "string"}, "description": "Filler words/phrases to cut (default: uh, uhh, umm, erm, ehm, mmm, hmm, mhm; pass um/uma explicitly if desired)"},
"clip_name": {"type": "string", "description": "Only clean the clip with this name"}, "clip_name": {"type": "string", "description": "Only clean the clip with this name"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"}, "model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"padding": {"type": "number", "default": 0.02, "description": "Seconds to widen each cut on both sides (0-2, default 0.02)"}, "padding": {"type": "number", "default": 0.02, "description": "Seconds to widen each cut on both sides (0-2, default 0.02)"},
+5 -2
View File
@@ -348,8 +348,9 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
language = arguments.get("language") language = arguments.get("language")
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers() num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
output_dir = arguments.get("output_dir")
transcript, reason = _load_or_transcribe(media_path, model, language) transcript, reason = _load_or_transcribe(media_path, model, language, output_dir)
if transcript is None: if transcript is None:
return _text_result( return _text_result(
f"# Voice Timeline\n\nCould not obtain a transcript " f"# Voice Timeline\n\nCould not obtain a transcript "
@@ -365,9 +366,10 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
weights=EmphasisWeights.from_dict(config["emphasis_weights"]), weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
peak_percentile=config["peak_percentile"], peak_percentile=config["peak_percentile"],
emphasis_floor=config["emphasis_floor"], emphasis_floor=config["emphasis_floor"],
emotion_enabled=config["emotion_enabled"],
emotion_sensitivity=config["emotion_sensitivity"],
) )
output_dir = arguments.get("output_dir")
json_path = Path(_validate_output_path( json_path = Path(_validate_output_path(
str(voice_timeline_path(media_path, output_dir)), str(voice_timeline_path(media_path, output_dir)),
anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent), anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent),
@@ -398,6 +400,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
"yes" if layers["acoustics"] else "FAILED — every acoustic value is 0", "yes" if layers["acoustics"] else "FAILED — every acoustic value is 0",
], ],
["Speakers", "yes" if layers["speakers"] else "not run — single default speaker"], ["Speakers", "yes" if layers["speakers"] else "not run — single default speaker"],
["Emotion", "yes" if layers.get("emotion") else "not run"],
], ],
) + "\n" ) + "\n"
+16
View File
@@ -29,6 +29,7 @@ from fcpxml.text_layout import (
ink_extent, ink_extent,
) )
from fcpxml.writer import FCPXMLModifier from fcpxml.writer import FCPXMLModifier
from server_tools.subtitles import _words_overlapping_clip
SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml" SAMPLE = Path(__file__).parent.parent / "examples" / "sample.fcpxml"
def font_points(style) -> float: def font_points(style) -> float:
@@ -51,6 +52,21 @@ WORDS = [
] ]
def test_words_overlapping_clip_keeps_word_that_starts_just_before_in_point():
words = [
{"word": "Aquela", "start": 2.03, "end": 2.69},
{"word": "mama", "start": 2.69, "end": 2.89},
{"word": "fora", "start": 10.0, "end": 10.2},
]
clip_words = _words_overlapping_clip(words, 2.0437166666666666, 3.0)
assert clip_words == [
{"word": "Aquela", "start": 0.0, "end": pytest.approx(0.6462833333333332)},
{"word": "mama", "start": pytest.approx(0.6462833333333332), "end": pytest.approx(0.8462833333333334)},
]
@pytest.fixture @pytest.fixture
def temp_fcpxml(): def temp_fcpxml():
with tempfile.NamedTemporaryFile(suffix=".fcpxml", delete=False) as f: with tempfile.NamedTemporaryFile(suffix=".fcpxml", delete=False) as f:
+479
View File
@@ -0,0 +1,479 @@
"""Tests for the phrase review model (voice timeline + AI actions → editable script)."""
import json
import pytest
from fcpxml.phrase_review import (
TRACK_BACKSTAGE,
TRACK_SCRIPT,
ZOOM_SCALE_BY_LEVEL,
build_phrase_review,
load_phrase_review,
merge_saved_decisions,
phrase_review_to_actions,
resolve_source,
review_paths,
save_phrase_review,
snap_to_words,
)
def _words(spans, emphasis=0.0):
return [
{
"text": f"w{i}",
"start": start,
"end": end,
"energy": 0.5,
"emphasis": emphasis,
}
for i, (start, end) in enumerate(spans)
]
def _timeline(segments):
return {"source": "/tmp/take.mov", "speakers": ["SPEAKER_00"], "segments": segments}
def _segment(start, end, text="linha", peak=0.1, take_boundary=False, words=None):
return {
"start": start,
"end": end,
"text": text,
"speaker": "SPEAKER_00",
"peak_emphasis": peak,
"take_boundary": take_boundary,
"gap_before": 0.0,
"words": words if words is not None else _words([(start, end)]),
}
class TestBuildFromAcoustics:
def test_emphasis_levels_follow_peak_thresholds(self):
review = build_phrase_review(
_timeline(
[
_segment(0, 1, peak=0.10),
_segment(1, 2, peak=0.30),
_segment(2, 3, peak=0.50),
_segment(3, 4, peak=0.90),
]
)
)
assert [p["emphasis"] for p in review["phrases"]] == [0, 1, 2, 3]
def test_every_phrase_starts_active_without_actions(self):
review = build_phrase_review(_timeline([_segment(0, 1), _segment(1, 2)]))
assert all(p["active"] for p in review["phrases"])
assert all(p["track"] == TRACK_SCRIPT for p in review["phrases"])
def test_carries_text_speaker_and_words(self):
review = build_phrase_review(_timeline([_segment(0, 2, text=" olá ")]))
phrase = review["phrases"][0]
assert phrase["text"] == "olá"
assert phrase["speaker"] == "SPEAKER_00"
assert phrase["words"][0]["text"] == "w0"
assert review["duration"] == 2.0
class TestCutsDeactivate:
def test_fully_cut_phrase_is_inactive(self):
review = build_phrase_review(
_timeline([_segment(0, 2), _segment(2, 4)]),
{"actions": [{"kind": "cut", "start": 0, "end": 2, "reason": "gaguejou"}]},
)
assert review["phrases"][0]["active"] is False
assert review["phrases"][0]["reason"] == "gaguejou"
assert review["phrases"][1]["active"] is True
def test_small_overlap_keeps_the_phrase(self):
# 0.2s off a 2s line is a trim, not a removal.
review = build_phrase_review(
_timeline([_segment(1, 3, words=_words([(1, 1.2), (1.2, 3)]))]),
{"actions": [{"kind": "cut", "start": 0.5, "end": 1.2}]},
)
assert review["phrases"][0]["active"] is True
def test_majority_overlap_deactivates(self):
review = build_phrase_review(
_timeline([_segment(0, 2)]),
{"actions": [{"kind": "cut", "start": 0, "end": 1.5}]},
)
assert review["phrases"][0]["active"] is False
def test_inactive_after_take_boundary_is_backstage(self):
review = build_phrase_review(
_timeline([_segment(10, 12, take_boundary=True)]),
{"actions": [{"kind": "cut", "start": 10, "end": 12}]},
)
assert review["phrases"][0]["track"] == TRACK_BACKSTAGE
class TestTrimFromPartialCuts:
def test_head_cut_becomes_a_trim_snapped_to_a_word(self):
review = build_phrase_review(
_timeline([_segment(1, 4, words=_words([(1, 1.4), (1.4, 4)]))]),
{"actions": [{"kind": "cut", "start": 0.8, "end": 1.35}]},
)
phrase = review["phrases"][0]
assert phrase["active"] is True
assert phrase["trim_start"] == 1.4 # snapped to the second word's start
assert phrase["trim_end"] == 4.0
def test_tail_cut_becomes_a_trim(self):
review = build_phrase_review(
_timeline([_segment(0, 3, words=_words([(0, 2.5), (2.5, 3)]))]),
{"actions": [{"kind": "cut", "start": 2.6, "end": 3.5}]},
)
phrase = review["phrases"][0]
assert phrase["trim_start"] == 0.0
assert phrase["trim_end"] == 2.5
def test_untouched_phrase_trims_to_its_own_bounds(self):
review = build_phrase_review(_timeline([_segment(0, 2)]))
phrase = review["phrases"][0]
assert (phrase["trim_start"], phrase["trim_end"]) == (0.0, 2.0)
class TestAIDirectionWins:
def test_zoom_action_sets_the_level_over_the_heuristic(self):
review = build_phrase_review(
_timeline([_segment(0, 2, peak=0.05)]),
{
"actions": [
{
"kind": "zoom",
"start": 0.5,
"end": 0.9,
"params": {"scale": 1.5},
"reason": "virada da história",
}
]
},
)
phrase = review["phrases"][0]
assert phrase["emphasis"] == 3
assert phrase["reason"] == "virada da história"
def test_text_action_marks_emphasis(self):
review = build_phrase_review(
_timeline([_segment(0, 2, peak=0.0)]),
{
"actions": [
{
"kind": "text",
"start": 0.5,
"end": 1.0,
"params": {"content": "3x mais rápido"},
}
]
},
)
assert review["phrases"][0]["emphasis"] == 2
def test_highest_level_wins_when_several_actions_overlap(self):
review = build_phrase_review(
_timeline([_segment(0, 4)]),
{
"actions": [
{"kind": "zoom", "start": 0.2, "end": 0.5, "params": {"scale": 1.15}},
{"kind": "zoom", "start": 2.0, "end": 2.4, "params": {"scale": 1.5}},
]
},
)
assert review["phrases"][0]["emphasis"] == 3
def test_malformed_rows_are_reported_not_fatal(self):
review = build_phrase_review(
_timeline([_segment(0, 2)]),
{"actions": [{"kind": "voar", "start": 0, "end": 1}]},
)
assert len(review["errors"]) == 1
assert review["phrases"][0]["active"] is True
class TestBackToActions:
def test_inactive_phrase_becomes_a_cut(self):
review = build_phrase_review(_timeline([_segment(0, 2), _segment(2, 4)]))
review["phrases"][0]["active"] = False
result = phrase_review_to_actions(review)
cuts = [a for a in result["actions"] if a["kind"] == "cut"]
assert len(cuts) == 1
assert (cuts[0]["start"], cuts[0]["end"]) == (0.0, 2.0)
def test_emphasis_becomes_a_zoom_and_a_span(self):
review = build_phrase_review(_timeline([_segment(0, 2)]))
review["phrases"][0]["emphasis"] = 2
result = phrase_review_to_actions(review)
zooms = [a for a in result["actions"] if a["kind"] == "zoom"]
assert zooms[0]["params"]["scale"] == ZOOM_SCALE_BY_LEVEL[2]
assert result["emphasis_spans"] == [
{"start": 0.0, "end": 2.0, "level": 2, "text": "linha"}
]
def test_level_zero_produces_nothing(self):
review = build_phrase_review(_timeline([_segment(0, 2)]))
review["phrases"][0]["emphasis"] = 0
result = phrase_review_to_actions(review)
assert result["actions"] == []
assert result["emphasis_spans"] == []
def test_trim_becomes_head_and_tail_cuts(self):
review = build_phrase_review(
_timeline([_segment(0, 4, words=_words([(0, 1), (1, 3), (3, 4)]))])
)
review["phrases"][0]["trim_start"] = 1.0
review["phrases"][0]["trim_end"] = 3.0
result = phrase_review_to_actions(review)
spans = [(a["start"], a["end"]) for a in result["actions"] if a["kind"] == "cut"]
assert spans == [(0.0, 1.0), (3.0, 4.0)]
def test_inactive_phrase_is_cut_whole_ignoring_its_trim(self):
review = build_phrase_review(_timeline([_segment(0, 4)]))
review["phrases"][0].update({"active": False, "trim_start": 1.0, "trim_end": 3.0})
result = phrase_review_to_actions(review)
assert [(a["start"], a["end"]) for a in result["actions"]] == [(0.0, 4.0)]
def test_zoom_follows_the_trimmed_span(self):
review = build_phrase_review(_timeline([_segment(0, 4)]))
review["phrases"][0].update({"emphasis": 1, "trim_start": 1.0, "trim_end": 3.0})
result = phrase_review_to_actions(review)
zoom = next(a for a in result["actions"] if a["kind"] == "zoom")
assert (zoom["start"], zoom["end"]) == (1.0, 3.0)
def test_impossible_trim_is_ignored(self):
review = build_phrase_review(_timeline([_segment(0, 4)]))
review["phrases"][0].update({"trim_start": 3.0, "trim_end": 1.0})
result = phrase_review_to_actions(review)
assert result["actions"] == []
def test_emphasis_out_of_range_is_clamped(self):
review = build_phrase_review(_timeline([_segment(0, 2)]))
review["phrases"][0]["emphasis"] = 99
result = phrase_review_to_actions(review)
assert result["actions"][0]["params"]["scale"] == ZOOM_SCALE_BY_LEVEL[3]
def test_rows_that_make_no_sense_are_skipped(self):
result = phrase_review_to_actions(
{"phrases": ["nope", {"start": 5, "end": 1}, {"start": 0, "end": 1}]}
)
assert result["actions"] == []
class TestManualZooms:
def test_manual_zoom_becomes_an_action_without_a_scale(self):
review = build_phrase_review(_timeline([_segment(0, 10)]))
review["zooms"] = [{"start": 2.0, "end": 4.0}]
result = phrase_review_to_actions(review)
zoom = next(a for a in result["actions"] if a["kind"] == "zoom")
assert (zoom["start"], zoom["end"]) == (2.0, 4.0)
# Sem scale: o aplicador usa o zoom_scale configurado pelo usuário.
assert "scale" not in zoom["params"]
def test_zoom_shorter_than_the_ramp_is_refused(self):
review = build_phrase_review(_timeline([_segment(0, 10)]))
review["zooms"] = [{"start": 2.0, "end": 2.1}]
assert phrase_review_to_actions(review)["actions"] == []
def test_manual_zoom_coexists_with_phrase_emphasis(self):
review = build_phrase_review(_timeline([_segment(0, 10)]))
review["phrases"][0]["emphasis"] = 2
review["zooms"] = [{"start": 2.0, "end": 4.0}]
zooms = [a for a in phrase_review_to_actions(review)["actions"] if a["kind"] == "zoom"]
assert len(zooms) == 2
def test_malformed_zoom_rows_are_skipped(self):
review = build_phrase_review(_timeline([_segment(0, 10)]))
review["zooms"] = ["nope", {"start": 5}, {"start": 4, "end": 1}]
assert phrase_review_to_actions(review)["actions"] == []
def test_saved_zooms_are_restored(self):
review = merge_saved_decisions(
build_phrase_review(_timeline([_segment(0, 10)])),
{"phrases": [], "zooms": [{"start": 1.0, "end": 3.0}]},
)
assert review["zooms"] == [{"start": 1.0, "end": 3.0}]
def test_new_review_starts_with_no_manual_zooms(self):
assert build_phrase_review(_timeline([_segment(0, 2)]))["zooms"] == []
class TestRoundTrip:
def test_review_survives_actions_and_back(self):
timeline = _timeline(
[_segment(0, 2, peak=0.9), _segment(2, 4), _segment(4, 6, peak=0.5)]
)
first = build_phrase_review(timeline)
first["phrases"][1]["active"] = False
actions = phrase_review_to_actions(first)
second = build_phrase_review(timeline, actions)
assert [p["active"] for p in second["phrases"]] == [True, False, True]
assert [p["emphasis"] for p in second["phrases"]] == [3, 0, 2]
class TestSnapToWords:
def test_snaps_to_the_nearest_start(self):
words = _words([(1.0, 1.5), (1.5, 2.0)])
assert snap_to_words(1.6, words, 1.6, "in") == 1.5
def test_snaps_to_the_nearest_end(self):
words = _words([(1.0, 1.5), (1.5, 2.0)])
assert snap_to_words(1.9, words, 1.9, "out") == 2.0
def test_falls_back_without_word_timings(self):
assert snap_to_words(1.2, [], 3.4, "in") == 3.4
class TestResolveSource:
def test_finds_the_media_beside_its_timeline(self, tmp_path):
media = tmp_path / "take.mov"
media.write_bytes(b"0")
timeline = tmp_path / "take_voice_timeline.json"
assert resolve_source("take.mov", str(timeline)) == str(media)
def test_falls_back_to_the_project_folder(self, tmp_path):
media_dir = tmp_path / "midia"
media_dir.mkdir()
media = media_dir / "take.mov"
media.write_bytes(b"0")
timeline = tmp_path / "json" / "take_voice_timeline.json"
assert resolve_source("take.mov", str(timeline), [str(media_dir)]) == str(media)
def test_absolute_path_is_used_as_is(self, tmp_path):
media = tmp_path / "take.mov"
media.write_bytes(b"0")
assert resolve_source(str(media), "") == str(media)
def test_missing_media_resolves_to_empty(self, tmp_path):
assert resolve_source("take.mov", str(tmp_path / "x_voice_timeline.json")) == ""
def test_stale_absolute_path_still_finds_the_file_by_name(self, tmp_path):
# The fixture's source is an absolute path that no longer exists (the
# everyday case: the project moved). Falling back to the file name next
# to the timeline is what keeps the preview working after a move.
media = tmp_path / "take.mov"
media.write_bytes(b"0")
assert resolve_source("/tmp/gone/take.mov", str(tmp_path / "t.json")) == str(media)
def test_review_carries_the_resolved_path(self, tmp_path):
media = tmp_path / "take.mov"
media.write_bytes(b"0")
review = build_phrase_review(
{**_timeline([_segment(0, 1)]), "source": "take.mov"},
voice_timeline_path=str(tmp_path / "take_voice_timeline.json"),
)
assert review["source_path"] == str(media)
def test_review_without_media_reports_no_path(self, tmp_path):
review = build_phrase_review(
_timeline([_segment(0, 1)]),
voice_timeline_path=str(tmp_path / "take_voice_timeline.json"),
)
assert review["source_path"] == ""
class TestEmotion:
def test_segment_emotion_reaches_the_phrase(self):
segment = _segment(0, 2)
segment["emotion"] = "excited"
segment["emotion_confidence"] = 0.72
review = build_phrase_review(_timeline([segment]))
assert review["phrases"][0]["emotion"] == "excited"
assert review["phrases"][0]["emotion_confidence"] == 0.72
def test_defaults_to_neutral_when_absent(self):
review = build_phrase_review(_timeline([_segment(0, 2)]))
assert review["phrases"][0]["emotion"] == "neutral"
assert review["phrases"][0]["emotion_confidence"] == 0.0
def test_availability_comes_from_the_analysis_layers(self):
assert build_phrase_review(_timeline([_segment(0, 1)]))["emotion_available"] is False
timeline = {**_timeline([_segment(0, 1)]), "layers": {"emotion": True}}
assert build_phrase_review(timeline)["emotion_available"] is True
class TestMergeSavedDecisions:
def test_saved_decisions_win_over_the_derivation(self):
timeline = _timeline([_segment(0, 2, peak=0.9), _segment(2, 4)])
saved = {
"phrases": [
{"index": 0, "start": 0.0, "emphasis": 0, "active": False,
"track": TRACK_BACKSTAGE, "text": "corrigido"},
]
}
review = merge_saved_decisions(build_phrase_review(timeline), saved)
first = review["phrases"][0]
assert (first["emphasis"], first["active"]) == (0, False)
assert first["track"] == TRACK_BACKSTAGE
assert first["text"] == "corrigido"
assert review["phrases"][1]["active"] is True
def test_fresh_analysis_fields_are_not_overwritten(self):
segment = _segment(0, 2, peak=0.9)
segment["emotion"] = "tense"
review = merge_saved_decisions(
build_phrase_review(_timeline([segment])),
{"phrases": [{"index": 0, "start": 0.0, "emphasis": 1}]},
)
assert review["phrases"][0]["emotion"] == "tense"
assert review["phrases"][0]["peak_emphasis"] == 0.9
def test_decision_is_dropped_when_the_line_moved(self):
review = merge_saved_decisions(
build_phrase_review(_timeline([_segment(10, 12, peak=0.9)])),
{"phrases": [{"index": 0, "start": 0.0, "active": False}]},
)
assert review["phrases"][0]["active"] is True
def test_saved_trim_is_restored(self):
review = merge_saved_decisions(
build_phrase_review(_timeline([_segment(0, 4)])),
{"phrases": [{"index": 0, "start": 0.0, "trim_start": 1.0, "trim_end": 3.0}]},
)
assert (review["phrases"][0]["trim_start"], review["phrases"][0]["trim_end"]) == (1.0, 3.0)
def test_impossible_saved_trim_is_ignored(self):
review = merge_saved_decisions(
build_phrase_review(_timeline([_segment(0, 4)])),
{"phrases": [{"index": 0, "start": 0.0, "trim_start": 9.0, "trim_end": 12.0}]},
)
assert (review["phrases"][0]["trim_start"], review["phrases"][0]["trim_end"]) == (0.0, 4.0)
def test_no_saved_review_is_a_no_op(self):
review = build_phrase_review(_timeline([_segment(0, 2)]))
assert merge_saved_decisions(review, None) is review
class TestPersistence:
def test_paths_are_named_after_the_timeline(self, tmp_path):
timeline_path = tmp_path / "take_voice_timeline.json"
review_path, actions_path = review_paths(str(timeline_path))
assert review_path.name == "take_phrase_review.json"
assert actions_path.name == "take_phrase_actions.json"
def test_save_writes_both_files_and_load_reads_it_back(self, tmp_path):
timeline_path = tmp_path / "take_voice_timeline.json"
review = build_phrase_review(_timeline([_segment(0, 2)]))
review["phrases"][0]["emphasis"] = 3
review_path, actions_path = save_phrase_review(str(timeline_path), review)
assert review_path.is_file() and actions_path.is_file()
written = json.loads(actions_path.read_text(encoding="utf-8"))
assert written["actions"][0]["kind"] == "zoom"
assert load_phrase_review(str(timeline_path))["phrases"][0]["emphasis"] == 3
def test_load_returns_none_when_absent_or_broken(self, tmp_path):
timeline_path = tmp_path / "take_voice_timeline.json"
assert load_phrase_review(str(timeline_path)) is None
review_path, _ = review_paths(str(timeline_path))
review_path.write_text("{ not json", encoding="utf-8")
assert load_phrase_review(str(timeline_path)) is None
if __name__ == "__main__":
pytest.main([__file__, "-v"])
+15 -7
View File
@@ -40,7 +40,7 @@ WORDS = [
w("the", 2.5, 2.6), w("the", 2.5, 2.6),
w("show", 2.65, 3.0), w("show", 2.65, 3.0),
w("is", 3.05, 3.15), w("is", 3.05, 3.15),
w("um", 3.2, 3.5), w("uh", 3.2, 3.5),
w("great.", 3.6, 4.0), w("great.", 3.6, 4.0),
] ]
@@ -60,7 +60,7 @@ class TestNormalizeWord:
class TestFindPhraseSpans: class TestFindPhraseSpans:
def test_single_word_multiple_hits(self): def test_single_word_multiple_hits(self):
spans = find_phrase_spans(WORDS, "um") spans = find_phrase_spans(WORDS, "um")
assert spans == [(0.4, 0.6), (3.2, 3.5)] assert spans == [(0.4, 0.6)]
def test_multi_word_phrase(self): def test_multi_word_phrase(self):
spans = find_phrase_spans(WORDS, "welcome to the show") spans = find_phrase_spans(WORDS, "welcome to the show")
@@ -84,9 +84,13 @@ class TestFindPhraseSpans:
class TestFindFillerSpans: class TestFindFillerSpans:
def test_default_fillers(self): def test_default_fillers(self):
spans = find_filler_spans(WORDS) spans = find_filler_spans(WORDS)
assert (0.4, 0.6) in spans assert (0.4, 0.6) not in spans
assert (3.2, 3.5) in spans assert (3.2, 3.5) in spans
def test_um_can_still_be_explicit(self):
spans = find_filler_spans(WORDS, fillers=("um",))
assert spans == [(0.4, 0.6)]
def test_multi_word_filler(self): def test_multi_word_filler(self):
spans = find_filler_spans(WORDS, fillers=("you know",)) spans = find_filler_spans(WORDS, fillers=("you know",))
assert spans == [(2.0, 2.4)] assert spans == [(2.0, 2.4)]
@@ -96,7 +100,10 @@ class TestFindFillerSpans:
assert spans == sorted(spans) assert spans == sorted(spans)
def test_defaults_are_conservative(self): def test_defaults_are_conservative(self):
# "like" and "so" are speech, not noise — must not be default-cut. # "um", "uma", "like" and "so" are speech, not noise — must not be
# default-cut.
assert "um" not in DEFAULT_FILLERS
assert "uma" not in DEFAULT_FILLERS
assert "like" not in DEFAULT_FILLERS assert "like" not in DEFAULT_FILLERS
assert "so" not in DEFAULT_FILLERS assert "so" not in DEFAULT_FILLERS
@@ -269,11 +276,12 @@ class TestRemoveFillerWordsHandler:
assert "_defillered" in result[0].text assert "_defillered" in result[0].text
segments = _spine_segments(str(tmp_path / "project_defillered.fcpxml")) segments = _spine_segments(str(tmp_path / "project_defillered.fcpxml"))
# um (0-0.5) trims the clip head, uh (3-3.5) splits -> 2 segments, ~7s total # Only uh (3-3.5) is removed by default; Portuguese "um" is preserved
# because it is often grammatical speech ("de um jeito").
assert len(segments) == 2 assert len(segments) == 2
total = sum(c.duration.seconds for c in segments) total = sum(c.duration.seconds for c in segments)
assert total == pytest.approx(7.0, abs=0.1) assert total == pytest.approx(7.5, abs=0.1)
assert segments[0].source_start.seconds == pytest.approx(0.5, abs=0.05) assert segments[0].source_start.seconds == pytest.approx(0.0, abs=0.05)
assert segments[1].source_start.seconds == pytest.approx(3.5, abs=0.05) assert segments[1].source_start.seconds == pytest.approx(3.5, abs=0.05)
async def test_no_fillers_found_saves_nothing(self, tmp_path): async def test_no_fillers_found_saves_nothing(self, tmp_path):
+7 -2
View File
@@ -76,9 +76,14 @@ class TestParseActions:
class TestZoomValidation: class TestZoomValidation:
def test_default_scale_when_absent(self): def test_absent_scale_is_left_absent(self):
# The parser no longer stamps a default: an omitted scale must reach the
# applier untouched so it can fall back to the user's configured
# `zoom_scale` (see server_tools/_shared.py). Filling one in here would
# silently override that setting for every action the model sends
# without an explicit scale.
actions, _ = parse_actions([{"kind": "zoom", "start": 1.0, "end": 2.0}]) actions, _ = parse_actions([{"kind": "zoom", "start": 1.0, "end": 2.0}])
assert actions[0].params["scale"] == 1.3 assert "scale" not in actions[0].params
def test_rejects_scale_below_one(self): def test_rejects_scale_below_one(self):
_, errors = parse_actions([ _, errors = parse_actions([
+26
View File
@@ -94,6 +94,32 @@ class TestApplyVoiceActionsHandler:
texts = [t.text for t in titles[0].iter() if t.text] texts = [t.text for t in titles[0].iter() if t.text]
assert any("SEGURANÇA" in t for t in texts) assert any("SEGURANÇA" in t for t in texts)
async def test_text_callout_defaults_fit_the_frame(self, project):
from fcpxml.writer import FCPXMLModifier
from server import handle_apply_voice_actions
await handle_apply_voice_actions({
"filepath": str(project),
"actions": [{
"kind": "text", "start": 3.0, "end": 4.0,
"params": {"content": "PRÓTESES DE SILICONE"},
}],
})
modifier = FCPXMLModifier(str(_out(project)))
report = modifier.validate_subtitle_layout()
assert report["summary"]["outside_frame"] == 0
title = modifier.root.find(".//title")
style = title.find("text-style-def/text-style")
position = next(
p.get("value")
for p in title.findall("param")
if p.get("name") == "Position"
)
assert float(style.get("fontSize")) < 530
assert position != "0 0"
async def test_applies_marker(self, project): async def test_applies_marker(self, project):
from server import handle_apply_voice_actions from server import handle_apply_voice_actions
+12
View File
@@ -11,6 +11,7 @@ import pytest
from fcpxml.voice_timeline import ( from fcpxml.voice_timeline import (
VOICE_TIMELINE_VERSION, VOICE_TIMELINE_VERSION,
annotate_emotions,
build_voice_timeline, build_voice_timeline,
enrich_words, enrich_words,
load_voice_timeline, load_voice_timeline,
@@ -60,6 +61,14 @@ class TestEnrichWords:
assert all(w["energy_norm"] == 0.0 for w in enriched) assert all(w["energy_norm"] == 0.0 for w in enriched)
assert all(w["pitch_delta"] == 0.0 for w in enriched) assert all(w["pitch_delta"] == 0.0 for w in enriched)
def test_emotion_labels_are_added_when_enabled(self):
enriched = enrich_words(_TRANSCRIPT["words"], _PITCH, _ENERGY)
emotional = annotate_emotions(enriched, enabled=True, sensitivity=0.1)
loudest = max(emotional, key=lambda w: w["arousal"])
assert loudest["word"] == "seguranca"
assert loudest["emotion"] in {"excited", "tense", "neutral"}
assert 0.0 <= loudest["emotion_confidence"] <= 1.0
class TestBuildVoiceTimeline: class TestBuildVoiceTimeline:
@pytest.fixture @pytest.fixture
@@ -74,6 +83,7 @@ class TestBuildVoiceTimeline:
for key in ("version", "source", "language", "scales", "summary", "speakers", "segments"): for key in ("version", "source", "language", "scales", "summary", "speakers", "segments"):
assert key in timeline assert key in timeline
assert timeline["version"] == VOICE_TIMELINE_VERSION assert timeline["version"] == VOICE_TIMELINE_VERSION
assert "emotion" in timeline["layers"]
def test_scales_document_every_word_metric(self, timeline): def test_scales_document_every_word_metric(self, timeline):
word = timeline["segments"][0]["words"][0] word = timeline["segments"][0]["words"][0]
@@ -106,6 +116,8 @@ class TestBuildVoiceTimeline:
quiet_segment = timeline["segments"][0] quiet_segment = timeline["segments"][0]
assert loud_segment["avg_energy"] > quiet_segment["avg_energy"] assert loud_segment["avg_energy"] > quiet_segment["avg_energy"]
assert loud_segment["peak_emphasis"] >= max(w["emphasis"] for w in loud_segment["words"]) assert loud_segment["peak_emphasis"] >= max(w["emphasis"] for w in loud_segment["words"])
assert "emotion" in loud_segment
assert "arousal" in loud_segment
def test_peak_moments_are_sorted_by_emphasis(self, timeline): def test_peak_moments_are_sorted_by_emphasis(self, timeline):
peaks = timeline["summary"]["peak_moments"] peaks = timeline["summary"]["peak_moments"]
+4
View File
@@ -94,6 +94,10 @@ class TestBuildVoiceTimelineHandler:
}, },
"emotion_enabled": False, "emotion_enabled": False,
"emotion_sensitivity": 0.5, "emotion_sensitivity": 0.5,
"zoom_scale": 1.3,
"zoom_mode": "in_out",
"zoom_ease_in": 0.25,
"zoom_ease_out": 0.04,
} }
monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(0.01)) monkeypatch.setattr(server_mod, "load_voice_analysis_config", lambda: config(0.01))