feat: etapa 5 do assistente — revisão de ênfases com timeline

Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-19 21:29:27 -04:00
co-authored by Claude Opus 5
parent e7748c2c58
commit 1bebee4359
31 changed files with 4622 additions and 83 deletions
+68 -4
View File
@@ -23,6 +23,7 @@ struct VoiceAnalysisView: View {
} else {
energySection
emphasisSection
zoomSection
weightsSection
emotionSection
resetSection
@@ -90,6 +91,44 @@ struct VoiceAnalysisView: View {
}
}
private var zoomSection: some View {
Section {
sliderRow(
title: "Zoom na ênfase",
value: $config.zoomScale,
range: 1.0...3.0,
readout: "\(Int(config.zoomScale * 100))%",
help: "Fator aplicado nos punch-ins de ênfase. 130% equivale a escala 1,30 no Final Cut."
)
Picker("Movimento", selection: $config.zoomMode) {
Text("Zoom in e out").tag("in_out")
Text("Só zoom in").tag("in")
Text("Só zoom out").tag("out")
}
.onChange(of: config.zoomMode) { _, _ in save() }
sliderRow(
title: "Velocidade do zoom in",
value: $config.zoomEaseIn,
range: 0.05...2.0,
readout: String(format: "%.2fs", config.zoomEaseIn),
help: "Duração da entrada do zoom. Menor é mais rápido."
)
sliderRow(
title: "Velocidade do zoom out",
value: $config.zoomEaseOut,
range: 0.01...2.0,
readout: String(format: "%.2fs", config.zoomEaseOut),
help: "Duração da saída do zoom. Menor é mais seco."
)
} header: {
Text("Zoom de Ênfase")
} footer: {
Text("Esses valores viram o padrão para ações de zoom que não trouxerem scale/ease/ease_out no JSON da edição por voz.")
.font(.caption)
.foregroundStyle(.secondary)
}
}
// MARK: - Emoção
private var emotionSection: some View {
@@ -129,13 +168,14 @@ struct VoiceAnalysisView: View {
title: String,
value: Binding<Double>,
range: ClosedRange<Double> = 0...1,
readout: String? = nil,
help: String? = nil
) -> some View {
VStack(alignment: .leading, spacing: 2) {
HStack {
Text(title)
Spacer()
Text(String(format: "%.2f", value.wrappedValue))
Text(readout ?? String(format: "%.2f", value.wrappedValue))
.monospacedDigit()
.foregroundStyle(.secondary)
}
@@ -188,6 +228,10 @@ struct VoiceAnalysisConfig {
var weightDuration: Double
var emotionEnabled: Bool
var emotionSensitivity: Double
var zoomScale: Double
var zoomMode: String
var zoomEaseIn: Double
var zoomEaseOut: Double
static let defaults = VoiceAnalysisConfig(
energyThreshold: 0.5,
@@ -198,7 +242,11 @@ struct VoiceAnalysisConfig {
weightPause: 0.15,
weightDuration: 0.10,
emotionEnabled: false,
emotionSensitivity: 0.5
emotionSensitivity: 0.5,
zoomScale: 1.30,
zoomMode: "in_out",
zoomEaseIn: 0.25,
zoomEaseOut: 0.04
)
init(
@@ -210,7 +258,11 @@ struct VoiceAnalysisConfig {
weightPause: Double,
weightDuration: Double,
emotionEnabled: Bool,
emotionSensitivity: Double
emotionSensitivity: Double,
zoomScale: Double,
zoomMode: String,
zoomEaseIn: Double,
zoomEaseOut: Double
) {
self.energyThreshold = energyThreshold
self.emphasisThreshold = emphasisThreshold
@@ -221,6 +273,10 @@ struct VoiceAnalysisConfig {
self.weightDuration = weightDuration
self.emotionEnabled = emotionEnabled
self.emotionSensitivity = emotionSensitivity
self.zoomScale = zoomScale
self.zoomMode = zoomMode
self.zoomEaseIn = zoomEaseIn
self.zoomEaseOut = zoomEaseOut
}
/// Lê a resposta do bridge, caindo no padrão para qualquer campo ausente.
@@ -236,7 +292,11 @@ struct VoiceAnalysisConfig {
weightPause: weights["pause_before"] as? Double ?? defaults.weightPause,
weightDuration: weights["duration"] as? Double ?? defaults.weightDuration,
emotionEnabled: json["emotion_enabled"] as? Bool ?? defaults.emotionEnabled,
emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity
emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity,
zoomScale: json["zoom_scale"] as? Double ?? defaults.zoomScale,
zoomMode: json["zoom_mode"] as? String ?? defaults.zoomMode,
zoomEaseIn: json["zoom_ease_in"] as? Double ?? defaults.zoomEaseIn,
zoomEaseOut: json["zoom_ease_out"] as? Double ?? defaults.zoomEaseOut
)
}
@@ -253,6 +313,10 @@ struct VoiceAnalysisConfig {
],
"emotion_enabled": emotionEnabled,
"emotion_sensitivity": emotionSensitivity,
"zoom_scale": zoomScale,
"zoom_mode": zoomMode,
"zoom_ease_in": zoomEaseIn,
"zoom_ease_out": zoomEaseOut,
]
}
}