Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
323 lines
12 KiB
Swift
323 lines
12 KiB
Swift
import SwiftUI
|
||
|
||
/// Guia "Análise de Voz" — parâmetros do motor de análise acústica da fala.
|
||
///
|
||
/// Controla os limiares que decidem quais palavras são candidatas a
|
||
/// punch-in/destaque: energia (intensidade da fala), o índice de ênfase e
|
||
/// seus pesos, e a camada opcional de emoção. Os valores ficam em
|
||
/// `~/.fcp-mcp-server/config.json` (via `model_manager.save_voice_analysis_config`),
|
||
/// os mesmos lidos pelas ferramentas de análise — não em UserDefaults, para
|
||
/// que a análise rode com exatamente o que a tela mostra.
|
||
struct VoiceAnalysisView: View {
|
||
@State private var config = VoiceAnalysisConfig.defaults
|
||
@State private var isLoading = true
|
||
@State private var errorMessage: String?
|
||
|
||
var body: some View {
|
||
Form {
|
||
if isLoading {
|
||
Section {
|
||
ProgressView().controlSize(.small)
|
||
.frame(maxWidth: .infinity, alignment: .center)
|
||
}
|
||
} else {
|
||
energySection
|
||
emphasisSection
|
||
zoomSection
|
||
weightsSection
|
||
emotionSection
|
||
resetSection
|
||
}
|
||
if let errorMessage {
|
||
Section {
|
||
Label(errorMessage, systemImage: "exclamationmark.triangle.fill")
|
||
.foregroundStyle(.red)
|
||
}
|
||
}
|
||
}
|
||
.formStyle(.grouped)
|
||
.task { await load() }
|
||
}
|
||
|
||
// MARK: - Energia
|
||
|
||
private var energySection: some View {
|
||
Section {
|
||
sliderRow(
|
||
title: "Limiar de energia",
|
||
value: $config.energyThreshold,
|
||
help: "Acima deste valor a fala conta como \"alta energia\"."
|
||
)
|
||
} header: {
|
||
Text("Energia da Fala")
|
||
} footer: {
|
||
Text("A energia de cada palavra é normalizada (0–1) pelo trecho mais alto do áudio. Valores mais baixos marcam mais palavras como intensas.")
|
||
.font(.caption)
|
||
.foregroundStyle(.secondary)
|
||
}
|
||
}
|
||
|
||
// MARK: - Ênfase
|
||
|
||
private var emphasisSection: some View {
|
||
Section {
|
||
sliderRow(
|
||
title: "Limiar de ênfase",
|
||
value: $config.emphasisThreshold,
|
||
help: "Acima deste valor a palavra vira candidata a punch-in/destaque."
|
||
)
|
||
} header: {
|
||
Text("Índice de Ênfase")
|
||
} footer: {
|
||
Text("O índice é a média ponderada de energia, variação de tom, variação de ritmo, pausa anterior e duração — com os pesos abaixo. Por ser média, picos reais ficam entre 0,55 e 0,80: acima de 0,85 quase nada é selecionado.")
|
||
.font(.caption)
|
||
.foregroundStyle(.secondary)
|
||
}
|
||
}
|
||
|
||
private var weightsSection: some View {
|
||
Section {
|
||
sliderRow(title: "Energia", value: $config.weightEnergy, range: 0...1)
|
||
sliderRow(title: "Variação de tom", value: $config.weightPitch, range: 0...1)
|
||
sliderRow(title: "Variação de ritmo", value: $config.weightRate, range: 0...1)
|
||
sliderRow(title: "Pausa anterior", value: $config.weightPause, range: 0...1)
|
||
sliderRow(title: "Duração da palavra", value: $config.weightDuration, range: 0...1)
|
||
} header: {
|
||
Text("Pesos do Índice de Ênfase")
|
||
} footer: {
|
||
Text("Não precisam somar 1 — são normalizados internamente. O que importa é a proporção entre eles.")
|
||
.font(.caption)
|
||
.foregroundStyle(.secondary)
|
||
}
|
||
}
|
||
|
||
private var zoomSection: some View {
|
||
Section {
|
||
sliderRow(
|
||
title: "Zoom na ênfase",
|
||
value: $config.zoomScale,
|
||
range: 1.0...3.0,
|
||
readout: "\(Int(config.zoomScale * 100))%",
|
||
help: "Fator aplicado nos punch-ins de ênfase. 130% equivale a escala 1,30 no Final Cut."
|
||
)
|
||
Picker("Movimento", selection: $config.zoomMode) {
|
||
Text("Zoom in e out").tag("in_out")
|
||
Text("Só zoom in").tag("in")
|
||
Text("Só zoom out").tag("out")
|
||
}
|
||
.onChange(of: config.zoomMode) { _, _ in save() }
|
||
sliderRow(
|
||
title: "Velocidade do zoom in",
|
||
value: $config.zoomEaseIn,
|
||
range: 0.05...2.0,
|
||
readout: String(format: "%.2fs", config.zoomEaseIn),
|
||
help: "Duração da entrada do zoom. Menor é mais rápido."
|
||
)
|
||
sliderRow(
|
||
title: "Velocidade do zoom out",
|
||
value: $config.zoomEaseOut,
|
||
range: 0.01...2.0,
|
||
readout: String(format: "%.2fs", config.zoomEaseOut),
|
||
help: "Duração da saída do zoom. Menor é mais seco."
|
||
)
|
||
} header: {
|
||
Text("Zoom de Ênfase")
|
||
} footer: {
|
||
Text("Esses valores viram o padrão para ações de zoom que não trouxerem scale/ease/ease_out no JSON da edição por voz.")
|
||
.font(.caption)
|
||
.foregroundStyle(.secondary)
|
||
}
|
||
}
|
||
|
||
// MARK: - Emoção
|
||
|
||
private var emotionSection: some View {
|
||
Section {
|
||
Toggle("Detectar emoção durante a fala", isOn: $config.emotionEnabled)
|
||
.onChange(of: config.emotionEnabled) { _, _ in save() }
|
||
if config.emotionEnabled {
|
||
sliderRow(
|
||
title: "Sensibilidade",
|
||
value: $config.emotionSensitivity,
|
||
help: "Confiança mínima para aceitar uma emoção detectada."
|
||
)
|
||
}
|
||
} header: {
|
||
Text("Emoção")
|
||
} footer: {
|
||
Text("A emoção nunca decide um corte sozinha — entra combinada com energia, tom e ênfase. Exige o componente opcional de emoção instalado; sem ele, a análise segue normalmente sem essa camada.")
|
||
.font(.caption)
|
||
.foregroundStyle(.secondary)
|
||
}
|
||
}
|
||
|
||
private var resetSection: some View {
|
||
Section {
|
||
Button("Restaurar padrões") {
|
||
config = .defaults
|
||
save()
|
||
}
|
||
}
|
||
}
|
||
|
||
// MARK: - Componentes
|
||
|
||
/// Slider com rótulo à esquerda e valor numérico à direita, salvando no
|
||
/// backend só quando o arraste termina (evita uma escrita por quadro).
|
||
private func sliderRow(
|
||
title: String,
|
||
value: Binding<Double>,
|
||
range: ClosedRange<Double> = 0...1,
|
||
readout: String? = nil,
|
||
help: String? = nil
|
||
) -> some View {
|
||
VStack(alignment: .leading, spacing: 2) {
|
||
HStack {
|
||
Text(title)
|
||
Spacer()
|
||
Text(readout ?? String(format: "%.2f", value.wrappedValue))
|
||
.monospacedDigit()
|
||
.foregroundStyle(.secondary)
|
||
}
|
||
Slider(value: value, in: range) { editing in
|
||
if !editing { save() }
|
||
}
|
||
if let help {
|
||
Text(help)
|
||
.font(.caption)
|
||
.foregroundStyle(.secondary)
|
||
}
|
||
}
|
||
}
|
||
|
||
// MARK: - Backend
|
||
|
||
@MainActor
|
||
private func load() async {
|
||
await withCheckedContinuation { continuation in
|
||
PythonBridge.call(command: "voice_analysis") { result, error in
|
||
DispatchQueue.main.async {
|
||
if let result {
|
||
config = VoiceAnalysisConfig(from: result)
|
||
} else if let error {
|
||
errorMessage = error
|
||
}
|
||
isLoading = false
|
||
continuation.resume()
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
private func save() {
|
||
PythonBridge.call(command: "set_voice_analysis", arguments: config.arguments()) { _, error in
|
||
DispatchQueue.main.async { errorMessage = error }
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Os parâmetros de análise de voz, no formato que a tela edita e o bridge
|
||
/// (`admin/models_api.py` → `set_voice_analysis`) persiste.
|
||
struct VoiceAnalysisConfig {
|
||
var energyThreshold: Double
|
||
var emphasisThreshold: Double
|
||
var weightEnergy: Double
|
||
var weightPitch: Double
|
||
var weightRate: Double
|
||
var weightPause: Double
|
||
var weightDuration: Double
|
||
var emotionEnabled: Bool
|
||
var emotionSensitivity: Double
|
||
var zoomScale: Double
|
||
var zoomMode: String
|
||
var zoomEaseIn: Double
|
||
var zoomEaseOut: Double
|
||
|
||
static let defaults = VoiceAnalysisConfig(
|
||
energyThreshold: 0.5,
|
||
emphasisThreshold: 0.60,
|
||
weightEnergy: 0.30,
|
||
weightPitch: 0.25,
|
||
weightRate: 0.20,
|
||
weightPause: 0.15,
|
||
weightDuration: 0.10,
|
||
emotionEnabled: false,
|
||
emotionSensitivity: 0.5,
|
||
zoomScale: 1.30,
|
||
zoomMode: "in_out",
|
||
zoomEaseIn: 0.25,
|
||
zoomEaseOut: 0.04
|
||
)
|
||
|
||
init(
|
||
energyThreshold: Double,
|
||
emphasisThreshold: Double,
|
||
weightEnergy: Double,
|
||
weightPitch: Double,
|
||
weightRate: Double,
|
||
weightPause: Double,
|
||
weightDuration: Double,
|
||
emotionEnabled: Bool,
|
||
emotionSensitivity: Double,
|
||
zoomScale: Double,
|
||
zoomMode: String,
|
||
zoomEaseIn: Double,
|
||
zoomEaseOut: Double
|
||
) {
|
||
self.energyThreshold = energyThreshold
|
||
self.emphasisThreshold = emphasisThreshold
|
||
self.weightEnergy = weightEnergy
|
||
self.weightPitch = weightPitch
|
||
self.weightRate = weightRate
|
||
self.weightPause = weightPause
|
||
self.weightDuration = weightDuration
|
||
self.emotionEnabled = emotionEnabled
|
||
self.emotionSensitivity = emotionSensitivity
|
||
self.zoomScale = zoomScale
|
||
self.zoomMode = zoomMode
|
||
self.zoomEaseIn = zoomEaseIn
|
||
self.zoomEaseOut = zoomEaseOut
|
||
}
|
||
|
||
/// Lê a resposta do bridge, caindo no padrão para qualquer campo ausente.
|
||
init(from json: [String: Any]) {
|
||
let defaults = VoiceAnalysisConfig.defaults
|
||
let weights = json["emphasis_weights"] as? [String: Any] ?? [:]
|
||
self.init(
|
||
energyThreshold: json["energy_threshold"] as? Double ?? defaults.energyThreshold,
|
||
emphasisThreshold: json["emphasis_threshold"] as? Double ?? defaults.emphasisThreshold,
|
||
weightEnergy: weights["energy"] as? Double ?? defaults.weightEnergy,
|
||
weightPitch: weights["pitch_variation"] as? Double ?? defaults.weightPitch,
|
||
weightRate: weights["rate_variation"] as? Double ?? defaults.weightRate,
|
||
weightPause: weights["pause_before"] as? Double ?? defaults.weightPause,
|
||
weightDuration: weights["duration"] as? Double ?? defaults.weightDuration,
|
||
emotionEnabled: json["emotion_enabled"] as? Bool ?? defaults.emotionEnabled,
|
||
emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity,
|
||
zoomScale: json["zoom_scale"] as? Double ?? defaults.zoomScale,
|
||
zoomMode: json["zoom_mode"] as? String ?? defaults.zoomMode,
|
||
zoomEaseIn: json["zoom_ease_in"] as? Double ?? defaults.zoomEaseIn,
|
||
zoomEaseOut: json["zoom_ease_out"] as? Double ?? defaults.zoomEaseOut
|
||
)
|
||
}
|
||
|
||
func arguments() -> [String: Any] {
|
||
[
|
||
"energy_threshold": energyThreshold,
|
||
"emphasis_threshold": emphasisThreshold,
|
||
"emphasis_weights": [
|
||
"energy": weightEnergy,
|
||
"pitch_variation": weightPitch,
|
||
"rate_variation": weightRate,
|
||
"pause_before": weightPause,
|
||
"duration": weightDuration,
|
||
],
|
||
"emotion_enabled": emotionEnabled,
|
||
"emotion_sensitivity": emotionSensitivity,
|
||
"zoom_scale": zoomScale,
|
||
"zoom_mode": zoomMode,
|
||
"zoom_ease_in": zoomEaseIn,
|
||
"zoom_ease_out": zoomEaseOut,
|
||
]
|
||
}
|
||
}
|