feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e7748c2c58
commit
1bebee4359
@@ -12,6 +12,7 @@ struct GArtApp: App {
|
||||
}
|
||||
|
||||
enum ActiveTab: Hashable {
|
||||
case wizard
|
||||
case project
|
||||
case captions
|
||||
case voiceAnalysis
|
||||
@@ -20,19 +21,23 @@ enum ActiveTab: Hashable {
|
||||
}
|
||||
|
||||
struct ContentView: View {
|
||||
@State private var activeTab: ActiveTab? = .project
|
||||
@State private var activeTab: ActiveTab? = .wizard
|
||||
|
||||
var body: some View {
|
||||
NavigationSplitView {
|
||||
List(selection: $activeTab) {
|
||||
Label("Projeto", systemImage: "film")
|
||||
.tag(ActiveTab.project)
|
||||
Label("Legendas Dinâmicas", systemImage: "captions.bubble")
|
||||
.tag(ActiveTab.captions)
|
||||
Label("Análise de Voz", systemImage: "waveform")
|
||||
.tag(ActiveTab.voiceAnalysis)
|
||||
Label("Modelos", systemImage: "tray.and.arrow.down")
|
||||
.tag(ActiveTab.models)
|
||||
Label("Assistente", systemImage: "wand.and.stars")
|
||||
.tag(ActiveTab.wizard)
|
||||
Section("Avançado") {
|
||||
Label("Projeto", systemImage: "film")
|
||||
.tag(ActiveTab.project)
|
||||
Label("Legendas", systemImage: "captions.bubble")
|
||||
.tag(ActiveTab.captions)
|
||||
Label("Análise de Voz", systemImage: "waveform")
|
||||
.tag(ActiveTab.voiceAnalysis)
|
||||
Label("Modelos", systemImage: "tray.and.arrow.down")
|
||||
.tag(ActiveTab.models)
|
||||
}
|
||||
Label("Sobre", systemImage: "info.circle")
|
||||
.tag(ActiveTab.about)
|
||||
}
|
||||
@@ -40,19 +45,22 @@ struct ContentView: View {
|
||||
.navigationSplitViewColumnWidth(min: 180, ideal: 200)
|
||||
} detail: {
|
||||
switch activeTab {
|
||||
case .wizard, nil:
|
||||
WizardView().id(UUID())
|
||||
.navigationTitle("Assistente")
|
||||
case .project:
|
||||
ProjectView().id(UUID())
|
||||
.navigationTitle("Projeto")
|
||||
case .captions:
|
||||
CaptionsView().id(UUID())
|
||||
.navigationTitle("Legendas Dinâmicas")
|
||||
.navigationTitle("Legendas")
|
||||
case .voiceAnalysis:
|
||||
VoiceAnalysisView().id(UUID())
|
||||
.navigationTitle("Análise de Voz")
|
||||
case .models:
|
||||
ModelDownloadView().id(UUID())
|
||||
.navigationTitle("Modelos")
|
||||
case .about, nil:
|
||||
case .about:
|
||||
AboutView()
|
||||
.navigationTitle("Sobre")
|
||||
}
|
||||
|
||||
@@ -19,6 +19,7 @@ import UniformTypeIdentifiers
|
||||
/// assunto.
|
||||
struct CaptionsView: View {
|
||||
@State private var config = CaptionStyleConfig.defaults
|
||||
@State private var plainConfig = PlainSubtitleConfig.defaults
|
||||
@State private var isLoading = true
|
||||
@State private var errorMessage: String?
|
||||
|
||||
@@ -53,6 +54,13 @@ struct CaptionsView: View {
|
||||
)
|
||||
}
|
||||
|
||||
private func plainBound<T>(_ keyPath: WritableKeyPath<PlainSubtitleConfig, T>) -> Binding<T> {
|
||||
Binding(
|
||||
get: { plainConfig[keyPath: keyPath] },
|
||||
set: { plainConfig[keyPath: keyPath] = $0; savePlain() }
|
||||
)
|
||||
}
|
||||
|
||||
private func colorBound(_ keyPath: WritableKeyPath<CaptionStyleConfig, String>) -> Binding<Color> {
|
||||
Binding(
|
||||
get: { Color(rgbaString: config[keyPath: keyPath]) },
|
||||
@@ -60,6 +68,13 @@ struct CaptionsView: View {
|
||||
)
|
||||
}
|
||||
|
||||
private func plainColorBound(_ keyPath: WritableKeyPath<PlainSubtitleConfig, String>) -> Binding<Color> {
|
||||
Binding(
|
||||
get: { Color(rgbaString: plainConfig[keyPath: keyPath]) },
|
||||
set: { plainConfig[keyPath: keyPath] = $0.fcpxmlColorString; savePlain() }
|
||||
)
|
||||
}
|
||||
|
||||
var body: some View {
|
||||
HSplitView {
|
||||
previewColumn
|
||||
@@ -152,6 +167,7 @@ struct CaptionsView: View {
|
||||
positionSection
|
||||
bodySection
|
||||
emphasisSection
|
||||
plainSubtitleSection
|
||||
calibrationSection
|
||||
}
|
||||
if let errorMessage {
|
||||
@@ -225,6 +241,35 @@ struct CaptionsView: View {
|
||||
}
|
||||
}
|
||||
|
||||
private var plainSubtitleSection: some View {
|
||||
Section("Legenda comum") {
|
||||
Picker("Fonte", selection: plainBound(\.font)) {
|
||||
ForEach(fontChoices, id: \.self) { Text($0).tag($0) }
|
||||
}
|
||||
slider(
|
||||
"Tamanho",
|
||||
value: plainBound(\.fontSize), in: 28...300, step: 1,
|
||||
readout: "\(Int(plainConfig.fontSize))pt",
|
||||
help: "Tamanho da legenda comum editável no Final Cut."
|
||||
)
|
||||
slider(
|
||||
"Máximo de palavras",
|
||||
value: plainBound(\.maxWords), in: 1...14, step: 1,
|
||||
readout: "\(Int(plainConfig.maxWords))",
|
||||
help: "Quantidade máxima de palavras por bloco de legenda."
|
||||
)
|
||||
slider(
|
||||
"Altura",
|
||||
value: plainBound(\.positionY), in: -1200...300, step: 1,
|
||||
readout: "\(Int(plainConfig.positionY))",
|
||||
help: "Posição vertical da legenda comum no quadro; valores mais negativos descem."
|
||||
)
|
||||
ColorPicker("Cor", selection: plainColorBound(\.fontColor), supportsOpacity: true)
|
||||
Toggle("Usar letra maiúscula", isOn: plainBound(\.uppercase))
|
||||
Toggle("Manter vírgula e ponto", isOn: plainBound(\.keepPunctuation))
|
||||
}
|
||||
}
|
||||
|
||||
private var calibrationSection: some View {
|
||||
Section {
|
||||
slider(
|
||||
@@ -305,8 +350,17 @@ struct CaptionsView: View {
|
||||
} else if let error {
|
||||
errorMessage = error
|
||||
}
|
||||
isLoading = false
|
||||
continuation.resume()
|
||||
PythonBridge.call(command: "plain_subtitle_config") { plainResult, plainError in
|
||||
DispatchQueue.main.async {
|
||||
if let plainResult {
|
||||
plainConfig = PlainSubtitleConfig(from: plainResult)
|
||||
} else if let plainError {
|
||||
errorMessage = plainError
|
||||
}
|
||||
isLoading = false
|
||||
continuation.resume()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -317,6 +371,12 @@ struct CaptionsView: View {
|
||||
DispatchQueue.main.async { errorMessage = error }
|
||||
}
|
||||
}
|
||||
|
||||
private func savePlain() {
|
||||
PythonBridge.call(command: "set_plain_subtitle_config", arguments: plainConfig.arguments()) { _, error in
|
||||
DispatchQueue.main.async { errorMessage = error }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// O estilo das legendas dinâmicas, no formato que a tela edita e o bridge
|
||||
@@ -402,6 +462,69 @@ struct CaptionStyleConfig {
|
||||
}
|
||||
}
|
||||
|
||||
struct PlainSubtitleConfig {
|
||||
var font: String
|
||||
var fontSize: Double
|
||||
var fontColor: String
|
||||
var maxWords: Double
|
||||
var positionY: Double
|
||||
var uppercase: Bool
|
||||
var keepPunctuation: Bool
|
||||
var textScale: Double
|
||||
|
||||
static let defaults = PlainSubtitleConfig(
|
||||
font: "Helvetica Neue",
|
||||
fontSize: 82,
|
||||
fontColor: "1 1 1 1",
|
||||
maxWords: 7,
|
||||
positionY: -820,
|
||||
uppercase: false,
|
||||
keepPunctuation: true,
|
||||
textScale: 2.0
|
||||
)
|
||||
|
||||
init(from json: [String: Any]) {
|
||||
let d = PlainSubtitleConfig.defaults
|
||||
self.init(
|
||||
font: json["font"] as? String ?? d.font,
|
||||
fontSize: (json["font_size"] as? NSNumber)?.doubleValue ?? d.fontSize,
|
||||
fontColor: json["font_color"] as? String ?? d.fontColor,
|
||||
maxWords: (json["max_words"] as? NSNumber)?.doubleValue ?? d.maxWords,
|
||||
positionY: (json["position_y"] as? NSNumber)?.doubleValue ?? d.positionY,
|
||||
uppercase: json["uppercase"] as? Bool ?? d.uppercase,
|
||||
keepPunctuation: json["keep_punctuation"] as? Bool ?? d.keepPunctuation,
|
||||
textScale: (json["text_scale"] as? NSNumber)?.doubleValue ?? d.textScale
|
||||
)
|
||||
}
|
||||
|
||||
init(
|
||||
font: String, fontSize: Double, fontColor: String, maxWords: Double,
|
||||
positionY: Double, uppercase: Bool, keepPunctuation: Bool, textScale: Double
|
||||
) {
|
||||
self.font = font
|
||||
self.fontSize = fontSize
|
||||
self.fontColor = fontColor
|
||||
self.maxWords = maxWords
|
||||
self.positionY = positionY
|
||||
self.uppercase = uppercase
|
||||
self.keepPunctuation = keepPunctuation
|
||||
self.textScale = textScale
|
||||
}
|
||||
|
||||
func arguments() -> [String: Any] {
|
||||
[
|
||||
"font": font,
|
||||
"font_size": Int(fontSize),
|
||||
"font_color": fontColor,
|
||||
"max_words": Int(maxWords),
|
||||
"position_y": positionY,
|
||||
"uppercase": uppercase,
|
||||
"keep_punctuation": keepPunctuation,
|
||||
"text_scale": textScale,
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
extension Color {
|
||||
/// Parses an FCPXML "R G B A" space-separated 0-1 string into a Color.
|
||||
init(rgbaString: String) {
|
||||
|
||||
@@ -15,6 +15,11 @@ struct ModelDownloadView: View {
|
||||
@State private var hfTokenText: String = ""
|
||||
@State private var numSpeakersText: String = ""
|
||||
@State private var language: String = "auto"
|
||||
@State private var acousticsAvailable: Bool?
|
||||
@State private var acousticsMessage: String = ""
|
||||
@State private var isInstallingAcoustics = false
|
||||
@State private var acousticsInstallLog: String = ""
|
||||
@State private var acousticsInstallError: String?
|
||||
|
||||
private let languages: [(String, String)] = [
|
||||
("auto", "Detectar automaticamente"),
|
||||
@@ -33,6 +38,7 @@ struct ModelDownloadView: View {
|
||||
var body: some View {
|
||||
Form {
|
||||
storageSection
|
||||
acousticsSection
|
||||
diarizationSection
|
||||
if let errorMessage {
|
||||
Section {
|
||||
@@ -65,7 +71,7 @@ struct ModelDownloadView: View {
|
||||
}
|
||||
}
|
||||
.formStyle(.grouped)
|
||||
.task { await refresh() }
|
||||
.task { await refresh(); checkAcoustics() }
|
||||
}
|
||||
|
||||
// MARK: - Transcription language
|
||||
@@ -95,6 +101,98 @@ struct ModelDownloadView: View {
|
||||
PythonBridge.call(command: "set_language", arguments: ["language": code]) { _, _ in }
|
||||
}
|
||||
|
||||
// MARK: - Acoustic analysis (librosa)
|
||||
|
||||
/// A ênfase de voz (pitch/energia) precisa do `librosa`, que é uma
|
||||
/// dependência opcional — sem ela `layers.acoustics` vem `false` na
|
||||
/// análise e a decisão de zoom fica sem base real. Antes disso só dava
|
||||
/// pra descobrir lendo o JSON exportado; agora o app já diz e resolve.
|
||||
private var acousticsSection: some View {
|
||||
Section {
|
||||
VStack(alignment: .leading, spacing: 10) {
|
||||
if let acousticsAvailable {
|
||||
Label(
|
||||
acousticsMessage.isEmpty
|
||||
? (acousticsAvailable ? "Disponível" : "Indisponível")
|
||||
: acousticsMessage,
|
||||
systemImage: acousticsAvailable ? "checkmark.circle.fill" : "exclamationmark.triangle.fill"
|
||||
)
|
||||
.font(.caption)
|
||||
.foregroundStyle(acousticsAvailable ? Color.green : Color.orange)
|
||||
} else {
|
||||
Label("Verificando…", systemImage: "hourglass")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
|
||||
if acousticsAvailable == false {
|
||||
Button {
|
||||
installAcoustics()
|
||||
} label: {
|
||||
if isInstallingAcoustics {
|
||||
HStack { ProgressView().controlSize(.small); Text("Instalando…") }
|
||||
} else {
|
||||
Label("Instalar (uv sync --all-extras)", systemImage: "arrow.down.circle")
|
||||
}
|
||||
}
|
||||
.disabled(isInstallingAcoustics)
|
||||
|
||||
if !acousticsInstallLog.isEmpty {
|
||||
ScrollView {
|
||||
Text(acousticsInstallLog)
|
||||
.font(.system(.caption2, design: .monospaced))
|
||||
.foregroundStyle(.secondary)
|
||||
.frame(maxWidth: .infinity, alignment: .leading)
|
||||
}
|
||||
.frame(height: 90)
|
||||
.background(RoundedRectangle(cornerRadius: 6).fill(Color.secondary.opacity(0.06)))
|
||||
}
|
||||
if let acousticsInstallError {
|
||||
Label(acousticsInstallError, systemImage: "xmark.circle.fill")
|
||||
.font(.caption).foregroundStyle(.red)
|
||||
}
|
||||
}
|
||||
}
|
||||
} header: {
|
||||
Text("Análise Acústica (zoom por voz)")
|
||||
} footer: {
|
||||
Text("Mede a energia e o tom de voz de verdade, para os candidatos a zoom da edição por voz. Sem isso, a análise ainda transcreve e decide cortes pelo texto — só o zoom fica sem base acústica.")
|
||||
.font(.caption)
|
||||
.foregroundStyle(.secondary)
|
||||
}
|
||||
}
|
||||
|
||||
private func checkAcoustics() {
|
||||
PythonBridge.call(command: "acoustics_capability") { result, err in
|
||||
DispatchQueue.main.async {
|
||||
guard let result, result["ok"] as? Bool == true else { return }
|
||||
acousticsAvailable = result["available"] as? Bool
|
||||
acousticsMessage = result["message"] as? String ?? ""
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func installAcoustics() {
|
||||
isInstallingAcoustics = true
|
||||
acousticsInstallLog = ""
|
||||
acousticsInstallError = nil
|
||||
// --all-extras, não só "intelligence": `uv sync` substitui o
|
||||
// ambiente pelos extras pedidos em vez de somar, então um sync
|
||||
// parcial aqui derrubaria dev/transcribe/diarização já instalados.
|
||||
PythonBridge.runUV(arguments: ["sync", "--all-extras"]) { line in
|
||||
DispatchQueue.main.async {
|
||||
acousticsInstallLog += (acousticsInstallLog.isEmpty ? "" : "\n") + line
|
||||
}
|
||||
} completion: { code, err in
|
||||
DispatchQueue.main.async {
|
||||
isInstallingAcoustics = false
|
||||
if code != 0 {
|
||||
acousticsInstallError = err ?? "Falha ao instalar."
|
||||
}
|
||||
checkAcoustics()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Diarization
|
||||
|
||||
private var diarizationSection: some View {
|
||||
|
||||
@@ -116,6 +116,145 @@ struct ZoomClip: Identifiable {
|
||||
}
|
||||
}
|
||||
|
||||
/// One word inside a phrase, with the acoustics that justify an emphasis.
|
||||
struct ReviewWord: Identifiable {
|
||||
let id: Int
|
||||
let text: String
|
||||
let start: Double
|
||||
let end: Double
|
||||
let energy: Double
|
||||
let emphasis: Double
|
||||
|
||||
init(id: Int, json: [String: Any]) {
|
||||
self.id = id
|
||||
text = json["text"] as? String ?? ""
|
||||
start = json["start"] as? Double ?? 0
|
||||
end = json["end"] as? Double ?? 0
|
||||
energy = json["energy"] as? Double ?? 0
|
||||
emphasis = json["emphasis"] as? Double ?? 0
|
||||
}
|
||||
}
|
||||
|
||||
/// A phrase in the review step — one spoken line plus the decision made about
|
||||
/// it. Mirrors `fcpxml/phrase_review.py`; `emphasis` is 0–3 and everything
|
||||
/// mutable here is what the editor is allowed to change.
|
||||
struct ReviewPhrase: Identifiable {
|
||||
let id: Int
|
||||
let start: Double
|
||||
let end: Double
|
||||
var trimStart: Double
|
||||
var trimEnd: Double
|
||||
var text: String
|
||||
let speaker: String
|
||||
var active: Bool
|
||||
var emphasis: Int
|
||||
var track: String
|
||||
let peakEmphasis: Double
|
||||
let emotion: String
|
||||
let emotionConfidence: Double
|
||||
let takeBoundary: Bool
|
||||
let gapBefore: Double
|
||||
let reason: String
|
||||
let words: [ReviewWord]
|
||||
|
||||
static let trackScript = "roteiro"
|
||||
static let trackBackstage = "bastidor"
|
||||
|
||||
/// Delivery emotion as the analysis names it, in the user's language plus a
|
||||
/// glyph — the label alone is too easy to skim past in a dense list.
|
||||
static func emotionLabel(_ emotion: String) -> (String, String) {
|
||||
switch emotion {
|
||||
case "excited": return ("Empolgado", "flame")
|
||||
case "tense": return ("Tenso", "bolt")
|
||||
case "calm": return ("Calmo", "leaf")
|
||||
case "reflective": return ("Reflexivo", "moon")
|
||||
default: return ("Neutro", "circle")
|
||||
}
|
||||
}
|
||||
|
||||
init(json: [String: Any]) {
|
||||
id = json["index"] as? Int ?? 0
|
||||
start = json["start"] as? Double ?? 0
|
||||
end = json["end"] as? Double ?? 0
|
||||
trimStart = json["trim_start"] as? Double ?? (json["start"] as? Double ?? 0)
|
||||
trimEnd = json["trim_end"] as? Double ?? (json["end"] as? Double ?? 0)
|
||||
text = json["text"] as? String ?? ""
|
||||
speaker = json["speaker"] as? String ?? ""
|
||||
active = json["active"] as? Bool ?? true
|
||||
emphasis = json["emphasis"] as? Int ?? 0
|
||||
track = json["track"] as? String ?? ReviewPhrase.trackScript
|
||||
peakEmphasis = json["peak_emphasis"] as? Double ?? 0
|
||||
emotion = json["emotion"] as? String ?? "neutral"
|
||||
emotionConfidence = json["emotion_confidence"] as? Double ?? 0
|
||||
takeBoundary = json["take_boundary"] as? Bool ?? false
|
||||
gapBefore = json["gap_before"] as? Double ?? 0
|
||||
reason = json["reason"] as? String ?? ""
|
||||
words = (json["words"] as? [[String: Any]] ?? [])
|
||||
.enumerated().map { ReviewWord(id: $0.offset, json: $0.element) }
|
||||
}
|
||||
|
||||
var asJSON: [String: Any] {
|
||||
[
|
||||
"index": id,
|
||||
"start": start,
|
||||
"end": end,
|
||||
"trim_start": trimStart,
|
||||
"trim_end": trimEnd,
|
||||
"text": text,
|
||||
"speaker": speaker,
|
||||
"active": active,
|
||||
"emphasis": emphasis,
|
||||
"track": track,
|
||||
"reason": reason,
|
||||
]
|
||||
}
|
||||
|
||||
var isBackstage: Bool { track == ReviewPhrase.trackBackstage }
|
||||
var isTrimmed: Bool { trimStart > start + 0.001 || trimEnd < end - 0.001 }
|
||||
var timecode: String {
|
||||
String(format: "%02d:%02d", Int(start) / 60, Int(start) % 60)
|
||||
}
|
||||
|
||||
/// The word boundaries a trim handle is allowed to land on.
|
||||
func snap(_ time: Double, edge: TrimEdge) -> Double {
|
||||
let boundaries = words.map { edge == .start ? $0.start : $0.end }.filter { $0 > 0 }
|
||||
guard let nearest = boundaries.min(by: { abs($0 - time) < abs($1 - time) }) else {
|
||||
return time
|
||||
}
|
||||
return nearest
|
||||
}
|
||||
}
|
||||
|
||||
enum TrimEdge { case start, end }
|
||||
|
||||
/// A punch-in the editor placed by hand over an arbitrary range, next to the
|
||||
/// whole-phrase zoom that an emphasis level produces. It stores only *when* —
|
||||
/// the scale and the ramp come from the Voice Analysis settings at render time.
|
||||
struct ManualZoom: Identifiable {
|
||||
let id = UUID()
|
||||
var start: Double
|
||||
var end: Double
|
||||
|
||||
/// Below this a punch-in has no room to ramp in and back out; the writer
|
||||
/// rejects the window, so offering it would place nothing.
|
||||
static let minimumDuration: Double = 0.4
|
||||
|
||||
init(start: Double, end: Double) {
|
||||
self.start = start
|
||||
self.end = end
|
||||
}
|
||||
|
||||
init?(json: [String: Any]) {
|
||||
guard let start = json["start"] as? Double, let end = json["end"] as? Double,
|
||||
end - start >= ManualZoom.minimumDuration
|
||||
else { return nil }
|
||||
self.start = start
|
||||
self.end = end
|
||||
}
|
||||
|
||||
var asJSON: [String: Any] { ["start": start, "end": end] }
|
||||
}
|
||||
|
||||
struct ZoomSegment: Identifiable {
|
||||
let id: Int
|
||||
let start: Double
|
||||
|
||||
@@ -0,0 +1,429 @@
|
||||
import AVFoundation
|
||||
import Combine
|
||||
import Foundation
|
||||
|
||||
/// State behind the wizard's emphasis-review step.
|
||||
///
|
||||
/// Holds the phrases, the selection, and the player — together, because they
|
||||
/// are one thing to the user: clicking a phrase moves the playhead, playing
|
||||
/// moves the selection, and skipping a removed line only works if whoever owns
|
||||
/// playback also knows which lines are removed.
|
||||
///
|
||||
/// The preview deliberately plays the *original* media and jumps over whatever
|
||||
/// the edit removes, instead of rendering a cut first. Rendering to check a
|
||||
/// toggle would put minutes between a decision and its result; jumping gives
|
||||
/// the same reading instantly, and the real cut is generated later from the
|
||||
/// exact same phrase list.
|
||||
@MainActor
|
||||
final class PhraseReviewModel: ObservableObject {
|
||||
@Published var phrases: [ReviewPhrase] = []
|
||||
@Published var selection: Int?
|
||||
@Published var isLoading = false
|
||||
@Published var errorMessage: String?
|
||||
@Published var currentTime: Double = 0
|
||||
@Published var isPlaying = false
|
||||
@Published var pixelsPerSecond: Double = 40
|
||||
@Published var skipRemoved = true
|
||||
@Published var zooms: [ManualZoom] = []
|
||||
/// In/out the editor dragged on the timeline, in source seconds.
|
||||
@Published var rangeStart: Double?
|
||||
@Published var rangeEnd: Double?
|
||||
/// Aspect ratio of the footage as recorded.
|
||||
@Published var videoAspect: Double = 16.0 / 9.0
|
||||
/// Aspect ratio the project delivers in, read from the .fcpxml. It is
|
||||
/// routinely *not* the footage's: these takes are shot horizontal and
|
||||
/// delivered vertical, so previewing the raw frame would show a crop the
|
||||
/// audience never sees — and the emphasis decisions are about what lands on
|
||||
/// screen. Nil until the project is known.
|
||||
@Published var projectAspect: Double?
|
||||
/// Whether the preview crops to the delivery frame. On by default whenever
|
||||
/// the two aspects disagree.
|
||||
@Published var matchProjectFraming = true
|
||||
|
||||
/// What the preview should actually draw.
|
||||
var previewAspect: Double {
|
||||
guard matchProjectFraming, let projectAspect else { return videoAspect }
|
||||
return projectAspect
|
||||
}
|
||||
|
||||
/// True when the delivery frame differs enough from the footage that the
|
||||
/// preview is showing a crop rather than the whole take.
|
||||
var isCropping: Bool {
|
||||
guard matchProjectFraming, let projectAspect else { return false }
|
||||
return abs(projectAspect - videoAspect) > 0.01
|
||||
}
|
||||
|
||||
private(set) var source = ""
|
||||
private(set) var sourcePath = ""
|
||||
private(set) var duration: Double = 0
|
||||
private(set) var speakers: [String] = []
|
||||
private(set) var emotionAvailable = false
|
||||
private(set) var player: AVPlayer?
|
||||
|
||||
private var voiceTimelinePath = ""
|
||||
private var timeObserver: Any?
|
||||
private var playbackLimit: Double?
|
||||
|
||||
let minPixelsPerSecond: Double = 8
|
||||
let maxPixelsPerSecond: Double = 400
|
||||
|
||||
deinit {
|
||||
if let timeObserver, let player {
|
||||
player.removeTimeObserver(timeObserver)
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Carregar
|
||||
|
||||
/// Builds the review from the voice timeline plus whatever the AI decided.
|
||||
/// A review saved on a previous visit wins — see `cmd_build_phrase_review`.
|
||||
/// Reads the delivery format from the project so the preview can frame the
|
||||
/// take the way it will actually be seen.
|
||||
func loadProjectFormat(projectPath: String) {
|
||||
PythonBridge.call(command: "inspect", arguments: ["path": projectPath]) { [weak self] result, _ in
|
||||
Task { @MainActor in
|
||||
guard let self,
|
||||
let timelines = result?["timelines"] as? [[String: Any]],
|
||||
let first = timelines.first,
|
||||
let width = first["width"] as? Int, let height = first["height"] as? Int,
|
||||
width > 0, height > 0
|
||||
else { return }
|
||||
self.projectAspect = Double(width) / Double(height)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func load(voiceTimelinePath: String, decisionsJSON: String,
|
||||
outputFolder: String? = nil, mediaFolder: String? = nil) {
|
||||
self.voiceTimelinePath = voiceTimelinePath
|
||||
isLoading = true
|
||||
errorMessage = nil
|
||||
|
||||
var arguments: [String: Any] = ["voice_timeline": voiceTimelinePath]
|
||||
if let outputFolder { arguments["output_dir"] = outputFolder }
|
||||
if let mediaFolder { arguments["media_dir"] = mediaFolder }
|
||||
if let data = decisionsJSON.data(using: .utf8),
|
||||
let parsed = try? JSONSerialization.jsonObject(with: data) {
|
||||
arguments["actions"] = parsed
|
||||
}
|
||||
|
||||
PythonBridge.call(command: "build_phrase_review", arguments: arguments) { [weak self] result, error in
|
||||
Task { @MainActor in
|
||||
guard let self else { return }
|
||||
self.isLoading = false
|
||||
if let error {
|
||||
self.errorMessage = error
|
||||
return
|
||||
}
|
||||
guard let result, result["ok"] as? Bool == true else {
|
||||
self.errorMessage = result?["error"] as? String ?? "Não foi possível montar a revisão."
|
||||
return
|
||||
}
|
||||
self.apply(result)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func apply(_ result: [String: Any]) {
|
||||
source = result["source"] as? String ?? ""
|
||||
// The timeline JSON stores only the media's file name; the bridge
|
||||
// resolves it to something openable (see phrase_review.resolve_source).
|
||||
sourcePath = result["source_path"] as? String ?? ""
|
||||
duration = result["duration"] as? Double ?? 0
|
||||
speakers = result["speakers"] as? [String] ?? []
|
||||
emotionAvailable = result["emotion_available"] as? Bool ?? false
|
||||
phrases = (result["phrases"] as? [[String: Any]] ?? []).map { ReviewPhrase(json: $0) }
|
||||
zooms = (result["zooms"] as? [[String: Any]] ?? []).compactMap { ManualZoom(json: $0) }
|
||||
selection = phrases.first?.id
|
||||
if let errors = result["errors"] as? [String], !errors.isEmpty {
|
||||
errorMessage = "A IA mandou \(errors.count) decisão(ões) que não deu para ler — o resto foi aplicado."
|
||||
}
|
||||
preparePlayer()
|
||||
}
|
||||
|
||||
/// Point the preview at a media file the user chose by hand — the way out
|
||||
/// when the footage moved somewhere the automatic lookup can't reach.
|
||||
func useMedia(at path: String) {
|
||||
sourcePath = path
|
||||
preparePlayer()
|
||||
}
|
||||
|
||||
private func preparePlayer() {
|
||||
guard !sourcePath.isEmpty, FileManager.default.fileExists(atPath: sourcePath) else {
|
||||
player = nil
|
||||
return
|
||||
}
|
||||
if let timeObserver, let player {
|
||||
player.removeTimeObserver(timeObserver)
|
||||
self.timeObserver = nil
|
||||
}
|
||||
let asset = AVURLAsset(url: URL(fileURLWithPath: sourcePath))
|
||||
let player = AVPlayer(playerItem: AVPlayerItem(asset: asset))
|
||||
self.player = player
|
||||
readAspect(from: asset)
|
||||
// 60 Hz: the same observer drives the playhead *and* decides when to
|
||||
// jump a removed stretch, so its period is the worst-case amount of cut
|
||||
// material that can be heard before the skip lands. At 20 Hz that was an
|
||||
// audible blip on every join.
|
||||
let interval = CMTime(seconds: 1.0 / 60.0, preferredTimescale: 600)
|
||||
timeObserver = player.addPeriodicTimeObserver(forInterval: interval, queue: .main) { [weak self] time in
|
||||
Task { @MainActor in
|
||||
self?.tick(time.seconds)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The displayed aspect ratio, honouring the rotation the camera recorded.
|
||||
/// A phone take is stored 1920×1080 with a 90° transform: reading
|
||||
/// `naturalSize` alone would call a vertical video horizontal.
|
||||
private func readAspect(from asset: AVURLAsset) {
|
||||
Task { [weak self] in
|
||||
guard let track = try? await asset.loadTracks(withMediaType: .video).first,
|
||||
let size = try? await track.load(.naturalSize),
|
||||
let transform = try? await track.load(.preferredTransform)
|
||||
else { return }
|
||||
let displayed = size.applying(transform)
|
||||
let width = abs(displayed.width), height = abs(displayed.height)
|
||||
guard width > 0, height > 0 else { return }
|
||||
await MainActor.run { self?.videoAspect = width / height }
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Reprodução
|
||||
|
||||
private func tick(_ time: Double) {
|
||||
currentTime = time
|
||||
guard isPlaying else { return }
|
||||
|
||||
// Playing a single phrase or a marked range stops at its out point
|
||||
// instead of running on into the rest of the take.
|
||||
if let limit = playbackLimit, time >= limit {
|
||||
pause()
|
||||
seek(to: limit)
|
||||
return
|
||||
}
|
||||
|
||||
if skipRemoved, let jump = nextKeptTime(after: time), jump > time {
|
||||
seek(to: jump)
|
||||
}
|
||||
if let phrase = phrase(at: time), selection != phrase.id {
|
||||
selection = phrase.id
|
||||
}
|
||||
}
|
||||
|
||||
/// Where playback should resume when `time` lands on removed material.
|
||||
/// Returns nil when the time is on material that survives.
|
||||
func nextKeptTime(after time: Double) -> Double? {
|
||||
for phrase in phrases where time >= phrase.start - 0.001 && time < phrase.end {
|
||||
if !phrase.active { return phrase.end }
|
||||
if time < phrase.trimStart { return phrase.trimStart }
|
||||
if time >= phrase.trimEnd { return phrase.end }
|
||||
return nil
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func togglePlay() {
|
||||
if isPlaying {
|
||||
pause()
|
||||
} else {
|
||||
playbackLimit = nil
|
||||
play()
|
||||
}
|
||||
}
|
||||
|
||||
private func play() {
|
||||
guard let player else { return }
|
||||
if skipRemoved, let jump = nextKeptTime(after: currentTime) { seek(to: jump) }
|
||||
player.play()
|
||||
isPlaying = true
|
||||
}
|
||||
|
||||
func pause() {
|
||||
player?.pause()
|
||||
isPlaying = false
|
||||
playbackLimit = nil
|
||||
}
|
||||
|
||||
/// Play exactly one span and stop — how a cut is judged: in context, at
|
||||
/// speed, without hunting for the out point by hand.
|
||||
func playRange(from start: Double, to end: Double) {
|
||||
guard end > start else { return }
|
||||
seek(to: start)
|
||||
playbackLimit = end
|
||||
player?.play()
|
||||
isPlaying = true
|
||||
}
|
||||
|
||||
func playSelectedPhrase() {
|
||||
guard let selection, let phrase = phrases.first(where: { $0.id == selection })
|
||||
else { return }
|
||||
playRange(from: phrase.active ? phrase.trimStart : phrase.start,
|
||||
to: phrase.active ? phrase.trimEnd : phrase.end)
|
||||
}
|
||||
|
||||
func seek(to time: Double) {
|
||||
currentTime = max(0, time)
|
||||
player?.seek(to: CMTime(seconds: max(0, time), preferredTimescale: 600),
|
||||
toleranceBefore: .zero, toleranceAfter: .zero)
|
||||
}
|
||||
|
||||
/// Move the playhead to a phrase and select it.
|
||||
func goTo(phraseID: Int) {
|
||||
guard let phrase = phrases.first(where: { $0.id == phraseID }) else { return }
|
||||
selection = phraseID
|
||||
seek(to: phrase.active ? phrase.trimStart : phrase.start)
|
||||
}
|
||||
|
||||
func phrase(at time: Double) -> ReviewPhrase? {
|
||||
phrases.first { time >= $0.start && time < $0.end }
|
||||
}
|
||||
|
||||
func selectNeighbour(_ delta: Int) {
|
||||
guard let selection, let index = phrases.firstIndex(where: { $0.id == selection }) else {
|
||||
if let first = phrases.first { goTo(phraseID: first.id) }
|
||||
return
|
||||
}
|
||||
let next = min(max(0, index + delta), phrases.count - 1)
|
||||
goTo(phraseID: phrases[next].id)
|
||||
}
|
||||
|
||||
// MARK: - Edições
|
||||
|
||||
private func update(_ id: Int, _ change: (inout ReviewPhrase) -> Void) {
|
||||
guard let index = phrases.firstIndex(where: { $0.id == id }) else { return }
|
||||
change(&phrases[index])
|
||||
}
|
||||
|
||||
func setEmphasis(_ level: Int, for id: Int) {
|
||||
update(id) { $0.emphasis = min(3, max(0, level)) }
|
||||
}
|
||||
|
||||
func toggleActive(_ id: Int) {
|
||||
update(id) { $0.active.toggle() }
|
||||
}
|
||||
|
||||
func setTrack(_ track: String, for id: Int) {
|
||||
update(id) { $0.track = track }
|
||||
}
|
||||
|
||||
func setText(_ text: String, for id: Int) {
|
||||
update(id) { $0.text = text }
|
||||
}
|
||||
|
||||
/// Trim a phrase's head or tail, landing on a word boundary.
|
||||
/// A trim that would swallow the whole line is refused — deactivating the
|
||||
/// phrase is the way to remove it, and doing it by accident with a drag
|
||||
/// would lose the emphasis decision along with the line.
|
||||
func trim(_ id: Int, edge: TrimEdge, to time: Double) {
|
||||
update(id) { phrase in
|
||||
let snapped = phrase.snap(time, edge: edge)
|
||||
switch edge {
|
||||
case .start:
|
||||
let value = min(max(phrase.start, snapped), phrase.trimEnd - 0.1)
|
||||
if value < phrase.trimEnd { phrase.trimStart = value }
|
||||
case .end:
|
||||
let value = max(min(phrase.end, snapped), phrase.trimStart + 0.1)
|
||||
if value > phrase.trimStart { phrase.trimEnd = value }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func resetTrim(_ id: Int) {
|
||||
update(id) { $0.trimStart = $0.start; $0.trimEnd = $0.end }
|
||||
}
|
||||
|
||||
/// Trim everything before/after a given word — the text-first way to cut,
|
||||
/// since the editor reads the line and points at where it should begin.
|
||||
func trimToWord(_ word: ReviewWord, edge: TrimEdge, in id: Int) {
|
||||
trim(id, edge: edge, to: edge == .start ? word.start : word.end)
|
||||
}
|
||||
|
||||
// MARK: - Trecho marcado e zooms
|
||||
|
||||
var hasRange: Bool {
|
||||
guard let rangeStart, let rangeEnd else { return false }
|
||||
return rangeEnd - rangeStart >= ManualZoom.minimumDuration
|
||||
}
|
||||
|
||||
var rangeSpan: (start: Double, end: Double)? {
|
||||
guard let rangeStart, let rangeEnd, rangeEnd > rangeStart else { return nil }
|
||||
return (rangeStart, rangeEnd)
|
||||
}
|
||||
|
||||
func setRange(from start: Double, to end: Double) {
|
||||
rangeStart = min(start, end)
|
||||
rangeEnd = max(start, end)
|
||||
}
|
||||
|
||||
func clearRange() {
|
||||
rangeStart = nil
|
||||
rangeEnd = nil
|
||||
}
|
||||
|
||||
/// Add a punch-in over the marked range. Scale and ramp are not stored:
|
||||
/// they come from the "Análise de Voz" settings when the edit is rendered,
|
||||
/// so changing the look there restyles every zoom at once.
|
||||
func addZoomForRange() {
|
||||
guard let span = rangeSpan, span.end - span.start >= ManualZoom.minimumDuration
|
||||
else { return }
|
||||
zooms.append(ManualZoom(start: span.start, end: span.end))
|
||||
zooms.sort { $0.start < $1.start }
|
||||
clearRange()
|
||||
}
|
||||
|
||||
func addZoomForPhrase(_ id: Int) {
|
||||
guard let phrase = phrases.first(where: { $0.id == id }) else { return }
|
||||
zooms.append(ManualZoom(start: phrase.trimStart, end: phrase.trimEnd))
|
||||
zooms.sort { $0.start < $1.start }
|
||||
}
|
||||
|
||||
func removeZoom(_ id: UUID) {
|
||||
zooms.removeAll { $0.id == id }
|
||||
}
|
||||
|
||||
func zoom(at time: Double) -> ManualZoom? {
|
||||
zooms.first { time >= $0.start && time <= $0.end }
|
||||
}
|
||||
|
||||
func setEmphasisForAll(_ level: Int) {
|
||||
for index in phrases.indices where phrases[index].active {
|
||||
phrases[index].emphasis = level
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Resumo e gravação
|
||||
|
||||
var emphasisCount: Int { phrases.filter { $0.active && $0.emphasis >= 1 }.count }
|
||||
var removedCount: Int { phrases.filter { !$0.active }.count }
|
||||
var keptDuration: Double {
|
||||
phrases.filter { $0.active }.reduce(0) { $0 + ($1.trimEnd - $1.trimStart) }
|
||||
}
|
||||
|
||||
/// Persists the edited review plus the actions derived from it. Called when
|
||||
/// the wizard advances — the render itself happens in the next step.
|
||||
func save(completion: @escaping (String?) -> Void) {
|
||||
guard !voiceTimelinePath.isEmpty, !phrases.isEmpty else {
|
||||
completion(nil)
|
||||
return
|
||||
}
|
||||
let arguments: [String: Any] = [
|
||||
"voice_timeline": voiceTimelinePath,
|
||||
"source": source,
|
||||
"duration": duration,
|
||||
"speakers": speakers,
|
||||
"phrases": phrases.map { $0.asJSON },
|
||||
"zooms": zooms.map { $0.asJSON },
|
||||
]
|
||||
PythonBridge.call(command: "save_phrase_review", arguments: arguments) { result, error in
|
||||
Task { @MainActor in
|
||||
if let error {
|
||||
completion(nil)
|
||||
_ = error
|
||||
return
|
||||
}
|
||||
completion(result?["review_path"] as? String)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,413 @@
|
||||
import AVFoundation
|
||||
import SwiftUI
|
||||
|
||||
/// The video surface, as a plain `AVPlayerLayer` in an `NSView`.
|
||||
///
|
||||
/// AVKit's `VideoPlayer` would be the obvious choice and is a trap here: this
|
||||
/// app is built by invoking `swiftc` directly (see `MacApp/build_app.sh`), and
|
||||
/// `_AVKit_SwiftUI` aborts at launch instantiating its generic metadata under
|
||||
/// that build. A player layer needs only AVFoundation, which links cleanly —
|
||||
/// and the transport controls live in the timeline's own toolbar anyway, so
|
||||
/// nothing is lost by dropping AVKit's chrome.
|
||||
private struct PlayerSurface: NSViewRepresentable {
|
||||
let player: AVPlayer
|
||||
/// When true the frame is filled and cropped instead of letterboxed — used
|
||||
/// to preview horizontal footage inside a vertical delivery frame.
|
||||
var fills: Bool
|
||||
|
||||
func makeNSView(context: Context) -> PlayerLayerView {
|
||||
let view = PlayerLayerView()
|
||||
view.player = player
|
||||
view.fills = fills
|
||||
return view
|
||||
}
|
||||
|
||||
func updateNSView(_ view: PlayerLayerView, context: Context) {
|
||||
if view.player !== player { view.player = player }
|
||||
view.fills = fills
|
||||
}
|
||||
}
|
||||
|
||||
final class PlayerLayerView: NSView {
|
||||
private let playerLayer = AVPlayerLayer()
|
||||
|
||||
var player: AVPlayer? {
|
||||
get { playerLayer.player }
|
||||
set { playerLayer.player = newValue }
|
||||
}
|
||||
|
||||
var fills: Bool = false {
|
||||
didSet { playerLayer.videoGravity = fills ? .resizeAspectFill : .resizeAspect }
|
||||
}
|
||||
|
||||
override init(frame frameRect: NSRect) {
|
||||
super.init(frame: frameRect)
|
||||
wantsLayer = true
|
||||
layer = CALayer()
|
||||
layer?.backgroundColor = NSColor.black.cgColor
|
||||
playerLayer.videoGravity = .resizeAspect
|
||||
layer?.addSublayer(playerLayer)
|
||||
}
|
||||
|
||||
required init?(coder: NSCoder) {
|
||||
super.init(coder: coder)
|
||||
wantsLayer = true
|
||||
layer = CALayer()
|
||||
playerLayer.videoGravity = .resizeAspect
|
||||
layer?.addSublayer(playerLayer)
|
||||
}
|
||||
|
||||
override func layout() {
|
||||
super.layout()
|
||||
playerLayer.frame = bounds
|
||||
}
|
||||
}
|
||||
|
||||
/// The wizard's emphasis-review step, laid out like an editing room: preview on
|
||||
/// top, timeline across the bottom, and the script as an inspector down the
|
||||
/// right side.
|
||||
///
|
||||
/// The arrangement is the point. Every decision here is about a *sentence*, so
|
||||
/// the same phrase has to be legible in all three places at once — a block on
|
||||
/// the timeline, a line of text in the inspector, and a moment in the preview.
|
||||
/// Selecting in any one of them selects in the other two.
|
||||
struct PhraseReviewView: View {
|
||||
@ObservedObject var model: PhraseReviewModel
|
||||
|
||||
var body: some View {
|
||||
VSplitView {
|
||||
HSplitView {
|
||||
previewPane
|
||||
.frame(minWidth: 320, idealWidth: 640)
|
||||
inspectorPane
|
||||
.frame(minWidth: 300, idealWidth: 360, maxWidth: 520)
|
||||
}
|
||||
.frame(minHeight: 240)
|
||||
|
||||
TimelineTracksView(model: model)
|
||||
.frame(minHeight: 190, idealHeight: 210)
|
||||
}
|
||||
.overlay { if model.isLoading { loadingOverlay } }
|
||||
.focusable()
|
||||
.onKeyPress(.space) { model.togglePlay(); return .handled }
|
||||
.onKeyPress(.return) { model.playSelectedPhrase(); return .handled }
|
||||
.onKeyPress(.leftArrow) { model.selectNeighbour(-1); return .handled }
|
||||
.onKeyPress(.rightArrow) { model.selectNeighbour(1); return .handled }
|
||||
.onKeyPress(characters: .decimalDigits) { press in
|
||||
guard let level = Int(press.characters), (0...3).contains(level),
|
||||
let selection = model.selection else { return .ignored }
|
||||
model.setEmphasis(level, for: selection)
|
||||
return .handled
|
||||
}
|
||||
}
|
||||
|
||||
private var loadingOverlay: some View {
|
||||
ZStack {
|
||||
Color(nsColor: .windowBackgroundColor).opacity(0.85)
|
||||
VStack(spacing: 10) {
|
||||
ProgressView()
|
||||
Text("Montando a revisão…").font(.callout).foregroundStyle(.secondary)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Preview
|
||||
|
||||
private var previewPane: some View {
|
||||
VStack(spacing: 0) {
|
||||
if let player = model.player {
|
||||
// The footage here is usually vertical. Sizing the surface to
|
||||
// the take's own aspect keeps a 9:16 frame as tall as the pane
|
||||
// allows instead of shrinking it to fit a horizontal box.
|
||||
// Framed to what the project delivers, not to what the camera
|
||||
// recorded: these takes are shot horizontal and cut vertical,
|
||||
// so the raw frame would show material the audience never sees.
|
||||
ZStack {
|
||||
Color.black
|
||||
PlayerSurface(player: player, fills: model.isCropping)
|
||||
.aspectRatio(model.previewAspect, contentMode: .fit)
|
||||
.clipped()
|
||||
}
|
||||
.overlay(alignment: .topTrailing) { framingBadge }
|
||||
} else {
|
||||
ZStack {
|
||||
Color.black.opacity(0.85)
|
||||
VStack(spacing: 10) {
|
||||
Image(systemName: "film.stack")
|
||||
.font(.system(size: 28)).foregroundStyle(.secondary)
|
||||
Text(model.source.isEmpty
|
||||
? "A análise de voz não registrou qual mídia foi usada."
|
||||
: "Não achei \(model.source) na pasta do projeto.")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
Text("A revisão funciona igual sem o preview — ele só ajuda a conferir o corte.")
|
||||
.font(.caption).foregroundStyle(.tertiary)
|
||||
Button("Localizar a mídia…") { pickMedia() }
|
||||
.buttonStyle(.bordered)
|
||||
}
|
||||
.multilineTextAlignment(.center)
|
||||
.padding(.horizontal, 24)
|
||||
}
|
||||
}
|
||||
Divider()
|
||||
summaryBar
|
||||
}
|
||||
}
|
||||
|
||||
private var summaryBar: some View {
|
||||
HStack(spacing: 16) {
|
||||
summaryItem("text.quote", "\(model.phrases.count) frases")
|
||||
summaryItem("sparkles", "\(model.emphasisCount) com ênfase")
|
||||
summaryItem("scissors", "\(model.removedCount) fora do corte")
|
||||
summaryItem("clock", durationLabel(model.keptDuration))
|
||||
if !model.zooms.isEmpty {
|
||||
summaryItem("plus.magnifyingglass", "\(model.zooms.count) zooms")
|
||||
}
|
||||
Spacer()
|
||||
if let phrase = selectedPhrase, !phrase.reason.isEmpty {
|
||||
Label(phrase.reason, systemImage: "brain")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
.lineLimit(1).truncationMode(.tail)
|
||||
}
|
||||
}
|
||||
.padding(.horizontal, 14)
|
||||
.padding(.vertical, 8)
|
||||
}
|
||||
|
||||
private func summaryItem(_ icon: String, _ text: String) -> some View {
|
||||
Label(text, systemImage: icon).font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
|
||||
private func durationLabel(_ seconds: Double) -> String {
|
||||
String(format: "%02d:%02d finais", Int(seconds) / 60, Int(seconds) % 60)
|
||||
}
|
||||
|
||||
/// Says which frame is on screen, and lets the editor flip to the raw take.
|
||||
/// Without it a centred crop looks like the footage itself, and someone
|
||||
/// would judge framing on an approximation without knowing it.
|
||||
@ViewBuilder
|
||||
private var framingBadge: some View {
|
||||
if model.projectAspect != nil, abs((model.projectAspect ?? 0) - model.videoAspect) > 0.01 {
|
||||
Button {
|
||||
model.matchProjectFraming.toggle()
|
||||
} label: {
|
||||
Label(model.matchProjectFraming ? "Enquadramento do projeto" : "Mídia original",
|
||||
systemImage: model.matchProjectFraming ? "crop" : "rectangle.expand.vertical")
|
||||
.font(.caption2)
|
||||
}
|
||||
.buttonStyle(.borderless)
|
||||
.padding(6)
|
||||
.background(Capsule().fill(.black.opacity(0.45)))
|
||||
.foregroundStyle(.white)
|
||||
.padding(8)
|
||||
.help("A fonte é horizontal e o projeto é vertical — o preview mostra o corte central aproximado. O enquadramento real de cada clipe vem do Final Cut.")
|
||||
}
|
||||
}
|
||||
|
||||
private func pickMedia() {
|
||||
let panel = NSOpenPanel()
|
||||
panel.canChooseFiles = true
|
||||
panel.canChooseDirectories = false
|
||||
panel.allowsMultipleSelection = false
|
||||
panel.prompt = "Usar esta mídia"
|
||||
panel.message = model.source.isEmpty
|
||||
? "Escolha o arquivo de vídeo desta gravação."
|
||||
: "Escolha onde está \(model.source)."
|
||||
if panel.runModal() == .OK, let url = panel.url {
|
||||
model.useMedia(at: url.path)
|
||||
}
|
||||
}
|
||||
|
||||
private var selectedPhrase: ReviewPhrase? {
|
||||
guard let selection = model.selection else { return nil }
|
||||
return model.phrases.first { $0.id == selection }
|
||||
}
|
||||
|
||||
// MARK: - Inspector de frases
|
||||
|
||||
private var inspectorPane: some View {
|
||||
VStack(spacing: 0) {
|
||||
inspectorHeader
|
||||
Divider()
|
||||
List(selection: $model.selection) {
|
||||
ForEach($model.phrases) { $phrase in
|
||||
PhraseRow(phrase: $phrase, model: model)
|
||||
.tag(phrase.id)
|
||||
}
|
||||
}
|
||||
.listStyle(.inset)
|
||||
.onChange(of: model.selection) { _, newValue in
|
||||
if let newValue { model.goTo(phraseID: newValue) }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var inspectorHeader: some View {
|
||||
VStack(alignment: .leading, spacing: 6) {
|
||||
Text("Frases").font(.headline)
|
||||
Text("Só as frases com ênfase recebem zoom e legenda dinâmica. O resto fica com legenda comum.")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
if !model.emotionAvailable {
|
||||
Label("Emoção da fala não foi detectada nesta análise — ligue em Avançado → Análise de Voz e refaça o passo 3.",
|
||||
systemImage: "waveform.path.ecg")
|
||||
.font(.caption2).foregroundStyle(.secondary)
|
||||
}
|
||||
HStack(spacing: 8) {
|
||||
Button("Limpar ênfases") { model.setEmphasisForAll(0) }
|
||||
.buttonStyle(.link).font(.caption)
|
||||
Spacer()
|
||||
Text("0–3 no teclado · ← → navega")
|
||||
.font(.caption2).foregroundStyle(.secondary)
|
||||
}
|
||||
}
|
||||
.padding(12)
|
||||
}
|
||||
}
|
||||
|
||||
/// One phrase in the inspector: the line as it will be said, plus every
|
||||
/// decision attached to it. Kept in one row on purpose — jumping to a separate
|
||||
/// detail pane to set a toggle would double the clicks on the most repeated
|
||||
/// action in the screen.
|
||||
private struct PhraseRow: View {
|
||||
@Binding var phrase: ReviewPhrase
|
||||
@ObservedObject var model: PhraseReviewModel
|
||||
@State private var isEditing = false
|
||||
|
||||
var body: some View {
|
||||
VStack(alignment: .leading, spacing: 6) {
|
||||
HStack(spacing: 6) {
|
||||
Text(phrase.timecode)
|
||||
.font(.system(.caption2, design: .monospaced))
|
||||
.foregroundStyle(.secondary)
|
||||
if phrase.takeBoundary {
|
||||
Image(systemName: "scissors.badge.ellipsis")
|
||||
.font(.caption2).foregroundStyle(.orange)
|
||||
.help("Nova tomada começa aqui")
|
||||
}
|
||||
if phrase.isTrimmed {
|
||||
Image(systemName: "arrow.left.and.right.square")
|
||||
.font(.caption2).foregroundStyle(.blue)
|
||||
.help("Frase cortada nas pontas")
|
||||
}
|
||||
if model.emotionAvailable {
|
||||
emotionChip
|
||||
}
|
||||
Spacer()
|
||||
Toggle("", isOn: $phrase.active)
|
||||
.toggleStyle(.switch)
|
||||
.controlSize(.mini)
|
||||
.labelsHidden()
|
||||
.help(phrase.active ? "No corte" : "Fora do corte")
|
||||
}
|
||||
|
||||
if isEditing {
|
||||
TextField("Texto da frase", text: $phrase.text, axis: .vertical)
|
||||
.textFieldStyle(.roundedBorder)
|
||||
.font(.callout)
|
||||
.onSubmit { isEditing = false }
|
||||
} else {
|
||||
Text(phrase.text.isEmpty ? "(sem texto)" : phrase.text)
|
||||
.font(.callout)
|
||||
.foregroundStyle(phrase.active ? .primary : .secondary)
|
||||
.strikethrough(!phrase.active)
|
||||
.onTapGesture(count: 2) { isEditing = true }
|
||||
}
|
||||
|
||||
HStack(spacing: 8) {
|
||||
Picker("", selection: $phrase.emphasis) {
|
||||
ForEach(0..<4, id: \.self) { level in
|
||||
Text(EmphasisPalette.label(level)).tag(level)
|
||||
}
|
||||
}
|
||||
.pickerStyle(.segmented)
|
||||
.controlSize(.mini)
|
||||
.labelsHidden()
|
||||
.disabled(!phrase.active)
|
||||
|
||||
Picker("", selection: $phrase.track) {
|
||||
Text("Roteiro").tag(ReviewPhrase.trackScript)
|
||||
Text("Bastidor").tag(ReviewPhrase.trackBackstage)
|
||||
}
|
||||
.pickerStyle(.menu)
|
||||
.controlSize(.mini)
|
||||
.labelsHidden()
|
||||
.frame(width: 92)
|
||||
}
|
||||
|
||||
if model.selection == phrase.id && !phrase.words.isEmpty {
|
||||
wordTrimmer
|
||||
}
|
||||
}
|
||||
.padding(.vertical, 4)
|
||||
.opacity(phrase.active ? 1 : 0.55)
|
||||
}
|
||||
|
||||
/// The delivery emotion the acoustics suggest. Shown faded below its own
|
||||
/// confidence: a guess the analysis is unsure about should not compete for
|
||||
/// attention with the emphasis decision, which is the point of the row.
|
||||
private var emotionChip: some View {
|
||||
let (label, icon) = ReviewPhrase.emotionLabel(phrase.emotion)
|
||||
return Label(label, systemImage: icon)
|
||||
.font(.caption2)
|
||||
.padding(.horizontal, 5)
|
||||
.padding(.vertical, 1)
|
||||
.background(
|
||||
Capsule().fill(Color.secondary.opacity(0.12))
|
||||
)
|
||||
.foregroundStyle(phrase.emotionConfidence >= 0.5 ? .secondary : .tertiary)
|
||||
.help("Emoção da entrega: \(label) — confiança \(Int(phrase.emotionConfidence * 100))%")
|
||||
}
|
||||
|
||||
/// Trimming by pointing at the transcript: click a word to start the phrase
|
||||
/// there, option-click to end it there. Same edit as dragging the block's
|
||||
/// edge on the timeline, but reachable while reading the line.
|
||||
private var wordTrimmer: some View {
|
||||
VStack(alignment: .leading, spacing: 4) {
|
||||
HStack(spacing: 4) {
|
||||
Text("Cortar pelas palavras").font(.caption2).foregroundStyle(.secondary)
|
||||
Spacer()
|
||||
if phrase.isTrimmed {
|
||||
Button("Inteira") { model.resetTrim(phrase.id) }
|
||||
.buttonStyle(.link).font(.caption2)
|
||||
}
|
||||
}
|
||||
FlowWords(words: phrase.words, phrase: phrase) { word, edge in
|
||||
model.trimToWord(word, edge: edge, in: phrase.id)
|
||||
}
|
||||
Text("Clique = começa aqui · ⌥clique = termina aqui")
|
||||
.font(.caption2).foregroundStyle(.tertiary)
|
||||
}
|
||||
.padding(.top, 2)
|
||||
}
|
||||
}
|
||||
|
||||
/// The phrase's words as wrapping chips, dimmed where they fall outside the trim.
|
||||
private struct FlowWords: View {
|
||||
let words: [ReviewWord]
|
||||
let phrase: ReviewPhrase
|
||||
let onTrim: (ReviewWord, TrimEdge) -> Void
|
||||
|
||||
var body: some View {
|
||||
// A LazyVGrid with adaptive columns wraps chips without a custom layout;
|
||||
// phrases are short enough that the slight raggedness beats the cost of
|
||||
// hand-rolling a flow layout here.
|
||||
LazyVGrid(columns: [GridItem(.adaptive(minimum: 44), spacing: 3)],
|
||||
alignment: .leading, spacing: 3) {
|
||||
ForEach(words) { word in
|
||||
let kept = word.start >= phrase.trimStart - 0.001 && word.end <= phrase.trimEnd + 0.001
|
||||
Text(word.text)
|
||||
.font(.caption2)
|
||||
.padding(.horizontal, 4)
|
||||
.padding(.vertical, 2)
|
||||
.background(
|
||||
RoundedRectangle(cornerRadius: 3)
|
||||
.fill(kept ? Color.accentColor.opacity(0.12) : Color.secondary.opacity(0.08))
|
||||
)
|
||||
.foregroundStyle(kept ? .primary : .secondary)
|
||||
.strikethrough(!kept)
|
||||
.onTapGesture {
|
||||
onTrim(word, NSEvent.modifierFlags.contains(.option) ? .end : .start)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -44,12 +44,26 @@ enum PythonBridge {
|
||||
return ["python3", scriptURL.path]
|
||||
}
|
||||
|
||||
/// `admin/models_api.py` lives outside `code/`, but its dependencies
|
||||
/// (`pyproject.toml`, `.venv`) live inside it. `uv run` picks the
|
||||
/// environment from the process's cwd, not from the script path — so
|
||||
/// running with cwd at the repo root made `uv` create/use a second,
|
||||
/// empty `.venv` there, silently ignoring everything installed into
|
||||
/// `code/.venv` (this cost a real debugging session: librosa/pyannote
|
||||
/// installed successfully but the app kept reporting them missing).
|
||||
/// Every `uv run` must share the same cwd as `uv sync` to see the same
|
||||
/// environment.
|
||||
static var workingDirectory: URL {
|
||||
projectRoot
|
||||
codeDirectory
|
||||
}
|
||||
|
||||
/// Directory containing `pyproject.toml` — where `uv sync` must run from.
|
||||
static var codeDirectory: URL {
|
||||
projectRoot.appendingPathComponent("code")
|
||||
}
|
||||
|
||||
/// Locate `uv` on PATH or in common install locations.
|
||||
private static func findUV() -> String? {
|
||||
static func findUV() -> String? {
|
||||
if let onPath = which("uv") { return onPath }
|
||||
let candidates = [
|
||||
"/usr/local/bin/uv",
|
||||
@@ -148,6 +162,59 @@ enum PythonBridge {
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - uv sync (installing optional extras, e.g. acoustic analysis)
|
||||
|
||||
/// Runs `uv <arguments>` from `codeDirectory` (where `pyproject.toml`
|
||||
/// lives), streaming each output line as plain text — used for
|
||||
/// `sync --extra intelligence` so "Modelos" can install the librosa
|
||||
/// extra without the user opening a terminal.
|
||||
static func runUV(arguments: [String],
|
||||
onLine: @escaping (String) -> Void,
|
||||
completion: @escaping (Int, String?) -> Void) {
|
||||
guard let uv = findUV() else {
|
||||
completion(1, "uv não encontrado. Instale com: curl -LsSf https://astral.sh/uv/install.sh | sh")
|
||||
return
|
||||
}
|
||||
let process = Process()
|
||||
process.executableURL = URL(fileURLWithPath: "/usr/bin/env")
|
||||
process.arguments = [uv] + arguments
|
||||
process.currentDirectoryURL = codeDirectory
|
||||
|
||||
let pipe = Pipe()
|
||||
process.standardOutput = pipe
|
||||
process.standardError = pipe
|
||||
|
||||
var buffer = ""
|
||||
let lock = NSLock()
|
||||
pipe.fileHandleForReading.readabilityHandler = { handle in
|
||||
let data = handle.availableData
|
||||
guard !data.isEmpty, let s = String(data: data, encoding: .utf8) else { return }
|
||||
lock.lock()
|
||||
buffer += s
|
||||
let parts = buffer.split(separator: "\n", omittingEmptySubsequences: false)
|
||||
buffer = String(parts.last ?? "")
|
||||
let lines = parts.dropLast()
|
||||
lock.unlock()
|
||||
for line in lines where !line.isEmpty { onLine(String(line)) }
|
||||
}
|
||||
|
||||
process.terminationHandler = { p in
|
||||
pipe.fileHandleForReading.readabilityHandler = nil
|
||||
lock.lock()
|
||||
let last = buffer.trimmingCharacters(in: .whitespacesAndNewlines)
|
||||
buffer = ""
|
||||
lock.unlock()
|
||||
if !last.isEmpty { onLine(last) }
|
||||
completion(Int(p.terminationStatus), p.terminationStatus == 0 ? nil : "uv sync terminou com erro (código \(p.terminationStatus)).")
|
||||
}
|
||||
|
||||
do {
|
||||
try process.run()
|
||||
} catch {
|
||||
completion(1, error.localizedDescription)
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Convenience: single JSON result
|
||||
|
||||
/// Runs a command and delivers the first parsed JSON document as the result.
|
||||
|
||||
@@ -0,0 +1,506 @@
|
||||
import SwiftUI
|
||||
|
||||
/// Colors shared by the timeline and the inspector, so a block and its row in
|
||||
/// the list always read as the same thing.
|
||||
enum EmphasisPalette {
|
||||
static func color(_ level: Int) -> Color {
|
||||
switch level {
|
||||
case 1: return Color.blue
|
||||
case 2: return Color.orange
|
||||
case 3: return Color.pink
|
||||
default: return Color.secondary
|
||||
}
|
||||
}
|
||||
|
||||
static func label(_ level: Int) -> String {
|
||||
switch level {
|
||||
case 1: return "Leve"
|
||||
case 2: return "Média"
|
||||
case 3: return "Forte"
|
||||
default: return "Sem"
|
||||
}
|
||||
}
|
||||
|
||||
static func speakerColor(_ speaker: String, among speakers: [String]) -> Color {
|
||||
let palette: [Color] = [.teal, .purple, .green, .indigo, .brown, .cyan]
|
||||
guard let index = speakers.firstIndex(of: speaker) else { return .gray }
|
||||
return palette[index % palette.count]
|
||||
}
|
||||
}
|
||||
|
||||
/// The timeline strip: four stacked tracks over one shared time axis.
|
||||
///
|
||||
/// Phrases are laid out as real views rather than drawn into a Canvas, because
|
||||
/// every one of them is a target — click to select, drag its edge to trim,
|
||||
/// right-click to change emphasis. The dense per-word energy track *is* a
|
||||
/// Canvas: it has thousands of bars and nothing to hit.
|
||||
struct TimelineTracksView: View {
|
||||
@ObservedObject var model: PhraseReviewModel
|
||||
|
||||
private let rulerHeight: CGFloat = 18
|
||||
private let phraseHeight: CGFloat = 46
|
||||
private let energyHeight: CGFloat = 34
|
||||
private let stripHeight: CGFloat = 12
|
||||
private let handleWidth: CGFloat = 8
|
||||
|
||||
private let gutterWidth: CGFloat = 92
|
||||
private let trackSpacing: CGFloat = 4
|
||||
|
||||
private var pps: CGFloat { CGFloat(model.pixelsPerSecond) }
|
||||
private var contentWidth: CGFloat { max(320, CGFloat(model.duration) * pps) }
|
||||
|
||||
/// Name, icon and height of each lane, in the order they stack. The gutter
|
||||
/// and the tracks are built from this one list so a label can never drift
|
||||
/// off the lane it names.
|
||||
private var lanes: [(label: String, icon: String, height: CGFloat)] {
|
||||
[
|
||||
("", "", rulerHeight),
|
||||
("Zooms", "plus.magnifyingglass", stripHeight + 6),
|
||||
("Frases", "text.quote", phraseHeight),
|
||||
("Energia", "waveform", energyHeight),
|
||||
("Emoção", "face.smiling", stripHeight),
|
||||
("Locutor", "person.wave.2", stripHeight),
|
||||
("Roteiro", "list.bullet.rectangle", stripHeight),
|
||||
]
|
||||
}
|
||||
|
||||
var body: some View {
|
||||
VStack(spacing: 0) {
|
||||
toolbar
|
||||
Divider()
|
||||
HStack(alignment: .top, spacing: 0) {
|
||||
gutter
|
||||
Divider()
|
||||
timelineScroller
|
||||
}
|
||||
}
|
||||
.background(Color(nsColor: .underPageBackgroundColor))
|
||||
}
|
||||
|
||||
/// Fixed column naming each lane. Without it the stripes are six colours
|
||||
/// with no way to tell which one is emotion and which one is the speaker.
|
||||
private var gutter: some View {
|
||||
VStack(alignment: .leading, spacing: trackSpacing) {
|
||||
ForEach(lanes.indices, id: \.self) { index in
|
||||
let lane = lanes[index]
|
||||
HStack(spacing: 4) {
|
||||
if !lane.icon.isEmpty {
|
||||
Image(systemName: lane.icon).font(.system(size: 9))
|
||||
}
|
||||
Text(lane.label).font(.system(size: 10))
|
||||
Spacer(minLength: 0)
|
||||
}
|
||||
.foregroundStyle(.secondary)
|
||||
.frame(height: lane.height, alignment: .center)
|
||||
}
|
||||
}
|
||||
.padding(.horizontal, 8)
|
||||
.padding(.vertical, 8)
|
||||
.frame(width: gutterWidth, alignment: .leading)
|
||||
}
|
||||
|
||||
private var timelineScroller: some View {
|
||||
ScrollViewReader { proxy in
|
||||
ScrollView([.horizontal]) {
|
||||
ZStack(alignment: .topLeading) {
|
||||
VStack(alignment: .leading, spacing: trackSpacing) {
|
||||
ruler
|
||||
zoomTrack
|
||||
phraseTrack
|
||||
energyTrack
|
||||
emotionTrack
|
||||
speakerTrack
|
||||
scriptTrack
|
||||
}
|
||||
.frame(width: contentWidth, alignment: .leading)
|
||||
rangeOverlay
|
||||
playhead
|
||||
// Anchors the auto-scroll: one invisible marker per
|
||||
// phrase, so selecting a line off-screen brings it in.
|
||||
ForEach(model.phrases) { phrase in
|
||||
Color.clear
|
||||
.frame(width: 1, height: 1)
|
||||
.offset(x: x(phrase.start))
|
||||
.id(phrase.id)
|
||||
}
|
||||
}
|
||||
.padding(.vertical, 8)
|
||||
.contentShape(Rectangle())
|
||||
.gesture(scrubGesture)
|
||||
.contextMenu { timelineMenu }
|
||||
}
|
||||
.onChange(of: model.selection) { _, newValue in
|
||||
guard let newValue else { return }
|
||||
withAnimation(.easeOut(duration: 0.2)) {
|
||||
proxy.scrollTo(newValue, anchor: .center)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Barra de controles
|
||||
|
||||
private var toolbar: some View {
|
||||
HStack(spacing: 12) {
|
||||
Button {
|
||||
model.togglePlay()
|
||||
} label: {
|
||||
Image(systemName: model.isPlaying ? "pause.fill" : "play.fill")
|
||||
}
|
||||
.buttonStyle(.borderless)
|
||||
.help("Reproduzir (espaço)")
|
||||
.disabled(model.player == nil)
|
||||
|
||||
Text(timecode(model.currentTime))
|
||||
.font(.system(.caption, design: .monospaced))
|
||||
.foregroundStyle(.secondary)
|
||||
|
||||
Button {
|
||||
model.playSelectedPhrase()
|
||||
} label: {
|
||||
Image(systemName: "play.rectangle")
|
||||
}
|
||||
.buttonStyle(.borderless)
|
||||
.help("Tocar só a frase selecionada (⏎)")
|
||||
.disabled(model.player == nil || model.selection == nil)
|
||||
|
||||
Toggle("Pular removidos", isOn: $model.skipRemoved)
|
||||
.toggleStyle(.checkbox)
|
||||
.font(.caption)
|
||||
.help("Durante a reprodução, salta os trechos desativados — mostra como o corte ficou.")
|
||||
|
||||
Button {
|
||||
model.addZoomForRange()
|
||||
} label: {
|
||||
Label("Zoom no trecho", systemImage: "plus.magnifyingglass")
|
||||
}
|
||||
.buttonStyle(.borderless)
|
||||
.font(.caption)
|
||||
.disabled(!model.hasRange)
|
||||
.help("Arraste na timeline para marcar um trecho e crie um zoom nele. A escala vem de Análise de Voz.")
|
||||
|
||||
Spacer()
|
||||
|
||||
legend
|
||||
|
||||
Spacer()
|
||||
|
||||
Image(systemName: "minus.magnifyingglass").foregroundStyle(.secondary)
|
||||
Slider(value: $model.pixelsPerSecond,
|
||||
in: model.minPixelsPerSecond...model.maxPixelsPerSecond)
|
||||
.frame(width: 130)
|
||||
Image(systemName: "plus.magnifyingglass").foregroundStyle(.secondary)
|
||||
}
|
||||
.padding(.horizontal, 12)
|
||||
.padding(.vertical, 8)
|
||||
}
|
||||
|
||||
private var legend: some View {
|
||||
HStack(spacing: 10) {
|
||||
ForEach(0..<4, id: \.self) { level in
|
||||
HStack(spacing: 4) {
|
||||
RoundedRectangle(cornerRadius: 2)
|
||||
.fill(EmphasisPalette.color(level))
|
||||
.frame(width: 10, height: 10)
|
||||
Text(EmphasisPalette.label(level)).font(.caption2)
|
||||
}
|
||||
}
|
||||
}
|
||||
.foregroundStyle(.secondary)
|
||||
}
|
||||
|
||||
// MARK: - Trilhas
|
||||
|
||||
private var ruler: some View {
|
||||
Canvas { context, size in
|
||||
let step = tickStep()
|
||||
var time = 0.0
|
||||
while time <= model.duration {
|
||||
let position = x(time)
|
||||
context.stroke(
|
||||
Path { $0.move(to: CGPoint(x: position, y: size.height - 6))
|
||||
$0.addLine(to: CGPoint(x: position, y: size.height)) },
|
||||
with: .color(.secondary.opacity(0.5))
|
||||
)
|
||||
context.draw(
|
||||
Text(timecode(time)).font(.system(size: 9, design: .monospaced))
|
||||
.foregroundColor(.secondary),
|
||||
at: CGPoint(x: position + 18, y: 6)
|
||||
)
|
||||
time += step
|
||||
}
|
||||
}
|
||||
.frame(width: contentWidth, height: rulerHeight)
|
||||
}
|
||||
|
||||
private var phraseTrack: some View {
|
||||
ZStack(alignment: .topLeading) {
|
||||
RoundedRectangle(cornerRadius: 4)
|
||||
.fill(Color.secondary.opacity(0.06))
|
||||
.frame(width: contentWidth, height: phraseHeight)
|
||||
ForEach(model.phrases) { phrase in
|
||||
phraseBlock(phrase)
|
||||
}
|
||||
}
|
||||
.frame(width: contentWidth, height: phraseHeight, alignment: .topLeading)
|
||||
}
|
||||
|
||||
@ViewBuilder
|
||||
private func phraseBlock(_ phrase: ReviewPhrase) -> some View {
|
||||
let isSelected = model.selection == phrase.id
|
||||
let color = EmphasisPalette.color(phrase.emphasis)
|
||||
let fullWidth = max(2, width(from: phrase.start, to: phrase.end))
|
||||
let keptWidth = max(1, width(from: phrase.trimStart, to: phrase.trimEnd))
|
||||
|
||||
ZStack(alignment: .topLeading) {
|
||||
// The whole line, dim — what is there before the edit.
|
||||
RoundedRectangle(cornerRadius: 4)
|
||||
.fill(color.opacity(phrase.active ? 0.15 : 0.10))
|
||||
.frame(width: fullWidth, height: phraseHeight)
|
||||
|
||||
// What survives: the kept span, drawn solid over it.
|
||||
RoundedRectangle(cornerRadius: 4)
|
||||
.fill(color.opacity(phrase.active ? 0.55 : 0.12))
|
||||
.frame(width: keptWidth, height: phraseHeight)
|
||||
.offset(x: width(from: phrase.start, to: phrase.trimStart))
|
||||
|
||||
Text(phrase.text)
|
||||
.font(.system(size: 10))
|
||||
.lineLimit(2)
|
||||
.padding(.horizontal, 4)
|
||||
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
|
||||
.foregroundStyle(phrase.active ? .primary : .secondary)
|
||||
.strikethrough(!phrase.active)
|
||||
|
||||
RoundedRectangle(cornerRadius: 4)
|
||||
.stroke(isSelected ? Color.accentColor : color.opacity(0.4),
|
||||
lineWidth: isSelected ? 2 : 1)
|
||||
.frame(width: fullWidth, height: phraseHeight)
|
||||
|
||||
if isSelected && phrase.active {
|
||||
trimHandle(phrase, edge: .start)
|
||||
trimHandle(phrase, edge: .end)
|
||||
}
|
||||
}
|
||||
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
|
||||
.offset(x: x(phrase.start))
|
||||
.contentShape(Rectangle())
|
||||
.onTapGesture { model.goTo(phraseID: phrase.id) }
|
||||
.contextMenu { phraseMenu(phrase) }
|
||||
.help(phrase.reason.isEmpty ? phrase.text : "\(phrase.text)\n— \(phrase.reason)")
|
||||
}
|
||||
|
||||
private func trimHandle(_ phrase: ReviewPhrase, edge: TrimEdge) -> some View {
|
||||
let offset = edge == .start
|
||||
? width(from: phrase.start, to: phrase.trimStart)
|
||||
: width(from: phrase.start, to: phrase.trimEnd) - handleWidth
|
||||
return RoundedRectangle(cornerRadius: 2)
|
||||
.fill(Color.accentColor)
|
||||
.frame(width: handleWidth, height: phraseHeight)
|
||||
.offset(x: offset)
|
||||
.gesture(
|
||||
DragGesture(minimumDistance: 1)
|
||||
.onChanged { value in
|
||||
let time = phrase.start + Double((value.location.x) / pps)
|
||||
model.trim(phrase.id, edge: edge, to: time)
|
||||
}
|
||||
)
|
||||
.help(edge == .start ? "Arraste para cortar o começo (pula de palavra em palavra)"
|
||||
: "Arraste para cortar o fim (pula de palavra em palavra)")
|
||||
}
|
||||
|
||||
@ViewBuilder
|
||||
private func phraseMenu(_ phrase: ReviewPhrase) -> some View {
|
||||
Button("Tocar esta frase") {
|
||||
model.goTo(phraseID: phrase.id)
|
||||
model.playSelectedPhrase()
|
||||
}
|
||||
Button(phrase.active ? "Remover do corte" : "Trazer de volta") {
|
||||
model.toggleActive(phrase.id)
|
||||
}
|
||||
Button("Adicionar zoom nesta frase") { model.addZoomForPhrase(phrase.id) }
|
||||
Divider()
|
||||
ForEach(0..<4, id: \.self) { level in
|
||||
Button("Ênfase: \(EmphasisPalette.label(level))") {
|
||||
model.setEmphasis(level, for: phrase.id)
|
||||
}
|
||||
}
|
||||
Divider()
|
||||
Button(phrase.isBackstage ? "Marcar como roteiro" : "Marcar como bastidor") {
|
||||
model.setTrack(phrase.isBackstage ? ReviewPhrase.trackScript : ReviewPhrase.trackBackstage,
|
||||
for: phrase.id)
|
||||
}
|
||||
if phrase.isTrimmed {
|
||||
Divider()
|
||||
Button("Desfazer corte da frase") { model.resetTrim(phrase.id) }
|
||||
}
|
||||
}
|
||||
|
||||
/// Per-word energy/emphasis, straight from the voice timeline — the closest
|
||||
/// thing to a waveform without opening the audio again.
|
||||
private var energyTrack: some View {
|
||||
Canvas { context, size in
|
||||
for phrase in model.phrases {
|
||||
for word in phrase.words {
|
||||
let start = x(word.start)
|
||||
let barWidth = max(1, width(from: word.start, to: word.end) - 1)
|
||||
let height = size.height * CGFloat(max(0.04, word.energy))
|
||||
let rect = CGRect(x: start, y: size.height - height,
|
||||
width: barWidth, height: height)
|
||||
let color = word.emphasis >= 0.65 ? Color.pink
|
||||
: word.emphasis >= 0.45 ? Color.orange
|
||||
: Color.secondary
|
||||
context.fill(Path(rect),
|
||||
with: .color(color.opacity(phrase.active ? 0.6 : 0.2)))
|
||||
}
|
||||
}
|
||||
}
|
||||
.frame(width: contentWidth, height: energyHeight)
|
||||
.background(RoundedRectangle(cornerRadius: 4).fill(Color.secondary.opacity(0.06)))
|
||||
}
|
||||
|
||||
private var speakerTrack: some View {
|
||||
stripTrack { phrase in
|
||||
EmphasisPalette.speakerColor(phrase.speaker, among: model.speakers)
|
||||
}
|
||||
}
|
||||
|
||||
private var scriptTrack: some View {
|
||||
stripTrack { phrase in phrase.isBackstage ? Color.gray : Color.mint }
|
||||
}
|
||||
|
||||
private func stripTrack(_ color: @escaping (ReviewPhrase) -> Color) -> some View {
|
||||
Canvas { context, size in
|
||||
for phrase in model.phrases {
|
||||
let rect = CGRect(x: x(phrase.start), y: 0,
|
||||
width: max(1, width(from: phrase.start, to: phrase.end)),
|
||||
height: size.height)
|
||||
context.fill(Path(roundedRect: rect, cornerRadius: 2),
|
||||
with: .color(color(phrase).opacity(phrase.active ? 0.7 : 0.2)))
|
||||
}
|
||||
}
|
||||
.frame(width: contentWidth, height: stripHeight)
|
||||
}
|
||||
|
||||
private var playhead: some View {
|
||||
Rectangle()
|
||||
.fill(Color.red)
|
||||
.frame(width: 1.5)
|
||||
.offset(x: x(model.currentTime))
|
||||
.allowsHitTesting(false)
|
||||
}
|
||||
|
||||
/// One gesture, two meanings, decided by whether the mouse moved: a click
|
||||
/// parks the playhead, a drag marks in/out. Splitting them across separate
|
||||
/// controls would mean choosing a tool before every action, which is
|
||||
/// exactly the ceremony this screen is meant to avoid.
|
||||
private var scrubGesture: some Gesture {
|
||||
DragGesture(minimumDistance: 0)
|
||||
.onChanged { value in
|
||||
let from = Double(value.startLocation.x / pps)
|
||||
let to = Double(value.location.x / pps)
|
||||
if abs(value.translation.width) > 3 {
|
||||
model.setRange(from: from, to: to)
|
||||
model.seek(to: min(from, to))
|
||||
} else {
|
||||
model.clearRange()
|
||||
model.seek(to: to)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The marked in/out, drawn over every track so the span reads against the
|
||||
/// phrases and the energy at once.
|
||||
private var rangeOverlay: some View {
|
||||
Group {
|
||||
if let span = model.rangeSpan {
|
||||
Rectangle()
|
||||
.fill(Color.accentColor.opacity(0.18))
|
||||
.overlay(Rectangle().stroke(Color.accentColor.opacity(0.6), lineWidth: 1))
|
||||
.frame(width: max(1, width(from: span.start, to: span.end)))
|
||||
.offset(x: x(span.start))
|
||||
.allowsHitTesting(false)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@ViewBuilder
|
||||
private var timelineMenu: some View {
|
||||
if model.hasRange, let span = model.rangeSpan {
|
||||
Button("Adicionar zoom no trecho (\(secondsLabel(span.end - span.start)))") {
|
||||
model.addZoomForRange()
|
||||
}
|
||||
Button("Tocar o trecho") { model.playRange(from: span.start, to: span.end) }
|
||||
Button("Limpar seleção") { model.clearRange() }
|
||||
} else {
|
||||
Text("Arraste na timeline para marcar um trecho")
|
||||
}
|
||||
if let zoom = model.zoom(at: model.currentTime) {
|
||||
Divider()
|
||||
Button("Remover o zoom daqui") { model.removeZoom(zoom.id) }
|
||||
}
|
||||
}
|
||||
|
||||
private func secondsLabel(_ seconds: Double) -> String {
|
||||
String(format: "%.1fs", seconds)
|
||||
}
|
||||
|
||||
/// Punch-ins, on their own lane above the script: they are a second layer
|
||||
/// over the same time, not a property of a phrase.
|
||||
private var zoomTrack: some View {
|
||||
ZStack(alignment: .topLeading) {
|
||||
RoundedRectangle(cornerRadius: 3)
|
||||
.fill(Color.secondary.opacity(0.06))
|
||||
.frame(width: contentWidth, height: stripHeight + 6)
|
||||
ForEach(model.zooms) { zoom in
|
||||
RoundedRectangle(cornerRadius: 3)
|
||||
.fill(Color.yellow.opacity(0.55))
|
||||
.overlay(
|
||||
Image(systemName: "plus.magnifyingglass")
|
||||
.font(.system(size: 8)).foregroundStyle(.black.opacity(0.6))
|
||||
)
|
||||
.frame(width: max(6, width(from: zoom.start, to: zoom.end)),
|
||||
height: stripHeight + 6)
|
||||
.offset(x: x(zoom.start))
|
||||
.help("Zoom marcado — \(secondsLabel(zoom.end - zoom.start)). A escala vem de Análise de Voz.")
|
||||
.contextMenu {
|
||||
Button("Remover este zoom") { model.removeZoom(zoom.id) }
|
||||
}
|
||||
}
|
||||
}
|
||||
.frame(width: contentWidth, height: stripHeight + 6, alignment: .topLeading)
|
||||
}
|
||||
|
||||
/// Delivery emotion per phrase — the fourth signal to read against the text.
|
||||
private var emotionTrack: some View {
|
||||
stripTrack { phrase in
|
||||
switch phrase.emotion {
|
||||
case "excited": return .orange
|
||||
case "tense": return .red
|
||||
case "calm": return .blue
|
||||
case "reflective": return .purple
|
||||
default: return .secondary
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Escala
|
||||
|
||||
private func x(_ time: Double) -> CGFloat { CGFloat(time) * pps }
|
||||
|
||||
private func width(from: Double, to: Double) -> CGFloat {
|
||||
max(0, CGFloat(to - from) * pps)
|
||||
}
|
||||
|
||||
/// Ruler spacing that keeps labels ~80pt apart at any zoom.
|
||||
private func tickStep() -> Double {
|
||||
let candidates: [Double] = [1, 2, 5, 10, 15, 30, 60, 120, 300, 600]
|
||||
let wanted = 80 / Double(pps)
|
||||
return candidates.first { $0 >= wanted } ?? 600
|
||||
}
|
||||
|
||||
private func timecode(_ seconds: Double) -> String {
|
||||
let total = Int(seconds.rounded(.down))
|
||||
return String(format: "%02d:%02d", total / 60, total % 60)
|
||||
}
|
||||
}
|
||||
@@ -271,7 +271,7 @@ struct TranscriptionView: View {
|
||||
Toggle("Marcar o que foi dito na timeline", isOn: $batchMarkers)
|
||||
|
||||
Divider()
|
||||
Toggle("Exportar legendas SRT", isOn: $batchSubtitles)
|
||||
Toggle("Gerar legenda comum (texto editável no FCP)", isOn: $batchSubtitles)
|
||||
|
||||
Divider()
|
||||
batchOptionRow(
|
||||
@@ -793,7 +793,7 @@ struct TranscriptionView: View {
|
||||
if batchFillers { operations.append("remove_filler_words") }
|
||||
if batchPhrases { operations.append("edit_by_transcript") }
|
||||
if batchMarkers { operations.append("transcript_markers") }
|
||||
if batchSubtitles { operations.append("export_srt") }
|
||||
if batchSubtitles { operations.append("generate_plain_subtitles") }
|
||||
// Runs last, on the timing already cut by any earlier steps (see the
|
||||
// "Abrir no Final Cut Pro" fallback chain and exportSubtitles()'s own
|
||||
// preference for `processedPath` — same reasoning).
|
||||
@@ -834,7 +834,7 @@ struct TranscriptionView: View {
|
||||
}
|
||||
let nextPath = result?["path"] as? String ?? currentPath
|
||||
if operation == "remove_silences" { processedPath = nextPath }
|
||||
if operation == "export_srt" { subtitlePaths = result?["paths"] as? [String] ?? [] }
|
||||
if operation == "generate_plain_subtitles" { subtitlePaths = [nextPath] }
|
||||
if operation == "generate_dynamic_subtitles" { dynamicSubtitlesPath = nextPath }
|
||||
processBatchStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder)
|
||||
}
|
||||
|
||||
@@ -23,6 +23,7 @@ struct VoiceAnalysisView: View {
|
||||
} else {
|
||||
energySection
|
||||
emphasisSection
|
||||
zoomSection
|
||||
weightsSection
|
||||
emotionSection
|
||||
resetSection
|
||||
@@ -90,6 +91,44 @@ struct VoiceAnalysisView: View {
|
||||
}
|
||||
}
|
||||
|
||||
private var zoomSection: some View {
|
||||
Section {
|
||||
sliderRow(
|
||||
title: "Zoom na ênfase",
|
||||
value: $config.zoomScale,
|
||||
range: 1.0...3.0,
|
||||
readout: "\(Int(config.zoomScale * 100))%",
|
||||
help: "Fator aplicado nos punch-ins de ênfase. 130% equivale a escala 1,30 no Final Cut."
|
||||
)
|
||||
Picker("Movimento", selection: $config.zoomMode) {
|
||||
Text("Zoom in e out").tag("in_out")
|
||||
Text("Só zoom in").tag("in")
|
||||
Text("Só zoom out").tag("out")
|
||||
}
|
||||
.onChange(of: config.zoomMode) { _, _ in save() }
|
||||
sliderRow(
|
||||
title: "Velocidade do zoom in",
|
||||
value: $config.zoomEaseIn,
|
||||
range: 0.05...2.0,
|
||||
readout: String(format: "%.2fs", config.zoomEaseIn),
|
||||
help: "Duração da entrada do zoom. Menor é mais rápido."
|
||||
)
|
||||
sliderRow(
|
||||
title: "Velocidade do zoom out",
|
||||
value: $config.zoomEaseOut,
|
||||
range: 0.01...2.0,
|
||||
readout: String(format: "%.2fs", config.zoomEaseOut),
|
||||
help: "Duração da saída do zoom. Menor é mais seco."
|
||||
)
|
||||
} header: {
|
||||
Text("Zoom de Ênfase")
|
||||
} footer: {
|
||||
Text("Esses valores viram o padrão para ações de zoom que não trouxerem scale/ease/ease_out no JSON da edição por voz.")
|
||||
.font(.caption)
|
||||
.foregroundStyle(.secondary)
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Emoção
|
||||
|
||||
private var emotionSection: some View {
|
||||
@@ -129,13 +168,14 @@ struct VoiceAnalysisView: View {
|
||||
title: String,
|
||||
value: Binding<Double>,
|
||||
range: ClosedRange<Double> = 0...1,
|
||||
readout: String? = nil,
|
||||
help: String? = nil
|
||||
) -> some View {
|
||||
VStack(alignment: .leading, spacing: 2) {
|
||||
HStack {
|
||||
Text(title)
|
||||
Spacer()
|
||||
Text(String(format: "%.2f", value.wrappedValue))
|
||||
Text(readout ?? String(format: "%.2f", value.wrappedValue))
|
||||
.monospacedDigit()
|
||||
.foregroundStyle(.secondary)
|
||||
}
|
||||
@@ -188,6 +228,10 @@ struct VoiceAnalysisConfig {
|
||||
var weightDuration: Double
|
||||
var emotionEnabled: Bool
|
||||
var emotionSensitivity: Double
|
||||
var zoomScale: Double
|
||||
var zoomMode: String
|
||||
var zoomEaseIn: Double
|
||||
var zoomEaseOut: Double
|
||||
|
||||
static let defaults = VoiceAnalysisConfig(
|
||||
energyThreshold: 0.5,
|
||||
@@ -198,7 +242,11 @@ struct VoiceAnalysisConfig {
|
||||
weightPause: 0.15,
|
||||
weightDuration: 0.10,
|
||||
emotionEnabled: false,
|
||||
emotionSensitivity: 0.5
|
||||
emotionSensitivity: 0.5,
|
||||
zoomScale: 1.30,
|
||||
zoomMode: "in_out",
|
||||
zoomEaseIn: 0.25,
|
||||
zoomEaseOut: 0.04
|
||||
)
|
||||
|
||||
init(
|
||||
@@ -210,7 +258,11 @@ struct VoiceAnalysisConfig {
|
||||
weightPause: Double,
|
||||
weightDuration: Double,
|
||||
emotionEnabled: Bool,
|
||||
emotionSensitivity: Double
|
||||
emotionSensitivity: Double,
|
||||
zoomScale: Double,
|
||||
zoomMode: String,
|
||||
zoomEaseIn: Double,
|
||||
zoomEaseOut: Double
|
||||
) {
|
||||
self.energyThreshold = energyThreshold
|
||||
self.emphasisThreshold = emphasisThreshold
|
||||
@@ -221,6 +273,10 @@ struct VoiceAnalysisConfig {
|
||||
self.weightDuration = weightDuration
|
||||
self.emotionEnabled = emotionEnabled
|
||||
self.emotionSensitivity = emotionSensitivity
|
||||
self.zoomScale = zoomScale
|
||||
self.zoomMode = zoomMode
|
||||
self.zoomEaseIn = zoomEaseIn
|
||||
self.zoomEaseOut = zoomEaseOut
|
||||
}
|
||||
|
||||
/// Lê a resposta do bridge, caindo no padrão para qualquer campo ausente.
|
||||
@@ -236,7 +292,11 @@ struct VoiceAnalysisConfig {
|
||||
weightPause: weights["pause_before"] as? Double ?? defaults.weightPause,
|
||||
weightDuration: weights["duration"] as? Double ?? defaults.weightDuration,
|
||||
emotionEnabled: json["emotion_enabled"] as? Bool ?? defaults.emotionEnabled,
|
||||
emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity
|
||||
emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity,
|
||||
zoomScale: json["zoom_scale"] as? Double ?? defaults.zoomScale,
|
||||
zoomMode: json["zoom_mode"] as? String ?? defaults.zoomMode,
|
||||
zoomEaseIn: json["zoom_ease_in"] as? Double ?? defaults.zoomEaseIn,
|
||||
zoomEaseOut: json["zoom_ease_out"] as? Double ?? defaults.zoomEaseOut
|
||||
)
|
||||
}
|
||||
|
||||
@@ -253,6 +313,10 @@ struct VoiceAnalysisConfig {
|
||||
],
|
||||
"emotion_enabled": emotionEnabled,
|
||||
"emotion_sensitivity": emotionSensitivity,
|
||||
"zoom_scale": zoomScale,
|
||||
"zoom_mode": zoomMode,
|
||||
"zoom_ease_in": zoomEaseIn,
|
||||
"zoom_ease_out": zoomEaseOut,
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,808 @@
|
||||
import SwiftUI
|
||||
import AppKit
|
||||
|
||||
/// Guia passo a passo do fluxo completo: projeto → transcrição → análise de
|
||||
/// voz → copiar para o chat e trazer as decisões → revisar as ênfases →
|
||||
/// processamento final. Existe para que o usuário não precise entender a ordem
|
||||
/// certa de botões espalhados em várias abas — cada etapa só libera a próxima
|
||||
/// quando o passo anterior terminou, e a "ponte" com o chat (que hoje exigia
|
||||
/// sair do app e escolher um arquivo na mão) vira copiar/colar assistido
|
||||
/// dentro da própria tela.
|
||||
enum WizardStep: Int, CaseIterable, Identifiable {
|
||||
case projeto, transcricao, analise, exportarChat, revisar, finalizar, concluido
|
||||
var id: Int { rawValue }
|
||||
|
||||
var titulo: String {
|
||||
switch self {
|
||||
case .projeto: return "Projeto"
|
||||
case .transcricao: return "Transcrever"
|
||||
case .analise: return "Analisar voz"
|
||||
case .exportarChat: return "Decisões da IA"
|
||||
case .revisar: return "Revisar ênfases"
|
||||
case .finalizar: return "Processar"
|
||||
case .concluido: return "Concluído"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct WizardView: View {
|
||||
@State private var step: WizardStep = .projeto
|
||||
|
||||
// Passo 1 — projeto
|
||||
@State private var outputFolder: String?
|
||||
@State private var projectPath: String?
|
||||
@State private var catalog: Catalog?
|
||||
|
||||
// Passo 2 — transcrição
|
||||
@State private var isTranscribing = false
|
||||
@State private var transcribeProgress: Double = 0
|
||||
@State private var transcribeStage = ""
|
||||
@State private var transcribeResults: [TranscriptResult] = []
|
||||
|
||||
// Passo 3 — análise de voz
|
||||
@State private var isAnalyzing = false
|
||||
@State private var voiceTimelinePath: String?
|
||||
@State private var voiceAnalysisMessage = ""
|
||||
@State private var acousticsAvailable: Bool?
|
||||
@State private var showVoiceTimelineReuseAlert = false
|
||||
@State private var existingVoiceTimelinePath: String?
|
||||
|
||||
// Passo 4 — enviar ao chat e trazer as decisões de volta
|
||||
@State private var copiedFeedback = ""
|
||||
@State private var decisionsText = ""
|
||||
@State private var isApplyingDecisions = false
|
||||
@State private var appliedPath: String?
|
||||
@State private var skippedVoiceEdit = false
|
||||
|
||||
// Passo 5 — revisar ênfases
|
||||
@StateObject private var reviewModel = PhraseReviewModel()
|
||||
@State private var reviewLoadedFor: String?
|
||||
@State private var phraseReviewPath: String?
|
||||
|
||||
// Passo 6 — processamento final
|
||||
@State private var finalSilences = true
|
||||
@State private var finalFillers = false
|
||||
@State private var finalSubtitles = true
|
||||
@State private var finalDynamicSubtitles = false
|
||||
@State private var isFinalizing = false
|
||||
@State private var finalStatus = ""
|
||||
@State private var finalPath: String?
|
||||
|
||||
@State private var errorMessage: String?
|
||||
|
||||
var body: some View {
|
||||
VStack(spacing: 0) {
|
||||
stepperHeader
|
||||
.padding(.horizontal, 24)
|
||||
.padding(.top, 20)
|
||||
.padding(.bottom, 16)
|
||||
|
||||
Divider()
|
||||
|
||||
// A revisão é uma sala de edição, não um formulário: ela precisa da
|
||||
// largura toda e rola por conta própria (timeline horizontal, lista
|
||||
// vertical). As demais etapas continuam na coluna estreita, que é o
|
||||
// que mantém um passo a passo legível.
|
||||
if step == .revisar {
|
||||
revisarStep
|
||||
} else {
|
||||
ScrollView {
|
||||
VStack(alignment: .leading, spacing: 18) {
|
||||
if let errorMessage, !errorMessage.isEmpty {
|
||||
Label(errorMessage, systemImage: "exclamationmark.triangle.fill")
|
||||
.foregroundStyle(.red)
|
||||
.padding(.top, 4)
|
||||
}
|
||||
content
|
||||
}
|
||||
.padding(24)
|
||||
.frame(maxWidth: 640, alignment: .leading)
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
|
||||
Divider()
|
||||
navFooter
|
||||
.padding(.horizontal, 24)
|
||||
.padding(.vertical, 16)
|
||||
}
|
||||
.task {
|
||||
loadProjectConfig()
|
||||
await loadCatalog()
|
||||
}
|
||||
.alert("Análise de voz já existe", isPresented: $showVoiceTimelineReuseAlert) {
|
||||
Button("Usar existente") {
|
||||
if let existingVoiceTimelinePath {
|
||||
voiceTimelinePath = existingVoiceTimelinePath
|
||||
voiceAnalysisMessage = "Reaproveitando análise existente: \(existingVoiceTimelinePath)"
|
||||
}
|
||||
}
|
||||
Button("Reprocessar") {
|
||||
analyzeVoice(forceReprocess: true)
|
||||
}
|
||||
Button("Cancelar", role: .cancel) {}
|
||||
} message: {
|
||||
Text("Já existe um arquivo voice_timeline para este projeto. Quer manter o processamento anterior para ganhar tempo?")
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Cabeçalho com os passos
|
||||
|
||||
private var stepperHeader: some View {
|
||||
HStack(spacing: 6) {
|
||||
ForEach(WizardStep.allCases) { s in
|
||||
HStack(spacing: 6) {
|
||||
ZStack {
|
||||
Circle()
|
||||
.fill(colorFor(s))
|
||||
.frame(width: 24, height: 24)
|
||||
if s.rawValue < step.rawValue {
|
||||
Image(systemName: "checkmark")
|
||||
.font(.caption2.weight(.bold))
|
||||
.foregroundStyle(.white)
|
||||
} else {
|
||||
Text("\(s.rawValue + 1)")
|
||||
.font(.caption2.weight(.bold))
|
||||
.foregroundStyle(s == step ? .white : .secondary)
|
||||
}
|
||||
}
|
||||
Text(s.titulo)
|
||||
.font(.caption)
|
||||
.foregroundStyle(s == step ? .primary : .secondary)
|
||||
.fontWeight(s == step ? .semibold : .regular)
|
||||
}
|
||||
if s != WizardStep.allCases.last {
|
||||
Rectangle()
|
||||
.fill(s.rawValue < step.rawValue ? Color.accentColor : Color.secondary.opacity(0.25))
|
||||
.frame(height: 2)
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func colorFor(_ s: WizardStep) -> Color {
|
||||
if s.rawValue < step.rawValue { return .accentColor }
|
||||
if s == step { return .accentColor }
|
||||
return Color.secondary.opacity(0.25)
|
||||
}
|
||||
|
||||
// MARK: - Conteúdo por etapa
|
||||
|
||||
@ViewBuilder
|
||||
private var content: some View {
|
||||
switch step {
|
||||
case .projeto: projetoStep
|
||||
case .transcricao: transcricaoStep
|
||||
case .analise: analiseStep
|
||||
case .exportarChat: exportarChatStep
|
||||
case .revisar: revisarStep
|
||||
case .finalizar: finalizarStep
|
||||
case .concluido: concluidoStep
|
||||
}
|
||||
}
|
||||
|
||||
private var projetoStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Text("1. Escolha o projeto").font(.title3.weight(.semibold))
|
||||
Text("A pasta é onde tudo o que for gerado nesse fluxo fica salvo. O arquivo é o .fcpxml exportado do Final Cut Pro.")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
|
||||
fieldRow(icon: "folder", label: outputFolder ?? "Nenhuma pasta selecionada", isSet: outputFolder != nil) {
|
||||
pickOutputFolder()
|
||||
}
|
||||
fieldRow(icon: "doc.text", label: projectPath.map { URL(fileURLWithPath: $0).lastPathComponent } ?? "Nenhum arquivo selecionado", isSet: projectPath != nil) {
|
||||
pickProjectFile()
|
||||
}
|
||||
|
||||
if looksLikeGeneratedFile(projectPath) {
|
||||
Label("Esse arquivo parece já ter sido processado por este fluxo (o nome tem um sufixo como \"_voice_edit\" ou \"_silence_removed\"). Rodar o wizard de novo em cima dele reaplica os cortes por cima de cortes já feitos. Selecione o .fcpxml original do Final Cut, a menos que a intenção seja mesmo reprocessar.",
|
||||
systemImage: "exclamationmark.triangle.fill")
|
||||
.font(.caption).foregroundStyle(.orange)
|
||||
}
|
||||
|
||||
if (catalog?.installedCount ?? 0) == 0 {
|
||||
Label("Nenhum modelo de transcrição instalado. Baixe um na aba \"Modelos\" antes de continuar.",
|
||||
systemImage: "exclamationmark.triangle.fill")
|
||||
.font(.caption).foregroundStyle(.orange)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var transcricaoStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Text("2. Transcreva o áudio").font(.title3.weight(.semibold))
|
||||
Text("Roda localmente com o modelo escolhido na aba Modelos. Vira a base de tudo que vem depois — o corte por voz, as legendas, os marcadores.")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
|
||||
Button {
|
||||
startTranscription()
|
||||
} label: {
|
||||
if isTranscribing {
|
||||
HStack { ProgressView().controlSize(.small); Text(transcribeStage.isEmpty ? "Transcrevendo…" : transcribeStage) }
|
||||
.frame(maxWidth: .infinity)
|
||||
} else {
|
||||
Label(transcribeResults.isEmpty ? "Transcrever" : "Transcrever novamente", systemImage: "waveform")
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
.buttonStyle(.borderedProminent)
|
||||
.controlSize(.large)
|
||||
.disabled(isTranscribing || projectPath == nil || outputFolder == nil)
|
||||
|
||||
if isTranscribing {
|
||||
VStack(alignment: .leading, spacing: 6) {
|
||||
ProgressView(value: transcribeProgress)
|
||||
Text("\(Int(transcribeProgress * 100))%").font(.caption).foregroundStyle(.secondary).monospacedDigit()
|
||||
}
|
||||
}
|
||||
|
||||
if !transcribeResults.isEmpty {
|
||||
ForEach(transcribeResults, id: \.media) { r in
|
||||
VStack(alignment: .leading, spacing: 4) {
|
||||
HStack {
|
||||
Image(systemName: "checkmark.circle.fill").foregroundStyle(.green)
|
||||
Text(r.media).font(.body.weight(.medium))
|
||||
Spacer()
|
||||
Text("\(r.language) · \(r.words) palavras").font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
Text(r.preview).font(.caption).foregroundStyle(.secondary).lineLimit(2)
|
||||
}
|
||||
.padding(12)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var analiseStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Text("3. Analise a voz").font(.title3.weight(.semibold))
|
||||
Text("Gera o JSON com transcrição, locutor e intensidade (pitch/energia/ritmo) por palavra — é esse arquivo que o chat lê para decidir o que cortar. Não corta nada sozinho.")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
|
||||
Button {
|
||||
analyzeVoice()
|
||||
} label: {
|
||||
if isAnalyzing {
|
||||
HStack { ProgressView().controlSize(.small); Text("Analisando…") }.frame(maxWidth: .infinity)
|
||||
} else {
|
||||
Label(voiceTimelinePath == nil ? "Analisar voz" : "Analisar novamente", systemImage: "waveform.badge.magnifyingglass")
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
.buttonStyle(.borderedProminent)
|
||||
.controlSize(.large)
|
||||
.disabled(isAnalyzing || projectPath == nil || outputFolder == nil)
|
||||
|
||||
if let voiceTimelinePath {
|
||||
VStack(alignment: .leading, spacing: 6) {
|
||||
Label("Análise pronta", systemImage: "checkmark.circle.fill").foregroundStyle(.green)
|
||||
Text(voiceTimelinePath).font(.caption).foregroundStyle(.secondary).lineLimit(1).truncationMode(.middle)
|
||||
}
|
||||
.padding(12)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
|
||||
|
||||
if acousticsAvailable == false {
|
||||
VStack(alignment: .leading, spacing: 4) {
|
||||
Label("Sem análise acústica real", systemImage: "exclamationmark.triangle.fill")
|
||||
.font(.caption.weight(.semibold)).foregroundStyle(.orange)
|
||||
Text("Falta o componente \"librosa\" — os cortes ainda são decididos pelo texto, mas o chat não vai propor zoom com confiança. Instale em Avançado → Modelos → \"Análise Acústica\", e refaça esta etapa depois.")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
.padding(12)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.orange.opacity(0.08)))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var exportarChatStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Text("4. Envie para o chat decidir os cortes").font(.title3.weight(.semibold))
|
||||
Text("Esta é a única etapa manual que sobra: o julgamento de qual tomada usar, onde dar zoom e o que escrever na tela é feito pela IA numa conversa, não por um botão. Copie abaixo, cole numa sessão do Claude e peça pra rodar a skill \"editar-por-voz\".")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
|
||||
if let voiceTimelinePath {
|
||||
Button {
|
||||
copyForChat(path: voiceTimelinePath)
|
||||
} label: {
|
||||
Label("Copiar para colar no chat", systemImage: "doc.on.clipboard")
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
.buttonStyle(.borderedProminent)
|
||||
.controlSize(.large)
|
||||
|
||||
if !copiedFeedback.isEmpty {
|
||||
Label(copiedFeedback, systemImage: "checkmark.circle.fill")
|
||||
.font(.caption).foregroundStyle(.green)
|
||||
}
|
||||
|
||||
VStack(alignment: .leading, spacing: 8) {
|
||||
Text("O que é copiado").font(.caption.weight(.semibold)).foregroundStyle(.secondary)
|
||||
Text("Um pedido pronto + o conteúdo de \(URL(fileURLWithPath: voiceTimelinePath).lastPathComponent), já formatado. É só colar (⌘V) numa conversa com o Claude.")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
.padding(12)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
|
||||
|
||||
Divider().padding(.vertical, 4)
|
||||
|
||||
Text("Cole aqui o que o chat devolveu").font(.callout.weight(.semibold))
|
||||
Text("Na próxima etapa essas decisões aparecem já marcadas na timeline, frase por frase, para você lapidar.")
|
||||
.font(.caption).foregroundStyle(.secondary)
|
||||
|
||||
HStack {
|
||||
Button {
|
||||
if let s = NSPasteboard.general.string(forType: .string) {
|
||||
decisionsText = s
|
||||
}
|
||||
} label: {
|
||||
Label("Colar da área de transferência", systemImage: "list.clipboard")
|
||||
}
|
||||
Spacer()
|
||||
if !decisionsText.isEmpty {
|
||||
Label(jsonIsValid ? "JSON válido" : "JSON inválido",
|
||||
systemImage: jsonIsValid ? "checkmark.circle.fill" : "xmark.circle.fill")
|
||||
.font(.caption)
|
||||
.foregroundStyle(jsonIsValid ? .green : .red)
|
||||
}
|
||||
}
|
||||
|
||||
TextEditor(text: $decisionsText)
|
||||
.font(.system(.caption, design: .monospaced))
|
||||
.frame(minHeight: 140)
|
||||
.padding(8)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
|
||||
.overlay(RoundedRectangle(cornerRadius: 8).stroke(Color.secondary.opacity(0.2)))
|
||||
|
||||
Button {
|
||||
applyDecisions()
|
||||
} label: {
|
||||
if isApplyingDecisions {
|
||||
HStack { ProgressView().controlSize(.small); Text("Aplicando…") }
|
||||
.frame(maxWidth: .infinity)
|
||||
} else {
|
||||
Label("Aplicar decisões", systemImage: "checkmark.seal")
|
||||
.frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
.buttonStyle(.borderedProminent)
|
||||
.controlSize(.large)
|
||||
.disabled(isApplyingDecisions || !jsonIsValid)
|
||||
|
||||
if let appliedPath {
|
||||
Label("Decisões aplicadas — \(URL(fileURLWithPath: appliedPath).lastPathComponent)",
|
||||
systemImage: "checkmark.circle.fill")
|
||||
.font(.caption).foregroundStyle(.green)
|
||||
}
|
||||
|
||||
Divider()
|
||||
Button("Pular esta etapa (revisar as ênfases direto, sem passar pela IA)") {
|
||||
skippedVoiceEdit = true
|
||||
appliedPath = nil
|
||||
decisionsText = ""
|
||||
}
|
||||
.buttonStyle(.plain)
|
||||
.font(.caption)
|
||||
.foregroundStyle(.secondary)
|
||||
} else {
|
||||
Label("Volte ao passo anterior e rode a análise de voz primeiro.", systemImage: "exclamationmark.triangle.fill")
|
||||
.font(.caption).foregroundStyle(.orange)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Etapa 5 — a sala de edição. Diferente das outras, não é um formulário
|
||||
/// dentro da coluna do assistente: ocupa a janela toda e se carrega sozinha
|
||||
/// na primeira vez que aparece para aquela análise de voz.
|
||||
private var revisarStep: some View {
|
||||
Group {
|
||||
if voiceTimelinePath != nil {
|
||||
PhraseReviewView(model: reviewModel)
|
||||
} else {
|
||||
VStack(spacing: 8) {
|
||||
Label("Volte ao passo 3 e rode a análise de voz primeiro.",
|
||||
systemImage: "exclamationmark.triangle.fill")
|
||||
.foregroundStyle(.orange)
|
||||
}
|
||||
.frame(maxWidth: .infinity, maxHeight: .infinity)
|
||||
}
|
||||
}
|
||||
.onAppear { loadReviewIfNeeded() }
|
||||
}
|
||||
|
||||
private var finalizarStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Text("6. Finalize o corte").font(.title3.weight(.semibold))
|
||||
Text("Últimos passos automáticos, sem decisão envolvida — rodam com os parâmetros já configurados na aba \"Análise de Voz\" / \"Legendas Dinâmicas\".")
|
||||
.font(.callout).foregroundStyle(.secondary)
|
||||
|
||||
Toggle("Remover silêncios do áudio", isOn: $finalSilences)
|
||||
Toggle("Remover palavras de preenchimento", isOn: $finalFillers)
|
||||
Toggle("Gerar legenda comum (texto editável no FCP)", isOn: $finalSubtitles)
|
||||
Toggle("Gerar legendas dinâmicas (estilo configurado na aba própria)", isOn: $finalDynamicSubtitles)
|
||||
|
||||
Button {
|
||||
finalizeProcessing()
|
||||
} label: {
|
||||
if isFinalizing {
|
||||
HStack { ProgressView().controlSize(.small); Text(finalStatus.isEmpty ? "Processando…" : finalStatus) }
|
||||
.frame(maxWidth: .infinity)
|
||||
} else {
|
||||
Label("Processar", systemImage: "play.fill").frame(maxWidth: .infinity)
|
||||
}
|
||||
}
|
||||
.buttonStyle(.borderedProminent)
|
||||
.controlSize(.large)
|
||||
.disabled(isFinalizing || (!finalSilences && !finalFillers && !finalSubtitles && !finalDynamicSubtitles))
|
||||
|
||||
if !finalStatus.isEmpty && !isFinalizing {
|
||||
Text(finalStatus).font(.caption).foregroundStyle(.secondary)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var concluidoStep: some View {
|
||||
VStack(alignment: .leading, spacing: 16) {
|
||||
Label("Concluído", systemImage: "checkmark.seal.fill")
|
||||
.font(.title3.weight(.semibold))
|
||||
.foregroundStyle(.green)
|
||||
if let finalPath {
|
||||
Text(finalPath).font(.caption).foregroundStyle(.secondary).lineLimit(1).truncationMode(.middle)
|
||||
HStack {
|
||||
Button("Abrir no Final Cut Pro") { NSWorkspace.shared.open(URL(fileURLWithPath: finalPath)) }
|
||||
.buttonStyle(.borderedProminent)
|
||||
Button("Mostrar no Finder") {
|
||||
NSWorkspace.shared.activateFileViewerSelecting([URL(fileURLWithPath: finalPath)])
|
||||
}
|
||||
}
|
||||
}
|
||||
Divider().padding(.vertical, 8)
|
||||
Button("Começar outro projeto") { resetWizard() }
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Navegação
|
||||
|
||||
private var navFooter: some View {
|
||||
HStack {
|
||||
if step != .projeto && step != .concluido {
|
||||
Button("Voltar") { goBack() }
|
||||
}
|
||||
Spacer()
|
||||
if step != .concluido {
|
||||
Button(step == .finalizar ? "Concluir" : "Continuar") { goNext() }
|
||||
.buttonStyle(.borderedProminent)
|
||||
.disabled(!canAdvance)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private var canAdvance: Bool {
|
||||
switch step {
|
||||
case .projeto: return outputFolder != nil && projectPath != nil
|
||||
case .transcricao: return !transcribeResults.isEmpty
|
||||
case .analise: return voiceTimelinePath != nil
|
||||
case .exportarChat: return appliedPath != nil || skippedVoiceEdit
|
||||
// Revisar é opcional: a sugestão da IA já é utilizável como veio, então
|
||||
// o botão nunca trava aqui — o passo existe para lapidar, não para
|
||||
// exigir mais uma confirmação.
|
||||
case .revisar: return true
|
||||
case .finalizar: return finalPath != nil && !isFinalizing
|
||||
case .concluido: return false
|
||||
}
|
||||
}
|
||||
|
||||
private func goNext() {
|
||||
guard let next = WizardStep(rawValue: step.rawValue + 1) else { return }
|
||||
// Sair da revisão grava o que foi decidido (e as ações derivadas dela)
|
||||
// ao lado da análise de voz. Nada é renderizado aqui: a etapa 6 é que
|
||||
// lê esse arquivo para dar zoom e legenda dinâmica só nas ênfases.
|
||||
if step == .revisar {
|
||||
reviewModel.save { path in
|
||||
phraseReviewPath = path
|
||||
}
|
||||
}
|
||||
step = next
|
||||
}
|
||||
|
||||
private func goBack() {
|
||||
guard let prev = WizardStep(rawValue: step.rawValue - 1) else { return }
|
||||
step = prev
|
||||
}
|
||||
|
||||
private func resetWizard() {
|
||||
step = .projeto
|
||||
transcribeResults = []
|
||||
voiceTimelinePath = nil
|
||||
voiceAnalysisMessage = ""
|
||||
decisionsText = ""
|
||||
appliedPath = nil
|
||||
skippedVoiceEdit = false
|
||||
reviewLoadedFor = nil
|
||||
phraseReviewPath = nil
|
||||
finalStatus = ""
|
||||
finalPath = nil
|
||||
errorMessage = nil
|
||||
}
|
||||
|
||||
// MARK: - Componentes auxiliares
|
||||
|
||||
@ViewBuilder
|
||||
private func fieldRow(icon: String, label: String, isSet: Bool, action: @escaping () -> Void) -> some View {
|
||||
HStack {
|
||||
Image(systemName: icon).foregroundStyle(isSet ? .primary : .secondary)
|
||||
Text(label).lineLimit(1).truncationMode(.middle).foregroundStyle(isSet ? .primary : .secondary)
|
||||
Spacer()
|
||||
Button("Escolher…", action: action)
|
||||
}
|
||||
.padding(12)
|
||||
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
|
||||
}
|
||||
|
||||
/// Todo output do fluxo carrega um destes sufixos no nome (ver
|
||||
/// `_derived_output` / suffixes usados por `apply_voice_actions`,
|
||||
/// `remove_silences`, `generate_dynamic_subtitles` em
|
||||
/// `admin/models_api.py`). Selecionar um deles como "o projeto" no passo
|
||||
/// 1 é o erro que gerou arquivos como `_voice_edit_voice_edit_...`: os
|
||||
/// cortes de voz assumem timestamps da mídia ORIGINAL, então reaplicá-los
|
||||
/// sobre um arquivo já cortado desloca tudo silenciosamente.
|
||||
private static let generatedSuffixes = [
|
||||
"_voice_edit", "_silence_removed", "_dynamic_subtitles",
|
||||
"_transcript_edit", "_fillers_removed", "_markers",
|
||||
]
|
||||
|
||||
private func looksLikeGeneratedFile(_ path: String?) -> Bool {
|
||||
guard let path else { return false }
|
||||
let stem = URL(fileURLWithPath: path).deletingPathExtension().lastPathComponent
|
||||
return Self.generatedSuffixes.contains { stem.contains($0) }
|
||||
}
|
||||
|
||||
private var jsonIsValid: Bool {
|
||||
guard let data = decisionsText.data(using: .utf8), !decisionsText.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { return false }
|
||||
return (try? JSONSerialization.jsonObject(with: data)) != nil
|
||||
}
|
||||
|
||||
// MARK: - Ações — Python bridge
|
||||
|
||||
private func loadProjectConfig() {
|
||||
PythonBridge.call(command: "project_config") { result, _ in
|
||||
DispatchQueue.main.async {
|
||||
guard let result, result["ok"] as? Bool == true else { return }
|
||||
if let folder = result["folder"] as? String, !folder.isEmpty { outputFolder = folder }
|
||||
if let file = result["file"] as? String, !file.isEmpty { projectPath = file }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func loadCatalog() async {
|
||||
PythonBridge.call(command: "catalog") { result, _ in
|
||||
DispatchQueue.main.async {
|
||||
if let result { catalog = Catalog(json: result) }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func pickOutputFolder() {
|
||||
let panel = NSOpenPanel()
|
||||
panel.canChooseFiles = false
|
||||
panel.canChooseDirectories = true
|
||||
panel.allowsMultipleSelection = false
|
||||
panel.prompt = "Usar esta pasta"
|
||||
panel.message = "Escolha a pasta onde os resultados serão salvos."
|
||||
if panel.runModal() == .OK, let url = panel.url {
|
||||
outputFolder = url.path
|
||||
PythonBridge.call(command: "set_project_config", arguments: ["folder": url.path]) { _, _ in }
|
||||
}
|
||||
}
|
||||
|
||||
private func pickProjectFile() {
|
||||
let panel = NSOpenPanel()
|
||||
panel.canChooseFiles = true
|
||||
panel.canChooseDirectories = false
|
||||
panel.allowsMultipleSelection = false
|
||||
panel.prompt = "Selecionar"
|
||||
panel.message = "Selecione o arquivo (.fcpxml) ou o bundle (.fcpxmld) exportado pelo Final Cut Pro."
|
||||
if panel.runModal() == .OK, let url = panel.url {
|
||||
let ext = url.pathExtension.lowercased()
|
||||
if ext == "fcpxml" || ext == "fcpxmld" || ext == "xml" {
|
||||
projectPath = url.path
|
||||
PythonBridge.call(command: "set_project_config", arguments: ["file": url.path]) { _, _ in }
|
||||
} else {
|
||||
errorMessage = "Selecione um arquivo .fcpxml, .fcpxmld ou .xml do Final Cut Pro."
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func startTranscription() {
|
||||
guard let projectPath, let outputFolder else { return }
|
||||
isTranscribing = true
|
||||
errorMessage = nil
|
||||
transcribeResults = []
|
||||
transcribeProgress = 0
|
||||
PythonBridge.run(command: "transcribe", arguments: ["path": projectPath, "output_dir": outputFolder]) { obj in
|
||||
DispatchQueue.main.async {
|
||||
let type = obj["type"] as? String
|
||||
if type == "progress" {
|
||||
transcribeProgress = (obj["fraction"] as? NSNumber)?.doubleValue ?? 0
|
||||
transcribeStage = obj["stage"] as? String ?? ""
|
||||
} else if type == "error" {
|
||||
errorMessage = obj["message"] as? String ?? "Erro na transcrição."
|
||||
} else if type == "result", let arr = obj["transcripts"] as? [[String: Any]] {
|
||||
transcribeResults = arr.map(TranscriptResult.init)
|
||||
}
|
||||
}
|
||||
} completion: { code, err in
|
||||
DispatchQueue.main.async {
|
||||
isTranscribing = false
|
||||
transcribeProgress = 1
|
||||
if code != 0 && transcribeResults.isEmpty {
|
||||
errorMessage = err ?? "A transcrição falhou."
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func analyzeVoice(forceReprocess: Bool = false) {
|
||||
guard let projectPath, let outputFolder else { return }
|
||||
isAnalyzing = true
|
||||
errorMessage = nil
|
||||
PythonBridge.call(command: "analyze_voice", arguments: [
|
||||
"path": projectPath,
|
||||
"output_dir": outputFolder,
|
||||
"force_reprocess": forceReprocess,
|
||||
]) { result, err in
|
||||
DispatchQueue.main.async {
|
||||
isAnalyzing = false
|
||||
guard result?["ok"] as? Bool == true else {
|
||||
errorMessage = result?["error"] as? String ?? err ?? "Falha ao analisar a voz."
|
||||
return
|
||||
}
|
||||
if result?["reused"] as? Bool == true, !forceReprocess {
|
||||
let timelines = result?["timelines"] as? [String] ?? []
|
||||
existingVoiceTimelinePath = timelines.first ?? extractPath(from: result?["message"] as? String ?? "", marker: "**Timeline JSON**:")
|
||||
showVoiceTimelineReuseAlert = true
|
||||
return
|
||||
}
|
||||
let message = result?["message"] as? String ?? ""
|
||||
voiceAnalysisMessage = message
|
||||
if let path = extractPath(from: message, marker: "**Timeline JSON**:") {
|
||||
voiceTimelinePath = path
|
||||
} else {
|
||||
voiceTimelinePath = nil
|
||||
// ok:true não garante que a análise gerou timeline — se
|
||||
// não houver fala detectável no áudio, o Python volta com
|
||||
// sucesso mas sem "Timeline JSON" na mensagem. Sem isso
|
||||
// aqui, a etapa parecia não fazer nada.
|
||||
errorMessage = "A análise terminou mas não encontrou fala reconhecível no áudio. Mensagem do motor: " + (message.isEmpty ? "(vazia)" : message)
|
||||
}
|
||||
checkAcoustics()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A ênfase de voz (energia/tom) depende do `librosa`, dependência
|
||||
/// opcional. Sem ela, a análise ainda transcreve e corta pelo texto,
|
||||
/// mas nunca deveria propor zoom — por isso avisamos aqui, no ponto
|
||||
/// onde o usuário sentiria falta, em vez de só na aba Modelos.
|
||||
private func checkAcoustics() {
|
||||
PythonBridge.call(command: "acoustics_capability") { result, _ in
|
||||
DispatchQueue.main.async {
|
||||
guard let result, result["ok"] as? Bool == true else { return }
|
||||
acousticsAvailable = result["available"] as? Bool
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Localiza uma linha markdown do tipo "- **Marker**: valor" (usado nas
|
||||
/// mensagens do bridge Python) e devolve o valor. Aceita o marcador de
|
||||
/// lista "- " opcional antes dos asteriscos.
|
||||
private func extractPath(from message: String, marker: String) -> String? {
|
||||
for line in message.split(separator: "\n") {
|
||||
var trimmed = Substring(line.trimmingCharacters(in: .whitespaces))
|
||||
if trimmed.hasPrefix("- ") { trimmed = trimmed.dropFirst(2) }
|
||||
if trimmed.hasPrefix(marker) {
|
||||
return trimmed.dropFirst(marker.count).trimmingCharacters(in: .whitespaces)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
private func copyForChat(path: String) {
|
||||
guard let content = try? String(contentsOfFile: path, encoding: .utf8) else {
|
||||
errorMessage = "Não foi possível ler \(path)."
|
||||
return
|
||||
}
|
||||
let prompt = """
|
||||
Use a skill "editar-por-voz" para decidir os cortes deste projeto a partir da timeline de voz abaixo. Devolva só o JSON de decisões (cortes, zooms, textos, marcadores) pronto para eu colar de volta no app.
|
||||
|
||||
```json
|
||||
\(content)
|
||||
```
|
||||
"""
|
||||
let pasteboard = NSPasteboard.general
|
||||
pasteboard.clearContents()
|
||||
pasteboard.setString(prompt, forType: .string)
|
||||
copiedFeedback = "Copiado — cole (⌘V) numa conversa com o Claude."
|
||||
}
|
||||
|
||||
/// Monta a revisão uma vez por análise de voz. Voltar e avançar de novo não
|
||||
/// recarrega: isso jogaria fora as edições manuais em silêncio, que é
|
||||
/// exatamente o que esta tela existe para preservar.
|
||||
private func loadReviewIfNeeded() {
|
||||
guard let voiceTimelinePath, reviewLoadedFor != voiceTimelinePath else { return }
|
||||
reviewLoadedFor = voiceTimelinePath
|
||||
// A pasta do projeto e a do .fcpxml entram como onde procurar a mídia:
|
||||
// a análise de voz guarda só o nome do arquivo, não o caminho.
|
||||
reviewModel.load(
|
||||
voiceTimelinePath: voiceTimelinePath,
|
||||
decisionsJSON: decisionsText,
|
||||
outputFolder: outputFolder,
|
||||
mediaFolder: projectPath.map { URL(fileURLWithPath: $0).deletingLastPathComponent().path }
|
||||
)
|
||||
if let projectPath { reviewModel.loadProjectFormat(projectPath: projectPath) }
|
||||
}
|
||||
|
||||
private func applyDecisions() {
|
||||
guard let projectPath, let outputFolder,
|
||||
let data = decisionsText.data(using: .utf8),
|
||||
let parsed = try? JSONSerialization.jsonObject(with: data) else { return }
|
||||
isApplyingDecisions = true
|
||||
errorMessage = nil
|
||||
PythonBridge.call(command: "apply_voice_actions", arguments: [
|
||||
"path": projectPath,
|
||||
"output_dir": outputFolder,
|
||||
"actions": parsed,
|
||||
]) { result, err in
|
||||
DispatchQueue.main.async {
|
||||
isApplyingDecisions = false
|
||||
guard result?["ok"] as? Bool == true else {
|
||||
errorMessage = result?["error"] as? String ?? err ?? "Falha ao aplicar as decisões."
|
||||
return
|
||||
}
|
||||
appliedPath = result?["path"] as? String ?? projectPath
|
||||
skippedVoiceEdit = false
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private func finalizeProcessing() {
|
||||
guard let outputFolder else { return }
|
||||
let startPath = appliedPath ?? projectPath
|
||||
guard let startPath else { return }
|
||||
var operations: [String] = []
|
||||
if finalSilences { operations.append("remove_silences") }
|
||||
if finalFillers { operations.append("remove_filler_words") }
|
||||
if finalSubtitles { operations.append("generate_plain_subtitles") }
|
||||
if finalDynamicSubtitles { operations.append("generate_dynamic_subtitles") }
|
||||
guard !operations.isEmpty else { return }
|
||||
isFinalizing = true
|
||||
errorMessage = nil
|
||||
finalStatus = "Iniciando…"
|
||||
finalizeStep(operations, index: 0, currentPath: startPath, outputFolder: outputFolder)
|
||||
}
|
||||
|
||||
private func finalizeStep(_ operations: [String], index: Int, currentPath: String, outputFolder: String) {
|
||||
guard index < operations.count else {
|
||||
isFinalizing = false
|
||||
finalStatus = "Processamento concluído."
|
||||
finalPath = currentPath
|
||||
return
|
||||
}
|
||||
let operation = operations[index]
|
||||
finalStatus = "Processando: \(operation)…"
|
||||
PythonBridge.call(command: operation, arguments: ["path": currentPath, "output_dir": outputFolder]) { result, err in
|
||||
DispatchQueue.main.async {
|
||||
guard result?["ok"] as? Bool == true else {
|
||||
isFinalizing = false
|
||||
errorMessage = result?["error"] as? String ?? err ?? "Falha em \(operation)."
|
||||
finalStatus = "Processamento interrompido."
|
||||
return
|
||||
}
|
||||
let nextPath = result?["path"] as? String ?? currentPath
|
||||
finalizeStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user