feat: etapa 5 do assistente — revisão de ênfases com timeline

Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-19 21:29:27 -04:00
co-authored by Claude Opus 5
parent e7748c2c58
commit 1bebee4359
31 changed files with 4622 additions and 83 deletions
+19 -11
View File
@@ -12,6 +12,7 @@ struct GArtApp: App {
}
enum ActiveTab: Hashable {
case wizard
case project
case captions
case voiceAnalysis
@@ -20,19 +21,23 @@ enum ActiveTab: Hashable {
}
struct ContentView: View {
@State private var activeTab: ActiveTab? = .project
@State private var activeTab: ActiveTab? = .wizard
var body: some View {
NavigationSplitView {
List(selection: $activeTab) {
Label("Projeto", systemImage: "film")
.tag(ActiveTab.project)
Label("Legendas Dinâmicas", systemImage: "captions.bubble")
.tag(ActiveTab.captions)
Label("Análise de Voz", systemImage: "waveform")
.tag(ActiveTab.voiceAnalysis)
Label("Modelos", systemImage: "tray.and.arrow.down")
.tag(ActiveTab.models)
Label("Assistente", systemImage: "wand.and.stars")
.tag(ActiveTab.wizard)
Section("Avançado") {
Label("Projeto", systemImage: "film")
.tag(ActiveTab.project)
Label("Legendas", systemImage: "captions.bubble")
.tag(ActiveTab.captions)
Label("Análise de Voz", systemImage: "waveform")
.tag(ActiveTab.voiceAnalysis)
Label("Modelos", systemImage: "tray.and.arrow.down")
.tag(ActiveTab.models)
}
Label("Sobre", systemImage: "info.circle")
.tag(ActiveTab.about)
}
@@ -40,19 +45,22 @@ struct ContentView: View {
.navigationSplitViewColumnWidth(min: 180, ideal: 200)
} detail: {
switch activeTab {
case .wizard, nil:
WizardView().id(UUID())
.navigationTitle("Assistente")
case .project:
ProjectView().id(UUID())
.navigationTitle("Projeto")
case .captions:
CaptionsView().id(UUID())
.navigationTitle("Legendas Dinâmicas")
.navigationTitle("Legendas")
case .voiceAnalysis:
VoiceAnalysisView().id(UUID())
.navigationTitle("Análise de Voz")
case .models:
ModelDownloadView().id(UUID())
.navigationTitle("Modelos")
case .about, nil:
case .about:
AboutView()
.navigationTitle("Sobre")
}
+125 -2
View File
@@ -19,6 +19,7 @@ import UniformTypeIdentifiers
/// assunto.
struct CaptionsView: View {
@State private var config = CaptionStyleConfig.defaults
@State private var plainConfig = PlainSubtitleConfig.defaults
@State private var isLoading = true
@State private var errorMessage: String?
@@ -53,6 +54,13 @@ struct CaptionsView: View {
)
}
private func plainBound<T>(_ keyPath: WritableKeyPath<PlainSubtitleConfig, T>) -> Binding<T> {
Binding(
get: { plainConfig[keyPath: keyPath] },
set: { plainConfig[keyPath: keyPath] = $0; savePlain() }
)
}
private func colorBound(_ keyPath: WritableKeyPath<CaptionStyleConfig, String>) -> Binding<Color> {
Binding(
get: { Color(rgbaString: config[keyPath: keyPath]) },
@@ -60,6 +68,13 @@ struct CaptionsView: View {
)
}
private func plainColorBound(_ keyPath: WritableKeyPath<PlainSubtitleConfig, String>) -> Binding<Color> {
Binding(
get: { Color(rgbaString: plainConfig[keyPath: keyPath]) },
set: { plainConfig[keyPath: keyPath] = $0.fcpxmlColorString; savePlain() }
)
}
var body: some View {
HSplitView {
previewColumn
@@ -152,6 +167,7 @@ struct CaptionsView: View {
positionSection
bodySection
emphasisSection
plainSubtitleSection
calibrationSection
}
if let errorMessage {
@@ -225,6 +241,35 @@ struct CaptionsView: View {
}
}
private var plainSubtitleSection: some View {
Section("Legenda comum") {
Picker("Fonte", selection: plainBound(\.font)) {
ForEach(fontChoices, id: \.self) { Text($0).tag($0) }
}
slider(
"Tamanho",
value: plainBound(\.fontSize), in: 28...300, step: 1,
readout: "\(Int(plainConfig.fontSize))pt",
help: "Tamanho da legenda comum editável no Final Cut."
)
slider(
"Máximo de palavras",
value: plainBound(\.maxWords), in: 1...14, step: 1,
readout: "\(Int(plainConfig.maxWords))",
help: "Quantidade máxima de palavras por bloco de legenda."
)
slider(
"Altura",
value: plainBound(\.positionY), in: -1200...300, step: 1,
readout: "\(Int(plainConfig.positionY))",
help: "Posição vertical da legenda comum no quadro; valores mais negativos descem."
)
ColorPicker("Cor", selection: plainColorBound(\.fontColor), supportsOpacity: true)
Toggle("Usar letra maiúscula", isOn: plainBound(\.uppercase))
Toggle("Manter vírgula e ponto", isOn: plainBound(\.keepPunctuation))
}
}
private var calibrationSection: some View {
Section {
slider(
@@ -305,8 +350,17 @@ struct CaptionsView: View {
} else if let error {
errorMessage = error
}
isLoading = false
continuation.resume()
PythonBridge.call(command: "plain_subtitle_config") { plainResult, plainError in
DispatchQueue.main.async {
if let plainResult {
plainConfig = PlainSubtitleConfig(from: plainResult)
} else if let plainError {
errorMessage = plainError
}
isLoading = false
continuation.resume()
}
}
}
}
}
@@ -317,6 +371,12 @@ struct CaptionsView: View {
DispatchQueue.main.async { errorMessage = error }
}
}
private func savePlain() {
PythonBridge.call(command: "set_plain_subtitle_config", arguments: plainConfig.arguments()) { _, error in
DispatchQueue.main.async { errorMessage = error }
}
}
}
/// O estilo das legendas dinâmicas, no formato que a tela edita e o bridge
@@ -402,6 +462,69 @@ struct CaptionStyleConfig {
}
}
struct PlainSubtitleConfig {
var font: String
var fontSize: Double
var fontColor: String
var maxWords: Double
var positionY: Double
var uppercase: Bool
var keepPunctuation: Bool
var textScale: Double
static let defaults = PlainSubtitleConfig(
font: "Helvetica Neue",
fontSize: 82,
fontColor: "1 1 1 1",
maxWords: 7,
positionY: -820,
uppercase: false,
keepPunctuation: true,
textScale: 2.0
)
init(from json: [String: Any]) {
let d = PlainSubtitleConfig.defaults
self.init(
font: json["font"] as? String ?? d.font,
fontSize: (json["font_size"] as? NSNumber)?.doubleValue ?? d.fontSize,
fontColor: json["font_color"] as? String ?? d.fontColor,
maxWords: (json["max_words"] as? NSNumber)?.doubleValue ?? d.maxWords,
positionY: (json["position_y"] as? NSNumber)?.doubleValue ?? d.positionY,
uppercase: json["uppercase"] as? Bool ?? d.uppercase,
keepPunctuation: json["keep_punctuation"] as? Bool ?? d.keepPunctuation,
textScale: (json["text_scale"] as? NSNumber)?.doubleValue ?? d.textScale
)
}
init(
font: String, fontSize: Double, fontColor: String, maxWords: Double,
positionY: Double, uppercase: Bool, keepPunctuation: Bool, textScale: Double
) {
self.font = font
self.fontSize = fontSize
self.fontColor = fontColor
self.maxWords = maxWords
self.positionY = positionY
self.uppercase = uppercase
self.keepPunctuation = keepPunctuation
self.textScale = textScale
}
func arguments() -> [String: Any] {
[
"font": font,
"font_size": Int(fontSize),
"font_color": fontColor,
"max_words": Int(maxWords),
"position_y": positionY,
"uppercase": uppercase,
"keep_punctuation": keepPunctuation,
"text_scale": textScale,
]
}
}
extension Color {
/// Parses an FCPXML "R G B A" space-separated 0-1 string into a Color.
init(rgbaString: String) {
+99 -1
View File
@@ -15,6 +15,11 @@ struct ModelDownloadView: View {
@State private var hfTokenText: String = ""
@State private var numSpeakersText: String = ""
@State private var language: String = "auto"
@State private var acousticsAvailable: Bool?
@State private var acousticsMessage: String = ""
@State private var isInstallingAcoustics = false
@State private var acousticsInstallLog: String = ""
@State private var acousticsInstallError: String?
private let languages: [(String, String)] = [
("auto", "Detectar automaticamente"),
@@ -33,6 +38,7 @@ struct ModelDownloadView: View {
var body: some View {
Form {
storageSection
acousticsSection
diarizationSection
if let errorMessage {
Section {
@@ -65,7 +71,7 @@ struct ModelDownloadView: View {
}
}
.formStyle(.grouped)
.task { await refresh() }
.task { await refresh(); checkAcoustics() }
}
// MARK: - Transcription language
@@ -95,6 +101,98 @@ struct ModelDownloadView: View {
PythonBridge.call(command: "set_language", arguments: ["language": code]) { _, _ in }
}
// MARK: - Acoustic analysis (librosa)
/// A ênfase de voz (pitch/energia) precisa do `librosa`, que é uma
/// dependência opcional — sem ela `layers.acoustics` vem `false` na
/// análise e a decisão de zoom fica sem base real. Antes disso só dava
/// pra descobrir lendo o JSON exportado; agora o app já diz e resolve.
private var acousticsSection: some View {
Section {
VStack(alignment: .leading, spacing: 10) {
if let acousticsAvailable {
Label(
acousticsMessage.isEmpty
? (acousticsAvailable ? "Disponível" : "Indisponível")
: acousticsMessage,
systemImage: acousticsAvailable ? "checkmark.circle.fill" : "exclamationmark.triangle.fill"
)
.font(.caption)
.foregroundStyle(acousticsAvailable ? Color.green : Color.orange)
} else {
Label("Verificando…", systemImage: "hourglass")
.font(.caption).foregroundStyle(.secondary)
}
if acousticsAvailable == false {
Button {
installAcoustics()
} label: {
if isInstallingAcoustics {
HStack { ProgressView().controlSize(.small); Text("Instalando…") }
} else {
Label("Instalar (uv sync --all-extras)", systemImage: "arrow.down.circle")
}
}
.disabled(isInstallingAcoustics)
if !acousticsInstallLog.isEmpty {
ScrollView {
Text(acousticsInstallLog)
.font(.system(.caption2, design: .monospaced))
.foregroundStyle(.secondary)
.frame(maxWidth: .infinity, alignment: .leading)
}
.frame(height: 90)
.background(RoundedRectangle(cornerRadius: 6).fill(Color.secondary.opacity(0.06)))
}
if let acousticsInstallError {
Label(acousticsInstallError, systemImage: "xmark.circle.fill")
.font(.caption).foregroundStyle(.red)
}
}
}
} header: {
Text("Análise Acústica (zoom por voz)")
} footer: {
Text("Mede a energia e o tom de voz de verdade, para os candidatos a zoom da edição por voz. Sem isso, a análise ainda transcreve e decide cortes pelo texto — só o zoom fica sem base acústica.")
.font(.caption)
.foregroundStyle(.secondary)
}
}
private func checkAcoustics() {
PythonBridge.call(command: "acoustics_capability") { result, err in
DispatchQueue.main.async {
guard let result, result["ok"] as? Bool == true else { return }
acousticsAvailable = result["available"] as? Bool
acousticsMessage = result["message"] as? String ?? ""
}
}
}
private func installAcoustics() {
isInstallingAcoustics = true
acousticsInstallLog = ""
acousticsInstallError = nil
// --all-extras, não só "intelligence": `uv sync` substitui o
// ambiente pelos extras pedidos em vez de somar, então um sync
// parcial aqui derrubaria dev/transcribe/diarização já instalados.
PythonBridge.runUV(arguments: ["sync", "--all-extras"]) { line in
DispatchQueue.main.async {
acousticsInstallLog += (acousticsInstallLog.isEmpty ? "" : "\n") + line
}
} completion: { code, err in
DispatchQueue.main.async {
isInstallingAcoustics = false
if code != 0 {
acousticsInstallError = err ?? "Falha ao instalar."
}
checkAcoustics()
}
}
}
// MARK: - Diarization
private var diarizationSection: some View {
+139
View File
@@ -116,6 +116,145 @@ struct ZoomClip: Identifiable {
}
}
/// One word inside a phrase, with the acoustics that justify an emphasis.
struct ReviewWord: Identifiable {
let id: Int
let text: String
let start: Double
let end: Double
let energy: Double
let emphasis: Double
init(id: Int, json: [String: Any]) {
self.id = id
text = json["text"] as? String ?? ""
start = json["start"] as? Double ?? 0
end = json["end"] as? Double ?? 0
energy = json["energy"] as? Double ?? 0
emphasis = json["emphasis"] as? Double ?? 0
}
}
/// A phrase in the review step — one spoken line plus the decision made about
/// it. Mirrors `fcpxml/phrase_review.py`; `emphasis` is 0–3 and everything
/// mutable here is what the editor is allowed to change.
struct ReviewPhrase: Identifiable {
let id: Int
let start: Double
let end: Double
var trimStart: Double
var trimEnd: Double
var text: String
let speaker: String
var active: Bool
var emphasis: Int
var track: String
let peakEmphasis: Double
let emotion: String
let emotionConfidence: Double
let takeBoundary: Bool
let gapBefore: Double
let reason: String
let words: [ReviewWord]
static let trackScript = "roteiro"
static let trackBackstage = "bastidor"
/// Delivery emotion as the analysis names it, in the user's language plus a
/// glyph — the label alone is too easy to skim past in a dense list.
static func emotionLabel(_ emotion: String) -> (String, String) {
switch emotion {
case "excited": return ("Empolgado", "flame")
case "tense": return ("Tenso", "bolt")
case "calm": return ("Calmo", "leaf")
case "reflective": return ("Reflexivo", "moon")
default: return ("Neutro", "circle")
}
}
init(json: [String: Any]) {
id = json["index"] as? Int ?? 0
start = json["start"] as? Double ?? 0
end = json["end"] as? Double ?? 0
trimStart = json["trim_start"] as? Double ?? (json["start"] as? Double ?? 0)
trimEnd = json["trim_end"] as? Double ?? (json["end"] as? Double ?? 0)
text = json["text"] as? String ?? ""
speaker = json["speaker"] as? String ?? ""
active = json["active"] as? Bool ?? true
emphasis = json["emphasis"] as? Int ?? 0
track = json["track"] as? String ?? ReviewPhrase.trackScript
peakEmphasis = json["peak_emphasis"] as? Double ?? 0
emotion = json["emotion"] as? String ?? "neutral"
emotionConfidence = json["emotion_confidence"] as? Double ?? 0
takeBoundary = json["take_boundary"] as? Bool ?? false
gapBefore = json["gap_before"] as? Double ?? 0
reason = json["reason"] as? String ?? ""
words = (json["words"] as? [[String: Any]] ?? [])
.enumerated().map { ReviewWord(id: $0.offset, json: $0.element) }
}
var asJSON: [String: Any] {
[
"index": id,
"start": start,
"end": end,
"trim_start": trimStart,
"trim_end": trimEnd,
"text": text,
"speaker": speaker,
"active": active,
"emphasis": emphasis,
"track": track,
"reason": reason,
]
}
var isBackstage: Bool { track == ReviewPhrase.trackBackstage }
var isTrimmed: Bool { trimStart > start + 0.001 || trimEnd < end - 0.001 }
var timecode: String {
String(format: "%02d:%02d", Int(start) / 60, Int(start) % 60)
}
/// The word boundaries a trim handle is allowed to land on.
func snap(_ time: Double, edge: TrimEdge) -> Double {
let boundaries = words.map { edge == .start ? $0.start : $0.end }.filter { $0 > 0 }
guard let nearest = boundaries.min(by: { abs($0 - time) < abs($1 - time) }) else {
return time
}
return nearest
}
}
enum TrimEdge { case start, end }
/// A punch-in the editor placed by hand over an arbitrary range, next to the
/// whole-phrase zoom that an emphasis level produces. It stores only *when* —
/// the scale and the ramp come from the Voice Analysis settings at render time.
struct ManualZoom: Identifiable {
let id = UUID()
var start: Double
var end: Double
/// Below this a punch-in has no room to ramp in and back out; the writer
/// rejects the window, so offering it would place nothing.
static let minimumDuration: Double = 0.4
init(start: Double, end: Double) {
self.start = start
self.end = end
}
init?(json: [String: Any]) {
guard let start = json["start"] as? Double, let end = json["end"] as? Double,
end - start >= ManualZoom.minimumDuration
else { return nil }
self.start = start
self.end = end
}
var asJSON: [String: Any] { ["start": start, "end": end] }
}
struct ZoomSegment: Identifiable {
let id: Int
let start: Double
+429
View File
@@ -0,0 +1,429 @@
import AVFoundation
import Combine
import Foundation
/// State behind the wizard's emphasis-review step.
///
/// Holds the phrases, the selection, and the player — together, because they
/// are one thing to the user: clicking a phrase moves the playhead, playing
/// moves the selection, and skipping a removed line only works if whoever owns
/// playback also knows which lines are removed.
///
/// The preview deliberately plays the *original* media and jumps over whatever
/// the edit removes, instead of rendering a cut first. Rendering to check a
/// toggle would put minutes between a decision and its result; jumping gives
/// the same reading instantly, and the real cut is generated later from the
/// exact same phrase list.
@MainActor
final class PhraseReviewModel: ObservableObject {
@Published var phrases: [ReviewPhrase] = []
@Published var selection: Int?
@Published var isLoading = false
@Published var errorMessage: String?
@Published var currentTime: Double = 0
@Published var isPlaying = false
@Published var pixelsPerSecond: Double = 40
@Published var skipRemoved = true
@Published var zooms: [ManualZoom] = []
/// In/out the editor dragged on the timeline, in source seconds.
@Published var rangeStart: Double?
@Published var rangeEnd: Double?
/// Aspect ratio of the footage as recorded.
@Published var videoAspect: Double = 16.0 / 9.0
/// Aspect ratio the project delivers in, read from the .fcpxml. It is
/// routinely *not* the footage's: these takes are shot horizontal and
/// delivered vertical, so previewing the raw frame would show a crop the
/// audience never sees — and the emphasis decisions are about what lands on
/// screen. Nil until the project is known.
@Published var projectAspect: Double?
/// Whether the preview crops to the delivery frame. On by default whenever
/// the two aspects disagree.
@Published var matchProjectFraming = true
/// What the preview should actually draw.
var previewAspect: Double {
guard matchProjectFraming, let projectAspect else { return videoAspect }
return projectAspect
}
/// True when the delivery frame differs enough from the footage that the
/// preview is showing a crop rather than the whole take.
var isCropping: Bool {
guard matchProjectFraming, let projectAspect else { return false }
return abs(projectAspect - videoAspect) > 0.01
}
private(set) var source = ""
private(set) var sourcePath = ""
private(set) var duration: Double = 0
private(set) var speakers: [String] = []
private(set) var emotionAvailable = false
private(set) var player: AVPlayer?
private var voiceTimelinePath = ""
private var timeObserver: Any?
private var playbackLimit: Double?
let minPixelsPerSecond: Double = 8
let maxPixelsPerSecond: Double = 400
deinit {
if let timeObserver, let player {
player.removeTimeObserver(timeObserver)
}
}
// MARK: - Carregar
/// Builds the review from the voice timeline plus whatever the AI decided.
/// A review saved on a previous visit wins — see `cmd_build_phrase_review`.
/// Reads the delivery format from the project so the preview can frame the
/// take the way it will actually be seen.
func loadProjectFormat(projectPath: String) {
PythonBridge.call(command: "inspect", arguments: ["path": projectPath]) { [weak self] result, _ in
Task { @MainActor in
guard let self,
let timelines = result?["timelines"] as? [[String: Any]],
let first = timelines.first,
let width = first["width"] as? Int, let height = first["height"] as? Int,
width > 0, height > 0
else { return }
self.projectAspect = Double(width) / Double(height)
}
}
}
func load(voiceTimelinePath: String, decisionsJSON: String,
outputFolder: String? = nil, mediaFolder: String? = nil) {
self.voiceTimelinePath = voiceTimelinePath
isLoading = true
errorMessage = nil
var arguments: [String: Any] = ["voice_timeline": voiceTimelinePath]
if let outputFolder { arguments["output_dir"] = outputFolder }
if let mediaFolder { arguments["media_dir"] = mediaFolder }
if let data = decisionsJSON.data(using: .utf8),
let parsed = try? JSONSerialization.jsonObject(with: data) {
arguments["actions"] = parsed
}
PythonBridge.call(command: "build_phrase_review", arguments: arguments) { [weak self] result, error in
Task { @MainActor in
guard let self else { return }
self.isLoading = false
if let error {
self.errorMessage = error
return
}
guard let result, result["ok"] as? Bool == true else {
self.errorMessage = result?["error"] as? String ?? "Não foi possível montar a revisão."
return
}
self.apply(result)
}
}
}
private func apply(_ result: [String: Any]) {
source = result["source"] as? String ?? ""
// The timeline JSON stores only the media's file name; the bridge
// resolves it to something openable (see phrase_review.resolve_source).
sourcePath = result["source_path"] as? String ?? ""
duration = result["duration"] as? Double ?? 0
speakers = result["speakers"] as? [String] ?? []
emotionAvailable = result["emotion_available"] as? Bool ?? false
phrases = (result["phrases"] as? [[String: Any]] ?? []).map { ReviewPhrase(json: $0) }
zooms = (result["zooms"] as? [[String: Any]] ?? []).compactMap { ManualZoom(json: $0) }
selection = phrases.first?.id
if let errors = result["errors"] as? [String], !errors.isEmpty {
errorMessage = "A IA mandou \(errors.count) decisão(ões) que não deu para ler — o resto foi aplicado."
}
preparePlayer()
}
/// Point the preview at a media file the user chose by hand — the way out
/// when the footage moved somewhere the automatic lookup can't reach.
func useMedia(at path: String) {
sourcePath = path
preparePlayer()
}
private func preparePlayer() {
guard !sourcePath.isEmpty, FileManager.default.fileExists(atPath: sourcePath) else {
player = nil
return
}
if let timeObserver, let player {
player.removeTimeObserver(timeObserver)
self.timeObserver = nil
}
let asset = AVURLAsset(url: URL(fileURLWithPath: sourcePath))
let player = AVPlayer(playerItem: AVPlayerItem(asset: asset))
self.player = player
readAspect(from: asset)
// 60 Hz: the same observer drives the playhead *and* decides when to
// jump a removed stretch, so its period is the worst-case amount of cut
// material that can be heard before the skip lands. At 20 Hz that was an
// audible blip on every join.
let interval = CMTime(seconds: 1.0 / 60.0, preferredTimescale: 600)
timeObserver = player.addPeriodicTimeObserver(forInterval: interval, queue: .main) { [weak self] time in
Task { @MainActor in
self?.tick(time.seconds)
}
}
}
/// The displayed aspect ratio, honouring the rotation the camera recorded.
/// A phone take is stored 1920×1080 with a 90° transform: reading
/// `naturalSize` alone would call a vertical video horizontal.
private func readAspect(from asset: AVURLAsset) {
Task { [weak self] in
guard let track = try? await asset.loadTracks(withMediaType: .video).first,
let size = try? await track.load(.naturalSize),
let transform = try? await track.load(.preferredTransform)
else { return }
let displayed = size.applying(transform)
let width = abs(displayed.width), height = abs(displayed.height)
guard width > 0, height > 0 else { return }
await MainActor.run { self?.videoAspect = width / height }
}
}
// MARK: - Reprodução
private func tick(_ time: Double) {
currentTime = time
guard isPlaying else { return }
// Playing a single phrase or a marked range stops at its out point
// instead of running on into the rest of the take.
if let limit = playbackLimit, time >= limit {
pause()
seek(to: limit)
return
}
if skipRemoved, let jump = nextKeptTime(after: time), jump > time {
seek(to: jump)
}
if let phrase = phrase(at: time), selection != phrase.id {
selection = phrase.id
}
}
/// Where playback should resume when `time` lands on removed material.
/// Returns nil when the time is on material that survives.
func nextKeptTime(after time: Double) -> Double? {
for phrase in phrases where time >= phrase.start - 0.001 && time < phrase.end {
if !phrase.active { return phrase.end }
if time < phrase.trimStart { return phrase.trimStart }
if time >= phrase.trimEnd { return phrase.end }
return nil
}
return nil
}
func togglePlay() {
if isPlaying {
pause()
} else {
playbackLimit = nil
play()
}
}
private func play() {
guard let player else { return }
if skipRemoved, let jump = nextKeptTime(after: currentTime) { seek(to: jump) }
player.play()
isPlaying = true
}
func pause() {
player?.pause()
isPlaying = false
playbackLimit = nil
}
/// Play exactly one span and stop — how a cut is judged: in context, at
/// speed, without hunting for the out point by hand.
func playRange(from start: Double, to end: Double) {
guard end > start else { return }
seek(to: start)
playbackLimit = end
player?.play()
isPlaying = true
}
func playSelectedPhrase() {
guard let selection, let phrase = phrases.first(where: { $0.id == selection })
else { return }
playRange(from: phrase.active ? phrase.trimStart : phrase.start,
to: phrase.active ? phrase.trimEnd : phrase.end)
}
func seek(to time: Double) {
currentTime = max(0, time)
player?.seek(to: CMTime(seconds: max(0, time), preferredTimescale: 600),
toleranceBefore: .zero, toleranceAfter: .zero)
}
/// Move the playhead to a phrase and select it.
func goTo(phraseID: Int) {
guard let phrase = phrases.first(where: { $0.id == phraseID }) else { return }
selection = phraseID
seek(to: phrase.active ? phrase.trimStart : phrase.start)
}
func phrase(at time: Double) -> ReviewPhrase? {
phrases.first { time >= $0.start && time < $0.end }
}
func selectNeighbour(_ delta: Int) {
guard let selection, let index = phrases.firstIndex(where: { $0.id == selection }) else {
if let first = phrases.first { goTo(phraseID: first.id) }
return
}
let next = min(max(0, index + delta), phrases.count - 1)
goTo(phraseID: phrases[next].id)
}
// MARK: - Edições
private func update(_ id: Int, _ change: (inout ReviewPhrase) -> Void) {
guard let index = phrases.firstIndex(where: { $0.id == id }) else { return }
change(&phrases[index])
}
func setEmphasis(_ level: Int, for id: Int) {
update(id) { $0.emphasis = min(3, max(0, level)) }
}
func toggleActive(_ id: Int) {
update(id) { $0.active.toggle() }
}
func setTrack(_ track: String, for id: Int) {
update(id) { $0.track = track }
}
func setText(_ text: String, for id: Int) {
update(id) { $0.text = text }
}
/// Trim a phrase's head or tail, landing on a word boundary.
/// A trim that would swallow the whole line is refused — deactivating the
/// phrase is the way to remove it, and doing it by accident with a drag
/// would lose the emphasis decision along with the line.
func trim(_ id: Int, edge: TrimEdge, to time: Double) {
update(id) { phrase in
let snapped = phrase.snap(time, edge: edge)
switch edge {
case .start:
let value = min(max(phrase.start, snapped), phrase.trimEnd - 0.1)
if value < phrase.trimEnd { phrase.trimStart = value }
case .end:
let value = max(min(phrase.end, snapped), phrase.trimStart + 0.1)
if value > phrase.trimStart { phrase.trimEnd = value }
}
}
}
func resetTrim(_ id: Int) {
update(id) { $0.trimStart = $0.start; $0.trimEnd = $0.end }
}
/// Trim everything before/after a given word — the text-first way to cut,
/// since the editor reads the line and points at where it should begin.
func trimToWord(_ word: ReviewWord, edge: TrimEdge, in id: Int) {
trim(id, edge: edge, to: edge == .start ? word.start : word.end)
}
// MARK: - Trecho marcado e zooms
var hasRange: Bool {
guard let rangeStart, let rangeEnd else { return false }
return rangeEnd - rangeStart >= ManualZoom.minimumDuration
}
var rangeSpan: (start: Double, end: Double)? {
guard let rangeStart, let rangeEnd, rangeEnd > rangeStart else { return nil }
return (rangeStart, rangeEnd)
}
func setRange(from start: Double, to end: Double) {
rangeStart = min(start, end)
rangeEnd = max(start, end)
}
func clearRange() {
rangeStart = nil
rangeEnd = nil
}
/// Add a punch-in over the marked range. Scale and ramp are not stored:
/// they come from the "Análise de Voz" settings when the edit is rendered,
/// so changing the look there restyles every zoom at once.
func addZoomForRange() {
guard let span = rangeSpan, span.end - span.start >= ManualZoom.minimumDuration
else { return }
zooms.append(ManualZoom(start: span.start, end: span.end))
zooms.sort { $0.start < $1.start }
clearRange()
}
func addZoomForPhrase(_ id: Int) {
guard let phrase = phrases.first(where: { $0.id == id }) else { return }
zooms.append(ManualZoom(start: phrase.trimStart, end: phrase.trimEnd))
zooms.sort { $0.start < $1.start }
}
func removeZoom(_ id: UUID) {
zooms.removeAll { $0.id == id }
}
func zoom(at time: Double) -> ManualZoom? {
zooms.first { time >= $0.start && time <= $0.end }
}
func setEmphasisForAll(_ level: Int) {
for index in phrases.indices where phrases[index].active {
phrases[index].emphasis = level
}
}
// MARK: - Resumo e gravação
var emphasisCount: Int { phrases.filter { $0.active && $0.emphasis >= 1 }.count }
var removedCount: Int { phrases.filter { !$0.active }.count }
var keptDuration: Double {
phrases.filter { $0.active }.reduce(0) { $0 + ($1.trimEnd - $1.trimStart) }
}
/// Persists the edited review plus the actions derived from it. Called when
/// the wizard advances — the render itself happens in the next step.
func save(completion: @escaping (String?) -> Void) {
guard !voiceTimelinePath.isEmpty, !phrases.isEmpty else {
completion(nil)
return
}
let arguments: [String: Any] = [
"voice_timeline": voiceTimelinePath,
"source": source,
"duration": duration,
"speakers": speakers,
"phrases": phrases.map { $0.asJSON },
"zooms": zooms.map { $0.asJSON },
]
PythonBridge.call(command: "save_phrase_review", arguments: arguments) { result, error in
Task { @MainActor in
if let error {
completion(nil)
_ = error
return
}
completion(result?["review_path"] as? String)
}
}
}
}
+413
View File
@@ -0,0 +1,413 @@
import AVFoundation
import SwiftUI
/// The video surface, as a plain `AVPlayerLayer` in an `NSView`.
///
/// AVKit's `VideoPlayer` would be the obvious choice and is a trap here: this
/// app is built by invoking `swiftc` directly (see `MacApp/build_app.sh`), and
/// `_AVKit_SwiftUI` aborts at launch instantiating its generic metadata under
/// that build. A player layer needs only AVFoundation, which links cleanly —
/// and the transport controls live in the timeline's own toolbar anyway, so
/// nothing is lost by dropping AVKit's chrome.
private struct PlayerSurface: NSViewRepresentable {
let player: AVPlayer
/// When true the frame is filled and cropped instead of letterboxed — used
/// to preview horizontal footage inside a vertical delivery frame.
var fills: Bool
func makeNSView(context: Context) -> PlayerLayerView {
let view = PlayerLayerView()
view.player = player
view.fills = fills
return view
}
func updateNSView(_ view: PlayerLayerView, context: Context) {
if view.player !== player { view.player = player }
view.fills = fills
}
}
final class PlayerLayerView: NSView {
private let playerLayer = AVPlayerLayer()
var player: AVPlayer? {
get { playerLayer.player }
set { playerLayer.player = newValue }
}
var fills: Bool = false {
didSet { playerLayer.videoGravity = fills ? .resizeAspectFill : .resizeAspect }
}
override init(frame frameRect: NSRect) {
super.init(frame: frameRect)
wantsLayer = true
layer = CALayer()
layer?.backgroundColor = NSColor.black.cgColor
playerLayer.videoGravity = .resizeAspect
layer?.addSublayer(playerLayer)
}
required init?(coder: NSCoder) {
super.init(coder: coder)
wantsLayer = true
layer = CALayer()
playerLayer.videoGravity = .resizeAspect
layer?.addSublayer(playerLayer)
}
override func layout() {
super.layout()
playerLayer.frame = bounds
}
}
/// The wizard's emphasis-review step, laid out like an editing room: preview on
/// top, timeline across the bottom, and the script as an inspector down the
/// right side.
///
/// The arrangement is the point. Every decision here is about a *sentence*, so
/// the same phrase has to be legible in all three places at once — a block on
/// the timeline, a line of text in the inspector, and a moment in the preview.
/// Selecting in any one of them selects in the other two.
struct PhraseReviewView: View {
@ObservedObject var model: PhraseReviewModel
var body: some View {
VSplitView {
HSplitView {
previewPane
.frame(minWidth: 320, idealWidth: 640)
inspectorPane
.frame(minWidth: 300, idealWidth: 360, maxWidth: 520)
}
.frame(minHeight: 240)
TimelineTracksView(model: model)
.frame(minHeight: 190, idealHeight: 210)
}
.overlay { if model.isLoading { loadingOverlay } }
.focusable()
.onKeyPress(.space) { model.togglePlay(); return .handled }
.onKeyPress(.return) { model.playSelectedPhrase(); return .handled }
.onKeyPress(.leftArrow) { model.selectNeighbour(-1); return .handled }
.onKeyPress(.rightArrow) { model.selectNeighbour(1); return .handled }
.onKeyPress(characters: .decimalDigits) { press in
guard let level = Int(press.characters), (0...3).contains(level),
let selection = model.selection else { return .ignored }
model.setEmphasis(level, for: selection)
return .handled
}
}
private var loadingOverlay: some View {
ZStack {
Color(nsColor: .windowBackgroundColor).opacity(0.85)
VStack(spacing: 10) {
ProgressView()
Text("Montando a revisão…").font(.callout).foregroundStyle(.secondary)
}
}
}
// MARK: - Preview
private var previewPane: some View {
VStack(spacing: 0) {
if let player = model.player {
// The footage here is usually vertical. Sizing the surface to
// the take's own aspect keeps a 9:16 frame as tall as the pane
// allows instead of shrinking it to fit a horizontal box.
// Framed to what the project delivers, not to what the camera
// recorded: these takes are shot horizontal and cut vertical,
// so the raw frame would show material the audience never sees.
ZStack {
Color.black
PlayerSurface(player: player, fills: model.isCropping)
.aspectRatio(model.previewAspect, contentMode: .fit)
.clipped()
}
.overlay(alignment: .topTrailing) { framingBadge }
} else {
ZStack {
Color.black.opacity(0.85)
VStack(spacing: 10) {
Image(systemName: "film.stack")
.font(.system(size: 28)).foregroundStyle(.secondary)
Text(model.source.isEmpty
? "A análise de voz não registrou qual mídia foi usada."
: "Não achei \(model.source) na pasta do projeto.")
.font(.callout).foregroundStyle(.secondary)
Text("A revisão funciona igual sem o preview — ele só ajuda a conferir o corte.")
.font(.caption).foregroundStyle(.tertiary)
Button("Localizar a mídia…") { pickMedia() }
.buttonStyle(.bordered)
}
.multilineTextAlignment(.center)
.padding(.horizontal, 24)
}
}
Divider()
summaryBar
}
}
private var summaryBar: some View {
HStack(spacing: 16) {
summaryItem("text.quote", "\(model.phrases.count) frases")
summaryItem("sparkles", "\(model.emphasisCount) com ênfase")
summaryItem("scissors", "\(model.removedCount) fora do corte")
summaryItem("clock", durationLabel(model.keptDuration))
if !model.zooms.isEmpty {
summaryItem("plus.magnifyingglass", "\(model.zooms.count) zooms")
}
Spacer()
if let phrase = selectedPhrase, !phrase.reason.isEmpty {
Label(phrase.reason, systemImage: "brain")
.font(.caption).foregroundStyle(.secondary)
.lineLimit(1).truncationMode(.tail)
}
}
.padding(.horizontal, 14)
.padding(.vertical, 8)
}
private func summaryItem(_ icon: String, _ text: String) -> some View {
Label(text, systemImage: icon).font(.caption).foregroundStyle(.secondary)
}
private func durationLabel(_ seconds: Double) -> String {
String(format: "%02d:%02d finais", Int(seconds) / 60, Int(seconds) % 60)
}
/// Says which frame is on screen, and lets the editor flip to the raw take.
/// Without it a centred crop looks like the footage itself, and someone
/// would judge framing on an approximation without knowing it.
@ViewBuilder
private var framingBadge: some View {
if model.projectAspect != nil, abs((model.projectAspect ?? 0) - model.videoAspect) > 0.01 {
Button {
model.matchProjectFraming.toggle()
} label: {
Label(model.matchProjectFraming ? "Enquadramento do projeto" : "Mídia original",
systemImage: model.matchProjectFraming ? "crop" : "rectangle.expand.vertical")
.font(.caption2)
}
.buttonStyle(.borderless)
.padding(6)
.background(Capsule().fill(.black.opacity(0.45)))
.foregroundStyle(.white)
.padding(8)
.help("A fonte é horizontal e o projeto é vertical — o preview mostra o corte central aproximado. O enquadramento real de cada clipe vem do Final Cut.")
}
}
private func pickMedia() {
let panel = NSOpenPanel()
panel.canChooseFiles = true
panel.canChooseDirectories = false
panel.allowsMultipleSelection = false
panel.prompt = "Usar esta mídia"
panel.message = model.source.isEmpty
? "Escolha o arquivo de vídeo desta gravação."
: "Escolha onde está \(model.source)."
if panel.runModal() == .OK, let url = panel.url {
model.useMedia(at: url.path)
}
}
private var selectedPhrase: ReviewPhrase? {
guard let selection = model.selection else { return nil }
return model.phrases.first { $0.id == selection }
}
// MARK: - Inspector de frases
private var inspectorPane: some View {
VStack(spacing: 0) {
inspectorHeader
Divider()
List(selection: $model.selection) {
ForEach($model.phrases) { $phrase in
PhraseRow(phrase: $phrase, model: model)
.tag(phrase.id)
}
}
.listStyle(.inset)
.onChange(of: model.selection) { _, newValue in
if let newValue { model.goTo(phraseID: newValue) }
}
}
}
private var inspectorHeader: some View {
VStack(alignment: .leading, spacing: 6) {
Text("Frases").font(.headline)
Text("Só as frases com ênfase recebem zoom e legenda dinâmica. O resto fica com legenda comum.")
.font(.caption).foregroundStyle(.secondary)
if !model.emotionAvailable {
Label("Emoção da fala não foi detectada nesta análise — ligue em Avançado → Análise de Voz e refaça o passo 3.",
systemImage: "waveform.path.ecg")
.font(.caption2).foregroundStyle(.secondary)
}
HStack(spacing: 8) {
Button("Limpar ênfases") { model.setEmphasisForAll(0) }
.buttonStyle(.link).font(.caption)
Spacer()
Text("0–3 no teclado · ← → navega")
.font(.caption2).foregroundStyle(.secondary)
}
}
.padding(12)
}
}
/// One phrase in the inspector: the line as it will be said, plus every
/// decision attached to it. Kept in one row on purpose — jumping to a separate
/// detail pane to set a toggle would double the clicks on the most repeated
/// action in the screen.
private struct PhraseRow: View {
@Binding var phrase: ReviewPhrase
@ObservedObject var model: PhraseReviewModel
@State private var isEditing = false
var body: some View {
VStack(alignment: .leading, spacing: 6) {
HStack(spacing: 6) {
Text(phrase.timecode)
.font(.system(.caption2, design: .monospaced))
.foregroundStyle(.secondary)
if phrase.takeBoundary {
Image(systemName: "scissors.badge.ellipsis")
.font(.caption2).foregroundStyle(.orange)
.help("Nova tomada começa aqui")
}
if phrase.isTrimmed {
Image(systemName: "arrow.left.and.right.square")
.font(.caption2).foregroundStyle(.blue)
.help("Frase cortada nas pontas")
}
if model.emotionAvailable {
emotionChip
}
Spacer()
Toggle("", isOn: $phrase.active)
.toggleStyle(.switch)
.controlSize(.mini)
.labelsHidden()
.help(phrase.active ? "No corte" : "Fora do corte")
}
if isEditing {
TextField("Texto da frase", text: $phrase.text, axis: .vertical)
.textFieldStyle(.roundedBorder)
.font(.callout)
.onSubmit { isEditing = false }
} else {
Text(phrase.text.isEmpty ? "(sem texto)" : phrase.text)
.font(.callout)
.foregroundStyle(phrase.active ? .primary : .secondary)
.strikethrough(!phrase.active)
.onTapGesture(count: 2) { isEditing = true }
}
HStack(spacing: 8) {
Picker("", selection: $phrase.emphasis) {
ForEach(0..<4, id: \.self) { level in
Text(EmphasisPalette.label(level)).tag(level)
}
}
.pickerStyle(.segmented)
.controlSize(.mini)
.labelsHidden()
.disabled(!phrase.active)
Picker("", selection: $phrase.track) {
Text("Roteiro").tag(ReviewPhrase.trackScript)
Text("Bastidor").tag(ReviewPhrase.trackBackstage)
}
.pickerStyle(.menu)
.controlSize(.mini)
.labelsHidden()
.frame(width: 92)
}
if model.selection == phrase.id && !phrase.words.isEmpty {
wordTrimmer
}
}
.padding(.vertical, 4)
.opacity(phrase.active ? 1 : 0.55)
}
/// The delivery emotion the acoustics suggest. Shown faded below its own
/// confidence: a guess the analysis is unsure about should not compete for
/// attention with the emphasis decision, which is the point of the row.
private var emotionChip: some View {
let (label, icon) = ReviewPhrase.emotionLabel(phrase.emotion)
return Label(label, systemImage: icon)
.font(.caption2)
.padding(.horizontal, 5)
.padding(.vertical, 1)
.background(
Capsule().fill(Color.secondary.opacity(0.12))
)
.foregroundStyle(phrase.emotionConfidence >= 0.5 ? .secondary : .tertiary)
.help("Emoção da entrega: \(label) — confiança \(Int(phrase.emotionConfidence * 100))%")
}
/// Trimming by pointing at the transcript: click a word to start the phrase
/// there, option-click to end it there. Same edit as dragging the block's
/// edge on the timeline, but reachable while reading the line.
private var wordTrimmer: some View {
VStack(alignment: .leading, spacing: 4) {
HStack(spacing: 4) {
Text("Cortar pelas palavras").font(.caption2).foregroundStyle(.secondary)
Spacer()
if phrase.isTrimmed {
Button("Inteira") { model.resetTrim(phrase.id) }
.buttonStyle(.link).font(.caption2)
}
}
FlowWords(words: phrase.words, phrase: phrase) { word, edge in
model.trimToWord(word, edge: edge, in: phrase.id)
}
Text("Clique = começa aqui · ⌥clique = termina aqui")
.font(.caption2).foregroundStyle(.tertiary)
}
.padding(.top, 2)
}
}
/// The phrase's words as wrapping chips, dimmed where they fall outside the trim.
private struct FlowWords: View {
let words: [ReviewWord]
let phrase: ReviewPhrase
let onTrim: (ReviewWord, TrimEdge) -> Void
var body: some View {
// A LazyVGrid with adaptive columns wraps chips without a custom layout;
// phrases are short enough that the slight raggedness beats the cost of
// hand-rolling a flow layout here.
LazyVGrid(columns: [GridItem(.adaptive(minimum: 44), spacing: 3)],
alignment: .leading, spacing: 3) {
ForEach(words) { word in
let kept = word.start >= phrase.trimStart - 0.001 && word.end <= phrase.trimEnd + 0.001
Text(word.text)
.font(.caption2)
.padding(.horizontal, 4)
.padding(.vertical, 2)
.background(
RoundedRectangle(cornerRadius: 3)
.fill(kept ? Color.accentColor.opacity(0.12) : Color.secondary.opacity(0.08))
)
.foregroundStyle(kept ? .primary : .secondary)
.strikethrough(!kept)
.onTapGesture {
onTrim(word, NSEvent.modifierFlags.contains(.option) ? .end : .start)
}
}
}
}
}
+69 -2
View File
@@ -44,12 +44,26 @@ enum PythonBridge {
return ["python3", scriptURL.path]
}
/// `admin/models_api.py` lives outside `code/`, but its dependencies
/// (`pyproject.toml`, `.venv`) live inside it. `uv run` picks the
/// environment from the process's cwd, not from the script path — so
/// running with cwd at the repo root made `uv` create/use a second,
/// empty `.venv` there, silently ignoring everything installed into
/// `code/.venv` (this cost a real debugging session: librosa/pyannote
/// installed successfully but the app kept reporting them missing).
/// Every `uv run` must share the same cwd as `uv sync` to see the same
/// environment.
static var workingDirectory: URL {
projectRoot
codeDirectory
}
/// Directory containing `pyproject.toml` — where `uv sync` must run from.
static var codeDirectory: URL {
projectRoot.appendingPathComponent("code")
}
/// Locate `uv` on PATH or in common install locations.
private static func findUV() -> String? {
static func findUV() -> String? {
if let onPath = which("uv") { return onPath }
let candidates = [
"/usr/local/bin/uv",
@@ -148,6 +162,59 @@ enum PythonBridge {
}
}
// MARK: - uv sync (installing optional extras, e.g. acoustic analysis)
/// Runs `uv <arguments>` from `codeDirectory` (where `pyproject.toml`
/// lives), streaming each output line as plain text — used for
/// `sync --extra intelligence` so "Modelos" can install the librosa
/// extra without the user opening a terminal.
static func runUV(arguments: [String],
onLine: @escaping (String) -> Void,
completion: @escaping (Int, String?) -> Void) {
guard let uv = findUV() else {
completion(1, "uv não encontrado. Instale com: curl -LsSf https://astral.sh/uv/install.sh | sh")
return
}
let process = Process()
process.executableURL = URL(fileURLWithPath: "/usr/bin/env")
process.arguments = [uv] + arguments
process.currentDirectoryURL = codeDirectory
let pipe = Pipe()
process.standardOutput = pipe
process.standardError = pipe
var buffer = ""
let lock = NSLock()
pipe.fileHandleForReading.readabilityHandler = { handle in
let data = handle.availableData
guard !data.isEmpty, let s = String(data: data, encoding: .utf8) else { return }
lock.lock()
buffer += s
let parts = buffer.split(separator: "\n", omittingEmptySubsequences: false)
buffer = String(parts.last ?? "")
let lines = parts.dropLast()
lock.unlock()
for line in lines where !line.isEmpty { onLine(String(line)) }
}
process.terminationHandler = { p in
pipe.fileHandleForReading.readabilityHandler = nil
lock.lock()
let last = buffer.trimmingCharacters(in: .whitespacesAndNewlines)
buffer = ""
lock.unlock()
if !last.isEmpty { onLine(last) }
completion(Int(p.terminationStatus), p.terminationStatus == 0 ? nil : "uv sync terminou com erro (código \(p.terminationStatus)).")
}
do {
try process.run()
} catch {
completion(1, error.localizedDescription)
}
}
// MARK: - Convenience: single JSON result
/// Runs a command and delivers the first parsed JSON document as the result.
@@ -0,0 +1,506 @@
import SwiftUI
/// Colors shared by the timeline and the inspector, so a block and its row in
/// the list always read as the same thing.
enum EmphasisPalette {
static func color(_ level: Int) -> Color {
switch level {
case 1: return Color.blue
case 2: return Color.orange
case 3: return Color.pink
default: return Color.secondary
}
}
static func label(_ level: Int) -> String {
switch level {
case 1: return "Leve"
case 2: return "Média"
case 3: return "Forte"
default: return "Sem"
}
}
static func speakerColor(_ speaker: String, among speakers: [String]) -> Color {
let palette: [Color] = [.teal, .purple, .green, .indigo, .brown, .cyan]
guard let index = speakers.firstIndex(of: speaker) else { return .gray }
return palette[index % palette.count]
}
}
/// The timeline strip: four stacked tracks over one shared time axis.
///
/// Phrases are laid out as real views rather than drawn into a Canvas, because
/// every one of them is a target — click to select, drag its edge to trim,
/// right-click to change emphasis. The dense per-word energy track *is* a
/// Canvas: it has thousands of bars and nothing to hit.
struct TimelineTracksView: View {
@ObservedObject var model: PhraseReviewModel
private let rulerHeight: CGFloat = 18
private let phraseHeight: CGFloat = 46
private let energyHeight: CGFloat = 34
private let stripHeight: CGFloat = 12
private let handleWidth: CGFloat = 8
private let gutterWidth: CGFloat = 92
private let trackSpacing: CGFloat = 4
private var pps: CGFloat { CGFloat(model.pixelsPerSecond) }
private var contentWidth: CGFloat { max(320, CGFloat(model.duration) * pps) }
/// Name, icon and height of each lane, in the order they stack. The gutter
/// and the tracks are built from this one list so a label can never drift
/// off the lane it names.
private var lanes: [(label: String, icon: String, height: CGFloat)] {
[
("", "", rulerHeight),
("Zooms", "plus.magnifyingglass", stripHeight + 6),
("Frases", "text.quote", phraseHeight),
("Energia", "waveform", energyHeight),
("Emoção", "face.smiling", stripHeight),
("Locutor", "person.wave.2", stripHeight),
("Roteiro", "list.bullet.rectangle", stripHeight),
]
}
var body: some View {
VStack(spacing: 0) {
toolbar
Divider()
HStack(alignment: .top, spacing: 0) {
gutter
Divider()
timelineScroller
}
}
.background(Color(nsColor: .underPageBackgroundColor))
}
/// Fixed column naming each lane. Without it the stripes are six colours
/// with no way to tell which one is emotion and which one is the speaker.
private var gutter: some View {
VStack(alignment: .leading, spacing: trackSpacing) {
ForEach(lanes.indices, id: \.self) { index in
let lane = lanes[index]
HStack(spacing: 4) {
if !lane.icon.isEmpty {
Image(systemName: lane.icon).font(.system(size: 9))
}
Text(lane.label).font(.system(size: 10))
Spacer(minLength: 0)
}
.foregroundStyle(.secondary)
.frame(height: lane.height, alignment: .center)
}
}
.padding(.horizontal, 8)
.padding(.vertical, 8)
.frame(width: gutterWidth, alignment: .leading)
}
private var timelineScroller: some View {
ScrollViewReader { proxy in
ScrollView([.horizontal]) {
ZStack(alignment: .topLeading) {
VStack(alignment: .leading, spacing: trackSpacing) {
ruler
zoomTrack
phraseTrack
energyTrack
emotionTrack
speakerTrack
scriptTrack
}
.frame(width: contentWidth, alignment: .leading)
rangeOverlay
playhead
// Anchors the auto-scroll: one invisible marker per
// phrase, so selecting a line off-screen brings it in.
ForEach(model.phrases) { phrase in
Color.clear
.frame(width: 1, height: 1)
.offset(x: x(phrase.start))
.id(phrase.id)
}
}
.padding(.vertical, 8)
.contentShape(Rectangle())
.gesture(scrubGesture)
.contextMenu { timelineMenu }
}
.onChange(of: model.selection) { _, newValue in
guard let newValue else { return }
withAnimation(.easeOut(duration: 0.2)) {
proxy.scrollTo(newValue, anchor: .center)
}
}
}
}
// MARK: - Barra de controles
private var toolbar: some View {
HStack(spacing: 12) {
Button {
model.togglePlay()
} label: {
Image(systemName: model.isPlaying ? "pause.fill" : "play.fill")
}
.buttonStyle(.borderless)
.help("Reproduzir (espaço)")
.disabled(model.player == nil)
Text(timecode(model.currentTime))
.font(.system(.caption, design: .monospaced))
.foregroundStyle(.secondary)
Button {
model.playSelectedPhrase()
} label: {
Image(systemName: "play.rectangle")
}
.buttonStyle(.borderless)
.help("Tocar só a frase selecionada (⏎)")
.disabled(model.player == nil || model.selection == nil)
Toggle("Pular removidos", isOn: $model.skipRemoved)
.toggleStyle(.checkbox)
.font(.caption)
.help("Durante a reprodução, salta os trechos desativados — mostra como o corte ficou.")
Button {
model.addZoomForRange()
} label: {
Label("Zoom no trecho", systemImage: "plus.magnifyingglass")
}
.buttonStyle(.borderless)
.font(.caption)
.disabled(!model.hasRange)
.help("Arraste na timeline para marcar um trecho e crie um zoom nele. A escala vem de Análise de Voz.")
Spacer()
legend
Spacer()
Image(systemName: "minus.magnifyingglass").foregroundStyle(.secondary)
Slider(value: $model.pixelsPerSecond,
in: model.minPixelsPerSecond...model.maxPixelsPerSecond)
.frame(width: 130)
Image(systemName: "plus.magnifyingglass").foregroundStyle(.secondary)
}
.padding(.horizontal, 12)
.padding(.vertical, 8)
}
private var legend: some View {
HStack(spacing: 10) {
ForEach(0..<4, id: \.self) { level in
HStack(spacing: 4) {
RoundedRectangle(cornerRadius: 2)
.fill(EmphasisPalette.color(level))
.frame(width: 10, height: 10)
Text(EmphasisPalette.label(level)).font(.caption2)
}
}
}
.foregroundStyle(.secondary)
}
// MARK: - Trilhas
private var ruler: some View {
Canvas { context, size in
let step = tickStep()
var time = 0.0
while time <= model.duration {
let position = x(time)
context.stroke(
Path { $0.move(to: CGPoint(x: position, y: size.height - 6))
$0.addLine(to: CGPoint(x: position, y: size.height)) },
with: .color(.secondary.opacity(0.5))
)
context.draw(
Text(timecode(time)).font(.system(size: 9, design: .monospaced))
.foregroundColor(.secondary),
at: CGPoint(x: position + 18, y: 6)
)
time += step
}
}
.frame(width: contentWidth, height: rulerHeight)
}
private var phraseTrack: some View {
ZStack(alignment: .topLeading) {
RoundedRectangle(cornerRadius: 4)
.fill(Color.secondary.opacity(0.06))
.frame(width: contentWidth, height: phraseHeight)
ForEach(model.phrases) { phrase in
phraseBlock(phrase)
}
}
.frame(width: contentWidth, height: phraseHeight, alignment: .topLeading)
}
@ViewBuilder
private func phraseBlock(_ phrase: ReviewPhrase) -> some View {
let isSelected = model.selection == phrase.id
let color = EmphasisPalette.color(phrase.emphasis)
let fullWidth = max(2, width(from: phrase.start, to: phrase.end))
let keptWidth = max(1, width(from: phrase.trimStart, to: phrase.trimEnd))
ZStack(alignment: .topLeading) {
// The whole line, dim — what is there before the edit.
RoundedRectangle(cornerRadius: 4)
.fill(color.opacity(phrase.active ? 0.15 : 0.10))
.frame(width: fullWidth, height: phraseHeight)
// What survives: the kept span, drawn solid over it.
RoundedRectangle(cornerRadius: 4)
.fill(color.opacity(phrase.active ? 0.55 : 0.12))
.frame(width: keptWidth, height: phraseHeight)
.offset(x: width(from: phrase.start, to: phrase.trimStart))
Text(phrase.text)
.font(.system(size: 10))
.lineLimit(2)
.padding(.horizontal, 4)
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
.foregroundStyle(phrase.active ? .primary : .secondary)
.strikethrough(!phrase.active)
RoundedRectangle(cornerRadius: 4)
.stroke(isSelected ? Color.accentColor : color.opacity(0.4),
lineWidth: isSelected ? 2 : 1)
.frame(width: fullWidth, height: phraseHeight)
if isSelected && phrase.active {
trimHandle(phrase, edge: .start)
trimHandle(phrase, edge: .end)
}
}
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
.offset(x: x(phrase.start))
.contentShape(Rectangle())
.onTapGesture { model.goTo(phraseID: phrase.id) }
.contextMenu { phraseMenu(phrase) }
.help(phrase.reason.isEmpty ? phrase.text : "\(phrase.text)\n— \(phrase.reason)")
}
private func trimHandle(_ phrase: ReviewPhrase, edge: TrimEdge) -> some View {
let offset = edge == .start
? width(from: phrase.start, to: phrase.trimStart)
: width(from: phrase.start, to: phrase.trimEnd) - handleWidth
return RoundedRectangle(cornerRadius: 2)
.fill(Color.accentColor)
.frame(width: handleWidth, height: phraseHeight)
.offset(x: offset)
.gesture(
DragGesture(minimumDistance: 1)
.onChanged { value in
let time = phrase.start + Double((value.location.x) / pps)
model.trim(phrase.id, edge: edge, to: time)
}
)
.help(edge == .start ? "Arraste para cortar o começo (pula de palavra em palavra)"
: "Arraste para cortar o fim (pula de palavra em palavra)")
}
@ViewBuilder
private func phraseMenu(_ phrase: ReviewPhrase) -> some View {
Button("Tocar esta frase") {
model.goTo(phraseID: phrase.id)
model.playSelectedPhrase()
}
Button(phrase.active ? "Remover do corte" : "Trazer de volta") {
model.toggleActive(phrase.id)
}
Button("Adicionar zoom nesta frase") { model.addZoomForPhrase(phrase.id) }
Divider()
ForEach(0..<4, id: \.self) { level in
Button("Ênfase: \(EmphasisPalette.label(level))") {
model.setEmphasis(level, for: phrase.id)
}
}
Divider()
Button(phrase.isBackstage ? "Marcar como roteiro" : "Marcar como bastidor") {
model.setTrack(phrase.isBackstage ? ReviewPhrase.trackScript : ReviewPhrase.trackBackstage,
for: phrase.id)
}
if phrase.isTrimmed {
Divider()
Button("Desfazer corte da frase") { model.resetTrim(phrase.id) }
}
}
/// Per-word energy/emphasis, straight from the voice timeline — the closest
/// thing to a waveform without opening the audio again.
private var energyTrack: some View {
Canvas { context, size in
for phrase in model.phrases {
for word in phrase.words {
let start = x(word.start)
let barWidth = max(1, width(from: word.start, to: word.end) - 1)
let height = size.height * CGFloat(max(0.04, word.energy))
let rect = CGRect(x: start, y: size.height - height,
width: barWidth, height: height)
let color = word.emphasis >= 0.65 ? Color.pink
: word.emphasis >= 0.45 ? Color.orange
: Color.secondary
context.fill(Path(rect),
with: .color(color.opacity(phrase.active ? 0.6 : 0.2)))
}
}
}
.frame(width: contentWidth, height: energyHeight)
.background(RoundedRectangle(cornerRadius: 4).fill(Color.secondary.opacity(0.06)))
}
private var speakerTrack: some View {
stripTrack { phrase in
EmphasisPalette.speakerColor(phrase.speaker, among: model.speakers)
}
}
private var scriptTrack: some View {
stripTrack { phrase in phrase.isBackstage ? Color.gray : Color.mint }
}
private func stripTrack(_ color: @escaping (ReviewPhrase) -> Color) -> some View {
Canvas { context, size in
for phrase in model.phrases {
let rect = CGRect(x: x(phrase.start), y: 0,
width: max(1, width(from: phrase.start, to: phrase.end)),
height: size.height)
context.fill(Path(roundedRect: rect, cornerRadius: 2),
with: .color(color(phrase).opacity(phrase.active ? 0.7 : 0.2)))
}
}
.frame(width: contentWidth, height: stripHeight)
}
private var playhead: some View {
Rectangle()
.fill(Color.red)
.frame(width: 1.5)
.offset(x: x(model.currentTime))
.allowsHitTesting(false)
}
/// One gesture, two meanings, decided by whether the mouse moved: a click
/// parks the playhead, a drag marks in/out. Splitting them across separate
/// controls would mean choosing a tool before every action, which is
/// exactly the ceremony this screen is meant to avoid.
private var scrubGesture: some Gesture {
DragGesture(minimumDistance: 0)
.onChanged { value in
let from = Double(value.startLocation.x / pps)
let to = Double(value.location.x / pps)
if abs(value.translation.width) > 3 {
model.setRange(from: from, to: to)
model.seek(to: min(from, to))
} else {
model.clearRange()
model.seek(to: to)
}
}
}
/// The marked in/out, drawn over every track so the span reads against the
/// phrases and the energy at once.
private var rangeOverlay: some View {
Group {
if let span = model.rangeSpan {
Rectangle()
.fill(Color.accentColor.opacity(0.18))
.overlay(Rectangle().stroke(Color.accentColor.opacity(0.6), lineWidth: 1))
.frame(width: max(1, width(from: span.start, to: span.end)))
.offset(x: x(span.start))
.allowsHitTesting(false)
}
}
}
@ViewBuilder
private var timelineMenu: some View {
if model.hasRange, let span = model.rangeSpan {
Button("Adicionar zoom no trecho (\(secondsLabel(span.end - span.start)))") {
model.addZoomForRange()
}
Button("Tocar o trecho") { model.playRange(from: span.start, to: span.end) }
Button("Limpar seleção") { model.clearRange() }
} else {
Text("Arraste na timeline para marcar um trecho")
}
if let zoom = model.zoom(at: model.currentTime) {
Divider()
Button("Remover o zoom daqui") { model.removeZoom(zoom.id) }
}
}
private func secondsLabel(_ seconds: Double) -> String {
String(format: "%.1fs", seconds)
}
/// Punch-ins, on their own lane above the script: they are a second layer
/// over the same time, not a property of a phrase.
private var zoomTrack: some View {
ZStack(alignment: .topLeading) {
RoundedRectangle(cornerRadius: 3)
.fill(Color.secondary.opacity(0.06))
.frame(width: contentWidth, height: stripHeight + 6)
ForEach(model.zooms) { zoom in
RoundedRectangle(cornerRadius: 3)
.fill(Color.yellow.opacity(0.55))
.overlay(
Image(systemName: "plus.magnifyingglass")
.font(.system(size: 8)).foregroundStyle(.black.opacity(0.6))
)
.frame(width: max(6, width(from: zoom.start, to: zoom.end)),
height: stripHeight + 6)
.offset(x: x(zoom.start))
.help("Zoom marcado — \(secondsLabel(zoom.end - zoom.start)). A escala vem de Análise de Voz.")
.contextMenu {
Button("Remover este zoom") { model.removeZoom(zoom.id) }
}
}
}
.frame(width: contentWidth, height: stripHeight + 6, alignment: .topLeading)
}
/// Delivery emotion per phrase — the fourth signal to read against the text.
private var emotionTrack: some View {
stripTrack { phrase in
switch phrase.emotion {
case "excited": return .orange
case "tense": return .red
case "calm": return .blue
case "reflective": return .purple
default: return .secondary
}
}
}
// MARK: - Escala
private func x(_ time: Double) -> CGFloat { CGFloat(time) * pps }
private func width(from: Double, to: Double) -> CGFloat {
max(0, CGFloat(to - from) * pps)
}
/// Ruler spacing that keeps labels ~80pt apart at any zoom.
private func tickStep() -> Double {
let candidates: [Double] = [1, 2, 5, 10, 15, 30, 60, 120, 300, 600]
let wanted = 80 / Double(pps)
return candidates.first { $0 >= wanted } ?? 600
}
private func timecode(_ seconds: Double) -> String {
let total = Int(seconds.rounded(.down))
return String(format: "%02d:%02d", total / 60, total % 60)
}
}
+3 -3
View File
@@ -271,7 +271,7 @@ struct TranscriptionView: View {
Toggle("Marcar o que foi dito na timeline", isOn: $batchMarkers)
Divider()
Toggle("Exportar legendas SRT", isOn: $batchSubtitles)
Toggle("Gerar legenda comum (texto editável no FCP)", isOn: $batchSubtitles)
Divider()
batchOptionRow(
@@ -793,7 +793,7 @@ struct TranscriptionView: View {
if batchFillers { operations.append("remove_filler_words") }
if batchPhrases { operations.append("edit_by_transcript") }
if batchMarkers { operations.append("transcript_markers") }
if batchSubtitles { operations.append("export_srt") }
if batchSubtitles { operations.append("generate_plain_subtitles") }
// Runs last, on the timing already cut by any earlier steps (see the
// "Abrir no Final Cut Pro" fallback chain and exportSubtitles()'s own
// preference for `processedPath` — same reasoning).
@@ -834,7 +834,7 @@ struct TranscriptionView: View {
}
let nextPath = result?["path"] as? String ?? currentPath
if operation == "remove_silences" { processedPath = nextPath }
if operation == "export_srt" { subtitlePaths = result?["paths"] as? [String] ?? [] }
if operation == "generate_plain_subtitles" { subtitlePaths = [nextPath] }
if operation == "generate_dynamic_subtitles" { dynamicSubtitlesPath = nextPath }
processBatchStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder)
}
+68 -4
View File
@@ -23,6 +23,7 @@ struct VoiceAnalysisView: View {
} else {
energySection
emphasisSection
zoomSection
weightsSection
emotionSection
resetSection
@@ -90,6 +91,44 @@ struct VoiceAnalysisView: View {
}
}
private var zoomSection: some View {
Section {
sliderRow(
title: "Zoom na ênfase",
value: $config.zoomScale,
range: 1.0...3.0,
readout: "\(Int(config.zoomScale * 100))%",
help: "Fator aplicado nos punch-ins de ênfase. 130% equivale a escala 1,30 no Final Cut."
)
Picker("Movimento", selection: $config.zoomMode) {
Text("Zoom in e out").tag("in_out")
Text("Só zoom in").tag("in")
Text("Só zoom out").tag("out")
}
.onChange(of: config.zoomMode) { _, _ in save() }
sliderRow(
title: "Velocidade do zoom in",
value: $config.zoomEaseIn,
range: 0.05...2.0,
readout: String(format: "%.2fs", config.zoomEaseIn),
help: "Duração da entrada do zoom. Menor é mais rápido."
)
sliderRow(
title: "Velocidade do zoom out",
value: $config.zoomEaseOut,
range: 0.01...2.0,
readout: String(format: "%.2fs", config.zoomEaseOut),
help: "Duração da saída do zoom. Menor é mais seco."
)
} header: {
Text("Zoom de Ênfase")
} footer: {
Text("Esses valores viram o padrão para ações de zoom que não trouxerem scale/ease/ease_out no JSON da edição por voz.")
.font(.caption)
.foregroundStyle(.secondary)
}
}
// MARK: - Emoção
private var emotionSection: some View {
@@ -129,13 +168,14 @@ struct VoiceAnalysisView: View {
title: String,
value: Binding<Double>,
range: ClosedRange<Double> = 0...1,
readout: String? = nil,
help: String? = nil
) -> some View {
VStack(alignment: .leading, spacing: 2) {
HStack {
Text(title)
Spacer()
Text(String(format: "%.2f", value.wrappedValue))
Text(readout ?? String(format: "%.2f", value.wrappedValue))
.monospacedDigit()
.foregroundStyle(.secondary)
}
@@ -188,6 +228,10 @@ struct VoiceAnalysisConfig {
var weightDuration: Double
var emotionEnabled: Bool
var emotionSensitivity: Double
var zoomScale: Double
var zoomMode: String
var zoomEaseIn: Double
var zoomEaseOut: Double
static let defaults = VoiceAnalysisConfig(
energyThreshold: 0.5,
@@ -198,7 +242,11 @@ struct VoiceAnalysisConfig {
weightPause: 0.15,
weightDuration: 0.10,
emotionEnabled: false,
emotionSensitivity: 0.5
emotionSensitivity: 0.5,
zoomScale: 1.30,
zoomMode: "in_out",
zoomEaseIn: 0.25,
zoomEaseOut: 0.04
)
init(
@@ -210,7 +258,11 @@ struct VoiceAnalysisConfig {
weightPause: Double,
weightDuration: Double,
emotionEnabled: Bool,
emotionSensitivity: Double
emotionSensitivity: Double,
zoomScale: Double,
zoomMode: String,
zoomEaseIn: Double,
zoomEaseOut: Double
) {
self.energyThreshold = energyThreshold
self.emphasisThreshold = emphasisThreshold
@@ -221,6 +273,10 @@ struct VoiceAnalysisConfig {
self.weightDuration = weightDuration
self.emotionEnabled = emotionEnabled
self.emotionSensitivity = emotionSensitivity
self.zoomScale = zoomScale
self.zoomMode = zoomMode
self.zoomEaseIn = zoomEaseIn
self.zoomEaseOut = zoomEaseOut
}
/// Lê a resposta do bridge, caindo no padrão para qualquer campo ausente.
@@ -236,7 +292,11 @@ struct VoiceAnalysisConfig {
weightPause: weights["pause_before"] as? Double ?? defaults.weightPause,
weightDuration: weights["duration"] as? Double ?? defaults.weightDuration,
emotionEnabled: json["emotion_enabled"] as? Bool ?? defaults.emotionEnabled,
emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity
emotionSensitivity: json["emotion_sensitivity"] as? Double ?? defaults.emotionSensitivity,
zoomScale: json["zoom_scale"] as? Double ?? defaults.zoomScale,
zoomMode: json["zoom_mode"] as? String ?? defaults.zoomMode,
zoomEaseIn: json["zoom_ease_in"] as? Double ?? defaults.zoomEaseIn,
zoomEaseOut: json["zoom_ease_out"] as? Double ?? defaults.zoomEaseOut
)
}
@@ -253,6 +313,10 @@ struct VoiceAnalysisConfig {
],
"emotion_enabled": emotionEnabled,
"emotion_sensitivity": emotionSensitivity,
"zoom_scale": zoomScale,
"zoom_mode": zoomMode,
"zoom_ease_in": zoomEaseIn,
"zoom_ease_out": zoomEaseOut,
]
}
}
+808
View File
@@ -0,0 +1,808 @@
import SwiftUI
import AppKit
/// Guia passo a passo do fluxo completo: projeto → transcrição → análise de
/// voz → copiar para o chat e trazer as decisões → revisar as ênfases →
/// processamento final. Existe para que o usuário não precise entender a ordem
/// certa de botões espalhados em várias abas — cada etapa só libera a próxima
/// quando o passo anterior terminou, e a "ponte" com o chat (que hoje exigia
/// sair do app e escolher um arquivo na mão) vira copiar/colar assistido
/// dentro da própria tela.
enum WizardStep: Int, CaseIterable, Identifiable {
case projeto, transcricao, analise, exportarChat, revisar, finalizar, concluido
var id: Int { rawValue }
var titulo: String {
switch self {
case .projeto: return "Projeto"
case .transcricao: return "Transcrever"
case .analise: return "Analisar voz"
case .exportarChat: return "Decisões da IA"
case .revisar: return "Revisar ênfases"
case .finalizar: return "Processar"
case .concluido: return "Concluído"
}
}
}
struct WizardView: View {
@State private var step: WizardStep = .projeto
// Passo 1 — projeto
@State private var outputFolder: String?
@State private var projectPath: String?
@State private var catalog: Catalog?
// Passo 2 — transcrição
@State private var isTranscribing = false
@State private var transcribeProgress: Double = 0
@State private var transcribeStage = ""
@State private var transcribeResults: [TranscriptResult] = []
// Passo 3 — análise de voz
@State private var isAnalyzing = false
@State private var voiceTimelinePath: String?
@State private var voiceAnalysisMessage = ""
@State private var acousticsAvailable: Bool?
@State private var showVoiceTimelineReuseAlert = false
@State private var existingVoiceTimelinePath: String?
// Passo 4 — enviar ao chat e trazer as decisões de volta
@State private var copiedFeedback = ""
@State private var decisionsText = ""
@State private var isApplyingDecisions = false
@State private var appliedPath: String?
@State private var skippedVoiceEdit = false
// Passo 5 — revisar ênfases
@StateObject private var reviewModel = PhraseReviewModel()
@State private var reviewLoadedFor: String?
@State private var phraseReviewPath: String?
// Passo 6 — processamento final
@State private var finalSilences = true
@State private var finalFillers = false
@State private var finalSubtitles = true
@State private var finalDynamicSubtitles = false
@State private var isFinalizing = false
@State private var finalStatus = ""
@State private var finalPath: String?
@State private var errorMessage: String?
var body: some View {
VStack(spacing: 0) {
stepperHeader
.padding(.horizontal, 24)
.padding(.top, 20)
.padding(.bottom, 16)
Divider()
// A revisão é uma sala de edição, não um formulário: ela precisa da
// largura toda e rola por conta própria (timeline horizontal, lista
// vertical). As demais etapas continuam na coluna estreita, que é o
// que mantém um passo a passo legível.
if step == .revisar {
revisarStep
} else {
ScrollView {
VStack(alignment: .leading, spacing: 18) {
if let errorMessage, !errorMessage.isEmpty {
Label(errorMessage, systemImage: "exclamationmark.triangle.fill")
.foregroundStyle(.red)
.padding(.top, 4)
}
content
}
.padding(24)
.frame(maxWidth: 640, alignment: .leading)
.frame(maxWidth: .infinity)
}
}
Divider()
navFooter
.padding(.horizontal, 24)
.padding(.vertical, 16)
}
.task {
loadProjectConfig()
await loadCatalog()
}
.alert("Análise de voz já existe", isPresented: $showVoiceTimelineReuseAlert) {
Button("Usar existente") {
if let existingVoiceTimelinePath {
voiceTimelinePath = existingVoiceTimelinePath
voiceAnalysisMessage = "Reaproveitando análise existente: \(existingVoiceTimelinePath)"
}
}
Button("Reprocessar") {
analyzeVoice(forceReprocess: true)
}
Button("Cancelar", role: .cancel) {}
} message: {
Text("Já existe um arquivo voice_timeline para este projeto. Quer manter o processamento anterior para ganhar tempo?")
}
}
// MARK: - Cabeçalho com os passos
private var stepperHeader: some View {
HStack(spacing: 6) {
ForEach(WizardStep.allCases) { s in
HStack(spacing: 6) {
ZStack {
Circle()
.fill(colorFor(s))
.frame(width: 24, height: 24)
if s.rawValue < step.rawValue {
Image(systemName: "checkmark")
.font(.caption2.weight(.bold))
.foregroundStyle(.white)
} else {
Text("\(s.rawValue + 1)")
.font(.caption2.weight(.bold))
.foregroundStyle(s == step ? .white : .secondary)
}
}
Text(s.titulo)
.font(.caption)
.foregroundStyle(s == step ? .primary : .secondary)
.fontWeight(s == step ? .semibold : .regular)
}
if s != WizardStep.allCases.last {
Rectangle()
.fill(s.rawValue < step.rawValue ? Color.accentColor : Color.secondary.opacity(0.25))
.frame(height: 2)
.frame(maxWidth: .infinity)
}
}
}
}
private func colorFor(_ s: WizardStep) -> Color {
if s.rawValue < step.rawValue { return .accentColor }
if s == step { return .accentColor }
return Color.secondary.opacity(0.25)
}
// MARK: - Conteúdo por etapa
@ViewBuilder
private var content: some View {
switch step {
case .projeto: projetoStep
case .transcricao: transcricaoStep
case .analise: analiseStep
case .exportarChat: exportarChatStep
case .revisar: revisarStep
case .finalizar: finalizarStep
case .concluido: concluidoStep
}
}
private var projetoStep: some View {
VStack(alignment: .leading, spacing: 16) {
Text("1. Escolha o projeto").font(.title3.weight(.semibold))
Text("A pasta é onde tudo o que for gerado nesse fluxo fica salvo. O arquivo é o .fcpxml exportado do Final Cut Pro.")
.font(.callout).foregroundStyle(.secondary)
fieldRow(icon: "folder", label: outputFolder ?? "Nenhuma pasta selecionada", isSet: outputFolder != nil) {
pickOutputFolder()
}
fieldRow(icon: "doc.text", label: projectPath.map { URL(fileURLWithPath: $0).lastPathComponent } ?? "Nenhum arquivo selecionado", isSet: projectPath != nil) {
pickProjectFile()
}
if looksLikeGeneratedFile(projectPath) {
Label("Esse arquivo parece já ter sido processado por este fluxo (o nome tem um sufixo como \"_voice_edit\" ou \"_silence_removed\"). Rodar o wizard de novo em cima dele reaplica os cortes por cima de cortes já feitos. Selecione o .fcpxml original do Final Cut, a menos que a intenção seja mesmo reprocessar.",
systemImage: "exclamationmark.triangle.fill")
.font(.caption).foregroundStyle(.orange)
}
if (catalog?.installedCount ?? 0) == 0 {
Label("Nenhum modelo de transcrição instalado. Baixe um na aba \"Modelos\" antes de continuar.",
systemImage: "exclamationmark.triangle.fill")
.font(.caption).foregroundStyle(.orange)
}
}
}
private var transcricaoStep: some View {
VStack(alignment: .leading, spacing: 16) {
Text("2. Transcreva o áudio").font(.title3.weight(.semibold))
Text("Roda localmente com o modelo escolhido na aba Modelos. Vira a base de tudo que vem depois — o corte por voz, as legendas, os marcadores.")
.font(.callout).foregroundStyle(.secondary)
Button {
startTranscription()
} label: {
if isTranscribing {
HStack { ProgressView().controlSize(.small); Text(transcribeStage.isEmpty ? "Transcrevendo…" : transcribeStage) }
.frame(maxWidth: .infinity)
} else {
Label(transcribeResults.isEmpty ? "Transcrever" : "Transcrever novamente", systemImage: "waveform")
.frame(maxWidth: .infinity)
}
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
.disabled(isTranscribing || projectPath == nil || outputFolder == nil)
if isTranscribing {
VStack(alignment: .leading, spacing: 6) {
ProgressView(value: transcribeProgress)
Text("\(Int(transcribeProgress * 100))%").font(.caption).foregroundStyle(.secondary).monospacedDigit()
}
}
if !transcribeResults.isEmpty {
ForEach(transcribeResults, id: \.media) { r in
VStack(alignment: .leading, spacing: 4) {
HStack {
Image(systemName: "checkmark.circle.fill").foregroundStyle(.green)
Text(r.media).font(.body.weight(.medium))
Spacer()
Text("\(r.language) · \(r.words) palavras").font(.caption).foregroundStyle(.secondary)
}
Text(r.preview).font(.caption).foregroundStyle(.secondary).lineLimit(2)
}
.padding(12)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
}
}
}
}
private var analiseStep: some View {
VStack(alignment: .leading, spacing: 16) {
Text("3. Analise a voz").font(.title3.weight(.semibold))
Text("Gera o JSON com transcrição, locutor e intensidade (pitch/energia/ritmo) por palavra — é esse arquivo que o chat lê para decidir o que cortar. Não corta nada sozinho.")
.font(.callout).foregroundStyle(.secondary)
Button {
analyzeVoice()
} label: {
if isAnalyzing {
HStack { ProgressView().controlSize(.small); Text("Analisando…") }.frame(maxWidth: .infinity)
} else {
Label(voiceTimelinePath == nil ? "Analisar voz" : "Analisar novamente", systemImage: "waveform.badge.magnifyingglass")
.frame(maxWidth: .infinity)
}
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
.disabled(isAnalyzing || projectPath == nil || outputFolder == nil)
if let voiceTimelinePath {
VStack(alignment: .leading, spacing: 6) {
Label("Análise pronta", systemImage: "checkmark.circle.fill").foregroundStyle(.green)
Text(voiceTimelinePath).font(.caption).foregroundStyle(.secondary).lineLimit(1).truncationMode(.middle)
}
.padding(12)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
if acousticsAvailable == false {
VStack(alignment: .leading, spacing: 4) {
Label("Sem análise acústica real", systemImage: "exclamationmark.triangle.fill")
.font(.caption.weight(.semibold)).foregroundStyle(.orange)
Text("Falta o componente \"librosa\" — os cortes ainda são decididos pelo texto, mas o chat não vai propor zoom com confiança. Instale em Avançado → Modelos → \"Análise Acústica\", e refaça esta etapa depois.")
.font(.caption).foregroundStyle(.secondary)
}
.padding(12)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.orange.opacity(0.08)))
}
}
}
}
private var exportarChatStep: some View {
VStack(alignment: .leading, spacing: 16) {
Text("4. Envie para o chat decidir os cortes").font(.title3.weight(.semibold))
Text("Esta é a única etapa manual que sobra: o julgamento de qual tomada usar, onde dar zoom e o que escrever na tela é feito pela IA numa conversa, não por um botão. Copie abaixo, cole numa sessão do Claude e peça pra rodar a skill \"editar-por-voz\".")
.font(.callout).foregroundStyle(.secondary)
if let voiceTimelinePath {
Button {
copyForChat(path: voiceTimelinePath)
} label: {
Label("Copiar para colar no chat", systemImage: "doc.on.clipboard")
.frame(maxWidth: .infinity)
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
if !copiedFeedback.isEmpty {
Label(copiedFeedback, systemImage: "checkmark.circle.fill")
.font(.caption).foregroundStyle(.green)
}
VStack(alignment: .leading, spacing: 8) {
Text("O que é copiado").font(.caption.weight(.semibold)).foregroundStyle(.secondary)
Text("Um pedido pronto + o conteúdo de \(URL(fileURLWithPath: voiceTimelinePath).lastPathComponent), já formatado. É só colar (⌘V) numa conversa com o Claude.")
.font(.caption).foregroundStyle(.secondary)
}
.padding(12)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
Divider().padding(.vertical, 4)
Text("Cole aqui o que o chat devolveu").font(.callout.weight(.semibold))
Text("Na próxima etapa essas decisões aparecem já marcadas na timeline, frase por frase, para você lapidar.")
.font(.caption).foregroundStyle(.secondary)
HStack {
Button {
if let s = NSPasteboard.general.string(forType: .string) {
decisionsText = s
}
} label: {
Label("Colar da área de transferência", systemImage: "list.clipboard")
}
Spacer()
if !decisionsText.isEmpty {
Label(jsonIsValid ? "JSON válido" : "JSON inválido",
systemImage: jsonIsValid ? "checkmark.circle.fill" : "xmark.circle.fill")
.font(.caption)
.foregroundStyle(jsonIsValid ? .green : .red)
}
}
TextEditor(text: $decisionsText)
.font(.system(.caption, design: .monospaced))
.frame(minHeight: 140)
.padding(8)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
.overlay(RoundedRectangle(cornerRadius: 8).stroke(Color.secondary.opacity(0.2)))
Button {
applyDecisions()
} label: {
if isApplyingDecisions {
HStack { ProgressView().controlSize(.small); Text("Aplicando…") }
.frame(maxWidth: .infinity)
} else {
Label("Aplicar decisões", systemImage: "checkmark.seal")
.frame(maxWidth: .infinity)
}
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
.disabled(isApplyingDecisions || !jsonIsValid)
if let appliedPath {
Label("Decisões aplicadas — \(URL(fileURLWithPath: appliedPath).lastPathComponent)",
systemImage: "checkmark.circle.fill")
.font(.caption).foregroundStyle(.green)
}
Divider()
Button("Pular esta etapa (revisar as ênfases direto, sem passar pela IA)") {
skippedVoiceEdit = true
appliedPath = nil
decisionsText = ""
}
.buttonStyle(.plain)
.font(.caption)
.foregroundStyle(.secondary)
} else {
Label("Volte ao passo anterior e rode a análise de voz primeiro.", systemImage: "exclamationmark.triangle.fill")
.font(.caption).foregroundStyle(.orange)
}
}
}
/// Etapa 5 — a sala de edição. Diferente das outras, não é um formulário
/// dentro da coluna do assistente: ocupa a janela toda e se carrega sozinha
/// na primeira vez que aparece para aquela análise de voz.
private var revisarStep: some View {
Group {
if voiceTimelinePath != nil {
PhraseReviewView(model: reviewModel)
} else {
VStack(spacing: 8) {
Label("Volte ao passo 3 e rode a análise de voz primeiro.",
systemImage: "exclamationmark.triangle.fill")
.foregroundStyle(.orange)
}
.frame(maxWidth: .infinity, maxHeight: .infinity)
}
}
.onAppear { loadReviewIfNeeded() }
}
private var finalizarStep: some View {
VStack(alignment: .leading, spacing: 16) {
Text("6. Finalize o corte").font(.title3.weight(.semibold))
Text("Últimos passos automáticos, sem decisão envolvida — rodam com os parâmetros já configurados na aba \"Análise de Voz\" / \"Legendas Dinâmicas\".")
.font(.callout).foregroundStyle(.secondary)
Toggle("Remover silêncios do áudio", isOn: $finalSilences)
Toggle("Remover palavras de preenchimento", isOn: $finalFillers)
Toggle("Gerar legenda comum (texto editável no FCP)", isOn: $finalSubtitles)
Toggle("Gerar legendas dinâmicas (estilo configurado na aba própria)", isOn: $finalDynamicSubtitles)
Button {
finalizeProcessing()
} label: {
if isFinalizing {
HStack { ProgressView().controlSize(.small); Text(finalStatus.isEmpty ? "Processando…" : finalStatus) }
.frame(maxWidth: .infinity)
} else {
Label("Processar", systemImage: "play.fill").frame(maxWidth: .infinity)
}
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
.disabled(isFinalizing || (!finalSilences && !finalFillers && !finalSubtitles && !finalDynamicSubtitles))
if !finalStatus.isEmpty && !isFinalizing {
Text(finalStatus).font(.caption).foregroundStyle(.secondary)
}
}
}
private var concluidoStep: some View {
VStack(alignment: .leading, spacing: 16) {
Label("Concluído", systemImage: "checkmark.seal.fill")
.font(.title3.weight(.semibold))
.foregroundStyle(.green)
if let finalPath {
Text(finalPath).font(.caption).foregroundStyle(.secondary).lineLimit(1).truncationMode(.middle)
HStack {
Button("Abrir no Final Cut Pro") { NSWorkspace.shared.open(URL(fileURLWithPath: finalPath)) }
.buttonStyle(.borderedProminent)
Button("Mostrar no Finder") {
NSWorkspace.shared.activateFileViewerSelecting([URL(fileURLWithPath: finalPath)])
}
}
}
Divider().padding(.vertical, 8)
Button("Começar outro projeto") { resetWizard() }
}
}
// MARK: - Navegação
private var navFooter: some View {
HStack {
if step != .projeto && step != .concluido {
Button("Voltar") { goBack() }
}
Spacer()
if step != .concluido {
Button(step == .finalizar ? "Concluir" : "Continuar") { goNext() }
.buttonStyle(.borderedProminent)
.disabled(!canAdvance)
}
}
}
private var canAdvance: Bool {
switch step {
case .projeto: return outputFolder != nil && projectPath != nil
case .transcricao: return !transcribeResults.isEmpty
case .analise: return voiceTimelinePath != nil
case .exportarChat: return appliedPath != nil || skippedVoiceEdit
// Revisar é opcional: a sugestão da IA já é utilizável como veio, então
// o botão nunca trava aqui — o passo existe para lapidar, não para
// exigir mais uma confirmação.
case .revisar: return true
case .finalizar: return finalPath != nil && !isFinalizing
case .concluido: return false
}
}
private func goNext() {
guard let next = WizardStep(rawValue: step.rawValue + 1) else { return }
// Sair da revisão grava o que foi decidido (e as ações derivadas dela)
// ao lado da análise de voz. Nada é renderizado aqui: a etapa 6 é que
// lê esse arquivo para dar zoom e legenda dinâmica só nas ênfases.
if step == .revisar {
reviewModel.save { path in
phraseReviewPath = path
}
}
step = next
}
private func goBack() {
guard let prev = WizardStep(rawValue: step.rawValue - 1) else { return }
step = prev
}
private func resetWizard() {
step = .projeto
transcribeResults = []
voiceTimelinePath = nil
voiceAnalysisMessage = ""
decisionsText = ""
appliedPath = nil
skippedVoiceEdit = false
reviewLoadedFor = nil
phraseReviewPath = nil
finalStatus = ""
finalPath = nil
errorMessage = nil
}
// MARK: - Componentes auxiliares
@ViewBuilder
private func fieldRow(icon: String, label: String, isSet: Bool, action: @escaping () -> Void) -> some View {
HStack {
Image(systemName: icon).foregroundStyle(isSet ? .primary : .secondary)
Text(label).lineLimit(1).truncationMode(.middle).foregroundStyle(isSet ? .primary : .secondary)
Spacer()
Button("Escolher…", action: action)
}
.padding(12)
.background(RoundedRectangle(cornerRadius: 8).fill(Color.secondary.opacity(0.06)))
}
/// Todo output do fluxo carrega um destes sufixos no nome (ver
/// `_derived_output` / suffixes usados por `apply_voice_actions`,
/// `remove_silences`, `generate_dynamic_subtitles` em
/// `admin/models_api.py`). Selecionar um deles como "o projeto" no passo
/// 1 é o erro que gerou arquivos como `_voice_edit_voice_edit_...`: os
/// cortes de voz assumem timestamps da mídia ORIGINAL, então reaplicá-los
/// sobre um arquivo já cortado desloca tudo silenciosamente.
private static let generatedSuffixes = [
"_voice_edit", "_silence_removed", "_dynamic_subtitles",
"_transcript_edit", "_fillers_removed", "_markers",
]
private func looksLikeGeneratedFile(_ path: String?) -> Bool {
guard let path else { return false }
let stem = URL(fileURLWithPath: path).deletingPathExtension().lastPathComponent
return Self.generatedSuffixes.contains { stem.contains($0) }
}
private var jsonIsValid: Bool {
guard let data = decisionsText.data(using: .utf8), !decisionsText.trimmingCharacters(in: .whitespacesAndNewlines).isEmpty else { return false }
return (try? JSONSerialization.jsonObject(with: data)) != nil
}
// MARK: - Ações — Python bridge
private func loadProjectConfig() {
PythonBridge.call(command: "project_config") { result, _ in
DispatchQueue.main.async {
guard let result, result["ok"] as? Bool == true else { return }
if let folder = result["folder"] as? String, !folder.isEmpty { outputFolder = folder }
if let file = result["file"] as? String, !file.isEmpty { projectPath = file }
}
}
}
private func loadCatalog() async {
PythonBridge.call(command: "catalog") { result, _ in
DispatchQueue.main.async {
if let result { catalog = Catalog(json: result) }
}
}
}
private func pickOutputFolder() {
let panel = NSOpenPanel()
panel.canChooseFiles = false
panel.canChooseDirectories = true
panel.allowsMultipleSelection = false
panel.prompt = "Usar esta pasta"
panel.message = "Escolha a pasta onde os resultados serão salvos."
if panel.runModal() == .OK, let url = panel.url {
outputFolder = url.path
PythonBridge.call(command: "set_project_config", arguments: ["folder": url.path]) { _, _ in }
}
}
private func pickProjectFile() {
let panel = NSOpenPanel()
panel.canChooseFiles = true
panel.canChooseDirectories = false
panel.allowsMultipleSelection = false
panel.prompt = "Selecionar"
panel.message = "Selecione o arquivo (.fcpxml) ou o bundle (.fcpxmld) exportado pelo Final Cut Pro."
if panel.runModal() == .OK, let url = panel.url {
let ext = url.pathExtension.lowercased()
if ext == "fcpxml" || ext == "fcpxmld" || ext == "xml" {
projectPath = url.path
PythonBridge.call(command: "set_project_config", arguments: ["file": url.path]) { _, _ in }
} else {
errorMessage = "Selecione um arquivo .fcpxml, .fcpxmld ou .xml do Final Cut Pro."
}
}
}
private func startTranscription() {
guard let projectPath, let outputFolder else { return }
isTranscribing = true
errorMessage = nil
transcribeResults = []
transcribeProgress = 0
PythonBridge.run(command: "transcribe", arguments: ["path": projectPath, "output_dir": outputFolder]) { obj in
DispatchQueue.main.async {
let type = obj["type"] as? String
if type == "progress" {
transcribeProgress = (obj["fraction"] as? NSNumber)?.doubleValue ?? 0
transcribeStage = obj["stage"] as? String ?? ""
} else if type == "error" {
errorMessage = obj["message"] as? String ?? "Erro na transcrição."
} else if type == "result", let arr = obj["transcripts"] as? [[String: Any]] {
transcribeResults = arr.map(TranscriptResult.init)
}
}
} completion: { code, err in
DispatchQueue.main.async {
isTranscribing = false
transcribeProgress = 1
if code != 0 && transcribeResults.isEmpty {
errorMessage = err ?? "A transcrição falhou."
}
}
}
}
private func analyzeVoice(forceReprocess: Bool = false) {
guard let projectPath, let outputFolder else { return }
isAnalyzing = true
errorMessage = nil
PythonBridge.call(command: "analyze_voice", arguments: [
"path": projectPath,
"output_dir": outputFolder,
"force_reprocess": forceReprocess,
]) { result, err in
DispatchQueue.main.async {
isAnalyzing = false
guard result?["ok"] as? Bool == true else {
errorMessage = result?["error"] as? String ?? err ?? "Falha ao analisar a voz."
return
}
if result?["reused"] as? Bool == true, !forceReprocess {
let timelines = result?["timelines"] as? [String] ?? []
existingVoiceTimelinePath = timelines.first ?? extractPath(from: result?["message"] as? String ?? "", marker: "**Timeline JSON**:")
showVoiceTimelineReuseAlert = true
return
}
let message = result?["message"] as? String ?? ""
voiceAnalysisMessage = message
if let path = extractPath(from: message, marker: "**Timeline JSON**:") {
voiceTimelinePath = path
} else {
voiceTimelinePath = nil
// ok:true não garante que a análise gerou timeline — se
// não houver fala detectável no áudio, o Python volta com
// sucesso mas sem "Timeline JSON" na mensagem. Sem isso
// aqui, a etapa parecia não fazer nada.
errorMessage = "A análise terminou mas não encontrou fala reconhecível no áudio. Mensagem do motor: " + (message.isEmpty ? "(vazia)" : message)
}
checkAcoustics()
}
}
}
/// A ênfase de voz (energia/tom) depende do `librosa`, dependência
/// opcional. Sem ela, a análise ainda transcreve e corta pelo texto,
/// mas nunca deveria propor zoom — por isso avisamos aqui, no ponto
/// onde o usuário sentiria falta, em vez de só na aba Modelos.
private func checkAcoustics() {
PythonBridge.call(command: "acoustics_capability") { result, _ in
DispatchQueue.main.async {
guard let result, result["ok"] as? Bool == true else { return }
acousticsAvailable = result["available"] as? Bool
}
}
}
/// Localiza uma linha markdown do tipo "- **Marker**: valor" (usado nas
/// mensagens do bridge Python) e devolve o valor. Aceita o marcador de
/// lista "- " opcional antes dos asteriscos.
private func extractPath(from message: String, marker: String) -> String? {
for line in message.split(separator: "\n") {
var trimmed = Substring(line.trimmingCharacters(in: .whitespaces))
if trimmed.hasPrefix("- ") { trimmed = trimmed.dropFirst(2) }
if trimmed.hasPrefix(marker) {
return trimmed.dropFirst(marker.count).trimmingCharacters(in: .whitespaces)
}
}
return nil
}
private func copyForChat(path: String) {
guard let content = try? String(contentsOfFile: path, encoding: .utf8) else {
errorMessage = "Não foi possível ler \(path)."
return
}
let prompt = """
Use a skill "editar-por-voz" para decidir os cortes deste projeto a partir da timeline de voz abaixo. Devolva só o JSON de decisões (cortes, zooms, textos, marcadores) pronto para eu colar de volta no app.
```json
\(content)
```
"""
let pasteboard = NSPasteboard.general
pasteboard.clearContents()
pasteboard.setString(prompt, forType: .string)
copiedFeedback = "Copiado — cole (⌘V) numa conversa com o Claude."
}
/// Monta a revisão uma vez por análise de voz. Voltar e avançar de novo não
/// recarrega: isso jogaria fora as edições manuais em silêncio, que é
/// exatamente o que esta tela existe para preservar.
private func loadReviewIfNeeded() {
guard let voiceTimelinePath, reviewLoadedFor != voiceTimelinePath else { return }
reviewLoadedFor = voiceTimelinePath
// A pasta do projeto e a do .fcpxml entram como onde procurar a mídia:
// a análise de voz guarda só o nome do arquivo, não o caminho.
reviewModel.load(
voiceTimelinePath: voiceTimelinePath,
decisionsJSON: decisionsText,
outputFolder: outputFolder,
mediaFolder: projectPath.map { URL(fileURLWithPath: $0).deletingLastPathComponent().path }
)
if let projectPath { reviewModel.loadProjectFormat(projectPath: projectPath) }
}
private func applyDecisions() {
guard let projectPath, let outputFolder,
let data = decisionsText.data(using: .utf8),
let parsed = try? JSONSerialization.jsonObject(with: data) else { return }
isApplyingDecisions = true
errorMessage = nil
PythonBridge.call(command: "apply_voice_actions", arguments: [
"path": projectPath,
"output_dir": outputFolder,
"actions": parsed,
]) { result, err in
DispatchQueue.main.async {
isApplyingDecisions = false
guard result?["ok"] as? Bool == true else {
errorMessage = result?["error"] as? String ?? err ?? "Falha ao aplicar as decisões."
return
}
appliedPath = result?["path"] as? String ?? projectPath
skippedVoiceEdit = false
}
}
}
private func finalizeProcessing() {
guard let outputFolder else { return }
let startPath = appliedPath ?? projectPath
guard let startPath else { return }
var operations: [String] = []
if finalSilences { operations.append("remove_silences") }
if finalFillers { operations.append("remove_filler_words") }
if finalSubtitles { operations.append("generate_plain_subtitles") }
if finalDynamicSubtitles { operations.append("generate_dynamic_subtitles") }
guard !operations.isEmpty else { return }
isFinalizing = true
errorMessage = nil
finalStatus = "Iniciando…"
finalizeStep(operations, index: 0, currentPath: startPath, outputFolder: outputFolder)
}
private func finalizeStep(_ operations: [String], index: Int, currentPath: String, outputFolder: String) {
guard index < operations.count else {
isFinalizing = false
finalStatus = "Processamento concluído."
finalPath = currentPath
return
}
let operation = operations[index]
finalStatus = "Processando: \(operation)…"
PythonBridge.call(command: operation, arguments: ["path": currentPath, "output_dir": outputFolder]) { result, err in
DispatchQueue.main.async {
guard result?["ok"] as? Bool == true else {
isFinalizing = false
errorMessage = result?["error"] as? String ?? err ?? "Falha em \(operation)."
finalStatus = "Processamento interrompido."
return
}
let nextPath = result?["path"] as? String ?? currentPath
finalizeStep(operations, index: index + 1, currentPath: nextPath, outputFolder: outputFolder)
}
}
}
}