Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
414 lines
16 KiB
Swift
414 lines
16 KiB
Swift
import AVFoundation
|
||
import SwiftUI
|
||
|
||
/// The video surface, as a plain `AVPlayerLayer` in an `NSView`.
|
||
///
|
||
/// AVKit's `VideoPlayer` would be the obvious choice and is a trap here: this
|
||
/// app is built by invoking `swiftc` directly (see `MacApp/build_app.sh`), and
|
||
/// `_AVKit_SwiftUI` aborts at launch instantiating its generic metadata under
|
||
/// that build. A player layer needs only AVFoundation, which links cleanly —
|
||
/// and the transport controls live in the timeline's own toolbar anyway, so
|
||
/// nothing is lost by dropping AVKit's chrome.
|
||
private struct PlayerSurface: NSViewRepresentable {
|
||
let player: AVPlayer
|
||
/// When true the frame is filled and cropped instead of letterboxed — used
|
||
/// to preview horizontal footage inside a vertical delivery frame.
|
||
var fills: Bool
|
||
|
||
func makeNSView(context: Context) -> PlayerLayerView {
|
||
let view = PlayerLayerView()
|
||
view.player = player
|
||
view.fills = fills
|
||
return view
|
||
}
|
||
|
||
func updateNSView(_ view: PlayerLayerView, context: Context) {
|
||
if view.player !== player { view.player = player }
|
||
view.fills = fills
|
||
}
|
||
}
|
||
|
||
final class PlayerLayerView: NSView {
|
||
private let playerLayer = AVPlayerLayer()
|
||
|
||
var player: AVPlayer? {
|
||
get { playerLayer.player }
|
||
set { playerLayer.player = newValue }
|
||
}
|
||
|
||
var fills: Bool = false {
|
||
didSet { playerLayer.videoGravity = fills ? .resizeAspectFill : .resizeAspect }
|
||
}
|
||
|
||
override init(frame frameRect: NSRect) {
|
||
super.init(frame: frameRect)
|
||
wantsLayer = true
|
||
layer = CALayer()
|
||
layer?.backgroundColor = NSColor.black.cgColor
|
||
playerLayer.videoGravity = .resizeAspect
|
||
layer?.addSublayer(playerLayer)
|
||
}
|
||
|
||
required init?(coder: NSCoder) {
|
||
super.init(coder: coder)
|
||
wantsLayer = true
|
||
layer = CALayer()
|
||
playerLayer.videoGravity = .resizeAspect
|
||
layer?.addSublayer(playerLayer)
|
||
}
|
||
|
||
override func layout() {
|
||
super.layout()
|
||
playerLayer.frame = bounds
|
||
}
|
||
}
|
||
|
||
/// The wizard's emphasis-review step, laid out like an editing room: preview on
|
||
/// top, timeline across the bottom, and the script as an inspector down the
|
||
/// right side.
|
||
///
|
||
/// The arrangement is the point. Every decision here is about a *sentence*, so
|
||
/// the same phrase has to be legible in all three places at once — a block on
|
||
/// the timeline, a line of text in the inspector, and a moment in the preview.
|
||
/// Selecting in any one of them selects in the other two.
|
||
struct PhraseReviewView: View {
|
||
@ObservedObject var model: PhraseReviewModel
|
||
|
||
var body: some View {
|
||
VSplitView {
|
||
HSplitView {
|
||
previewPane
|
||
.frame(minWidth: 320, idealWidth: 640)
|
||
inspectorPane
|
||
.frame(minWidth: 300, idealWidth: 360, maxWidth: 520)
|
||
}
|
||
.frame(minHeight: 240)
|
||
|
||
TimelineTracksView(model: model)
|
||
.frame(minHeight: 190, idealHeight: 210)
|
||
}
|
||
.overlay { if model.isLoading { loadingOverlay } }
|
||
.focusable()
|
||
.onKeyPress(.space) { model.togglePlay(); return .handled }
|
||
.onKeyPress(.return) { model.playSelectedPhrase(); return .handled }
|
||
.onKeyPress(.leftArrow) { model.selectNeighbour(-1); return .handled }
|
||
.onKeyPress(.rightArrow) { model.selectNeighbour(1); return .handled }
|
||
.onKeyPress(characters: .decimalDigits) { press in
|
||
guard let level = Int(press.characters), (0...3).contains(level),
|
||
let selection = model.selection else { return .ignored }
|
||
model.setEmphasis(level, for: selection)
|
||
return .handled
|
||
}
|
||
}
|
||
|
||
private var loadingOverlay: some View {
|
||
ZStack {
|
||
Color(nsColor: .windowBackgroundColor).opacity(0.85)
|
||
VStack(spacing: 10) {
|
||
ProgressView()
|
||
Text("Montando a revisão…").font(.callout).foregroundStyle(.secondary)
|
||
}
|
||
}
|
||
}
|
||
|
||
// MARK: - Preview
|
||
|
||
private var previewPane: some View {
|
||
VStack(spacing: 0) {
|
||
if let player = model.player {
|
||
// The footage here is usually vertical. Sizing the surface to
|
||
// the take's own aspect keeps a 9:16 frame as tall as the pane
|
||
// allows instead of shrinking it to fit a horizontal box.
|
||
// Framed to what the project delivers, not to what the camera
|
||
// recorded: these takes are shot horizontal and cut vertical,
|
||
// so the raw frame would show material the audience never sees.
|
||
ZStack {
|
||
Color.black
|
||
PlayerSurface(player: player, fills: model.isCropping)
|
||
.aspectRatio(model.previewAspect, contentMode: .fit)
|
||
.clipped()
|
||
}
|
||
.overlay(alignment: .topTrailing) { framingBadge }
|
||
} else {
|
||
ZStack {
|
||
Color.black.opacity(0.85)
|
||
VStack(spacing: 10) {
|
||
Image(systemName: "film.stack")
|
||
.font(.system(size: 28)).foregroundStyle(.secondary)
|
||
Text(model.source.isEmpty
|
||
? "A análise de voz não registrou qual mídia foi usada."
|
||
: "Não achei \(model.source) na pasta do projeto.")
|
||
.font(.callout).foregroundStyle(.secondary)
|
||
Text("A revisão funciona igual sem o preview — ele só ajuda a conferir o corte.")
|
||
.font(.caption).foregroundStyle(.tertiary)
|
||
Button("Localizar a mídia…") { pickMedia() }
|
||
.buttonStyle(.bordered)
|
||
}
|
||
.multilineTextAlignment(.center)
|
||
.padding(.horizontal, 24)
|
||
}
|
||
}
|
||
Divider()
|
||
summaryBar
|
||
}
|
||
}
|
||
|
||
private var summaryBar: some View {
|
||
HStack(spacing: 16) {
|
||
summaryItem("text.quote", "\(model.phrases.count) frases")
|
||
summaryItem("sparkles", "\(model.emphasisCount) com ênfase")
|
||
summaryItem("scissors", "\(model.removedCount) fora do corte")
|
||
summaryItem("clock", durationLabel(model.keptDuration))
|
||
if !model.zooms.isEmpty {
|
||
summaryItem("plus.magnifyingglass", "\(model.zooms.count) zooms")
|
||
}
|
||
Spacer()
|
||
if let phrase = selectedPhrase, !phrase.reason.isEmpty {
|
||
Label(phrase.reason, systemImage: "brain")
|
||
.font(.caption).foregroundStyle(.secondary)
|
||
.lineLimit(1).truncationMode(.tail)
|
||
}
|
||
}
|
||
.padding(.horizontal, 14)
|
||
.padding(.vertical, 8)
|
||
}
|
||
|
||
private func summaryItem(_ icon: String, _ text: String) -> some View {
|
||
Label(text, systemImage: icon).font(.caption).foregroundStyle(.secondary)
|
||
}
|
||
|
||
private func durationLabel(_ seconds: Double) -> String {
|
||
String(format: "%02d:%02d finais", Int(seconds) / 60, Int(seconds) % 60)
|
||
}
|
||
|
||
/// Says which frame is on screen, and lets the editor flip to the raw take.
|
||
/// Without it a centred crop looks like the footage itself, and someone
|
||
/// would judge framing on an approximation without knowing it.
|
||
@ViewBuilder
|
||
private var framingBadge: some View {
|
||
if model.projectAspect != nil, abs((model.projectAspect ?? 0) - model.videoAspect) > 0.01 {
|
||
Button {
|
||
model.matchProjectFraming.toggle()
|
||
} label: {
|
||
Label(model.matchProjectFraming ? "Enquadramento do projeto" : "Mídia original",
|
||
systemImage: model.matchProjectFraming ? "crop" : "rectangle.expand.vertical")
|
||
.font(.caption2)
|
||
}
|
||
.buttonStyle(.borderless)
|
||
.padding(6)
|
||
.background(Capsule().fill(.black.opacity(0.45)))
|
||
.foregroundStyle(.white)
|
||
.padding(8)
|
||
.help("A fonte é horizontal e o projeto é vertical — o preview mostra o corte central aproximado. O enquadramento real de cada clipe vem do Final Cut.")
|
||
}
|
||
}
|
||
|
||
private func pickMedia() {
|
||
let panel = NSOpenPanel()
|
||
panel.canChooseFiles = true
|
||
panel.canChooseDirectories = false
|
||
panel.allowsMultipleSelection = false
|
||
panel.prompt = "Usar esta mídia"
|
||
panel.message = model.source.isEmpty
|
||
? "Escolha o arquivo de vídeo desta gravação."
|
||
: "Escolha onde está \(model.source)."
|
||
if panel.runModal() == .OK, let url = panel.url {
|
||
model.useMedia(at: url.path)
|
||
}
|
||
}
|
||
|
||
private var selectedPhrase: ReviewPhrase? {
|
||
guard let selection = model.selection else { return nil }
|
||
return model.phrases.first { $0.id == selection }
|
||
}
|
||
|
||
// MARK: - Inspector de frases
|
||
|
||
private var inspectorPane: some View {
|
||
VStack(spacing: 0) {
|
||
inspectorHeader
|
||
Divider()
|
||
List(selection: $model.selection) {
|
||
ForEach($model.phrases) { $phrase in
|
||
PhraseRow(phrase: $phrase, model: model)
|
||
.tag(phrase.id)
|
||
}
|
||
}
|
||
.listStyle(.inset)
|
||
.onChange(of: model.selection) { _, newValue in
|
||
if let newValue { model.goTo(phraseID: newValue) }
|
||
}
|
||
}
|
||
}
|
||
|
||
private var inspectorHeader: some View {
|
||
VStack(alignment: .leading, spacing: 6) {
|
||
Text("Frases").font(.headline)
|
||
Text("Só as frases com ênfase recebem zoom e legenda dinâmica. O resto fica com legenda comum.")
|
||
.font(.caption).foregroundStyle(.secondary)
|
||
if !model.emotionAvailable {
|
||
Label("Emoção da fala não foi detectada nesta análise — ligue em Avançado → Análise de Voz e refaça o passo 3.",
|
||
systemImage: "waveform.path.ecg")
|
||
.font(.caption2).foregroundStyle(.secondary)
|
||
}
|
||
HStack(spacing: 8) {
|
||
Button("Limpar ênfases") { model.setEmphasisForAll(0) }
|
||
.buttonStyle(.link).font(.caption)
|
||
Spacer()
|
||
Text("0–3 no teclado · ← → navega")
|
||
.font(.caption2).foregroundStyle(.secondary)
|
||
}
|
||
}
|
||
.padding(12)
|
||
}
|
||
}
|
||
|
||
/// One phrase in the inspector: the line as it will be said, plus every
|
||
/// decision attached to it. Kept in one row on purpose — jumping to a separate
|
||
/// detail pane to set a toggle would double the clicks on the most repeated
|
||
/// action in the screen.
|
||
private struct PhraseRow: View {
|
||
@Binding var phrase: ReviewPhrase
|
||
@ObservedObject var model: PhraseReviewModel
|
||
@State private var isEditing = false
|
||
|
||
var body: some View {
|
||
VStack(alignment: .leading, spacing: 6) {
|
||
HStack(spacing: 6) {
|
||
Text(phrase.timecode)
|
||
.font(.system(.caption2, design: .monospaced))
|
||
.foregroundStyle(.secondary)
|
||
if phrase.takeBoundary {
|
||
Image(systemName: "scissors.badge.ellipsis")
|
||
.font(.caption2).foregroundStyle(.orange)
|
||
.help("Nova tomada começa aqui")
|
||
}
|
||
if phrase.isTrimmed {
|
||
Image(systemName: "arrow.left.and.right.square")
|
||
.font(.caption2).foregroundStyle(.blue)
|
||
.help("Frase cortada nas pontas")
|
||
}
|
||
if model.emotionAvailable {
|
||
emotionChip
|
||
}
|
||
Spacer()
|
||
Toggle("", isOn: $phrase.active)
|
||
.toggleStyle(.switch)
|
||
.controlSize(.mini)
|
||
.labelsHidden()
|
||
.help(phrase.active ? "No corte" : "Fora do corte")
|
||
}
|
||
|
||
if isEditing {
|
||
TextField("Texto da frase", text: $phrase.text, axis: .vertical)
|
||
.textFieldStyle(.roundedBorder)
|
||
.font(.callout)
|
||
.onSubmit { isEditing = false }
|
||
} else {
|
||
Text(phrase.text.isEmpty ? "(sem texto)" : phrase.text)
|
||
.font(.callout)
|
||
.foregroundStyle(phrase.active ? .primary : .secondary)
|
||
.strikethrough(!phrase.active)
|
||
.onTapGesture(count: 2) { isEditing = true }
|
||
}
|
||
|
||
HStack(spacing: 8) {
|
||
Picker("", selection: $phrase.emphasis) {
|
||
ForEach(0..<4, id: \.self) { level in
|
||
Text(EmphasisPalette.label(level)).tag(level)
|
||
}
|
||
}
|
||
.pickerStyle(.segmented)
|
||
.controlSize(.mini)
|
||
.labelsHidden()
|
||
.disabled(!phrase.active)
|
||
|
||
Picker("", selection: $phrase.track) {
|
||
Text("Roteiro").tag(ReviewPhrase.trackScript)
|
||
Text("Bastidor").tag(ReviewPhrase.trackBackstage)
|
||
}
|
||
.pickerStyle(.menu)
|
||
.controlSize(.mini)
|
||
.labelsHidden()
|
||
.frame(width: 92)
|
||
}
|
||
|
||
if model.selection == phrase.id && !phrase.words.isEmpty {
|
||
wordTrimmer
|
||
}
|
||
}
|
||
.padding(.vertical, 4)
|
||
.opacity(phrase.active ? 1 : 0.55)
|
||
}
|
||
|
||
/// The delivery emotion the acoustics suggest. Shown faded below its own
|
||
/// confidence: a guess the analysis is unsure about should not compete for
|
||
/// attention with the emphasis decision, which is the point of the row.
|
||
private var emotionChip: some View {
|
||
let (label, icon) = ReviewPhrase.emotionLabel(phrase.emotion)
|
||
return Label(label, systemImage: icon)
|
||
.font(.caption2)
|
||
.padding(.horizontal, 5)
|
||
.padding(.vertical, 1)
|
||
.background(
|
||
Capsule().fill(Color.secondary.opacity(0.12))
|
||
)
|
||
.foregroundStyle(phrase.emotionConfidence >= 0.5 ? .secondary : .tertiary)
|
||
.help("Emoção da entrega: \(label) — confiança \(Int(phrase.emotionConfidence * 100))%")
|
||
}
|
||
|
||
/// Trimming by pointing at the transcript: click a word to start the phrase
|
||
/// there, option-click to end it there. Same edit as dragging the block's
|
||
/// edge on the timeline, but reachable while reading the line.
|
||
private var wordTrimmer: some View {
|
||
VStack(alignment: .leading, spacing: 4) {
|
||
HStack(spacing: 4) {
|
||
Text("Cortar pelas palavras").font(.caption2).foregroundStyle(.secondary)
|
||
Spacer()
|
||
if phrase.isTrimmed {
|
||
Button("Inteira") { model.resetTrim(phrase.id) }
|
||
.buttonStyle(.link).font(.caption2)
|
||
}
|
||
}
|
||
FlowWords(words: phrase.words, phrase: phrase) { word, edge in
|
||
model.trimToWord(word, edge: edge, in: phrase.id)
|
||
}
|
||
Text("Clique = começa aqui · ⌥clique = termina aqui")
|
||
.font(.caption2).foregroundStyle(.tertiary)
|
||
}
|
||
.padding(.top, 2)
|
||
}
|
||
}
|
||
|
||
/// The phrase's words as wrapping chips, dimmed where they fall outside the trim.
|
||
private struct FlowWords: View {
|
||
let words: [ReviewWord]
|
||
let phrase: ReviewPhrase
|
||
let onTrim: (ReviewWord, TrimEdge) -> Void
|
||
|
||
var body: some View {
|
||
// A LazyVGrid with adaptive columns wraps chips without a custom layout;
|
||
// phrases are short enough that the slight raggedness beats the cost of
|
||
// hand-rolling a flow layout here.
|
||
LazyVGrid(columns: [GridItem(.adaptive(minimum: 44), spacing: 3)],
|
||
alignment: .leading, spacing: 3) {
|
||
ForEach(words) { word in
|
||
let kept = word.start >= phrase.trimStart - 0.001 && word.end <= phrase.trimEnd + 0.001
|
||
Text(word.text)
|
||
.font(.caption2)
|
||
.padding(.horizontal, 4)
|
||
.padding(.vertical, 2)
|
||
.background(
|
||
RoundedRectangle(cornerRadius: 3)
|
||
.fill(kept ? Color.accentColor.opacity(0.12) : Color.secondary.opacity(0.08))
|
||
)
|
||
.foregroundStyle(kept ? .primary : .secondary)
|
||
.strikethrough(!kept)
|
||
.onTapGesture {
|
||
onTrim(word, NSEvent.modifierFlags.contains(.option) ? .end : .start)
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|