feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão

Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-21 18:26:04 -04:00
co-authored by Claude Sonnet 5
parent 711c397dfe
commit 7b5aed79ee
36 changed files with 2922 additions and 624 deletions
+41 -141
View File
@@ -1,86 +1,32 @@
import AVFoundation
import SwiftUI
/// The video surface, as a plain `AVPlayerLayer` in an `NSView`.
/// The wizard's emphasis-review step.
///
/// AVKit's `VideoPlayer` would be the obvious choice and is a trap here: this
/// app is built by invoking `swiftc` directly (see `MacApp/build_app.sh`), and
/// `_AVKit_SwiftUI` aborts at launch instantiating its generic metadata under
/// that build. A player layer needs only AVFoundation, which links cleanly —
/// and the transport controls live in the timeline's own toolbar anyway, so
/// nothing is lost by dropping AVKit's chrome.
private struct PlayerSurface: NSViewRepresentable {
let player: AVPlayer
/// When true the frame is filled and cropped instead of letterboxed — used
/// to preview horizontal footage inside a vertical delivery frame.
var fills: Bool
func makeNSView(context: Context) -> PlayerLayerView {
let view = PlayerLayerView()
view.player = player
view.fills = fills
return view
}
func updateNSView(_ view: PlayerLayerView, context: Context) {
if view.player !== player { view.player = player }
view.fills = fills
}
}
final class PlayerLayerView: NSView {
private let playerLayer = AVPlayerLayer()
var player: AVPlayer? {
get { playerLayer.player }
set { playerLayer.player = newValue }
}
var fills: Bool = false {
didSet { playerLayer.videoGravity = fills ? .resizeAspectFill : .resizeAspect }
}
override init(frame frameRect: NSRect) {
super.init(frame: frameRect)
wantsLayer = true
layer = CALayer()
layer?.backgroundColor = NSColor.black.cgColor
playerLayer.videoGravity = .resizeAspect
layer?.addSublayer(playerLayer)
}
required init?(coder: NSCoder) {
super.init(coder: coder)
wantsLayer = true
layer = CALayer()
playerLayer.videoGravity = .resizeAspect
layer?.addSublayer(playerLayer)
}
override func layout() {
super.layout()
playerLayer.frame = bounds
}
}
/// The wizard's emphasis-review step, laid out like an editing room: preview on
/// top, timeline across the bottom, and the script as an inspector down the
/// right side.
///
/// The arrangement is the point. Every decision here is about a *sentence*, so
/// the same phrase has to be legible in all three places at once — a block on
/// the timeline, a line of text in the inspector, and a moment in the preview.
/// Selecting in any one of them selects in the other two.
/// Every decision here is about a *sentence* read from the original
/// transcription, so the phrases are listed in full — each line shows the text
/// as it will be said, a switch to keep or drop it from the cut, and the
/// emphasis level. Selecting a line in the list also selects its block on the
/// timeline below, and vice-versa.
struct PhraseReviewView: View {
@ObservedObject var model: PhraseReviewModel
var body: some View {
VSplitView {
HSplitView {
previewPane
.frame(minWidth: 320, idealWidth: 640)
inspectorPane
.frame(minWidth: 300, idealWidth: 360, maxWidth: 520)
VStack(spacing: 0) {
inspectorHeader
Divider()
List(selection: $model.selection) {
ForEach($model.phrases) { $phrase in
PhraseRow(phrase: $phrase, model: model)
.tag(phrase.id)
}
}
.listStyle(.inset)
.onChange(of: model.selection) { _, newValue in
if let newValue { model.goTo(phraseID: newValue) }
}
Divider()
summaryBar
}
.frame(minHeight: 240)
@@ -111,48 +57,6 @@ struct PhraseReviewView: View {
}
}
// MARK: - Preview
private var previewPane: some View {
VStack(spacing: 0) {
if let player = model.player {
// The footage here is usually vertical. Sizing the surface to
// the take's own aspect keeps a 9:16 frame as tall as the pane
// allows instead of shrinking it to fit a horizontal box.
// Framed to what the project delivers, not to what the camera
// recorded: these takes are shot horizontal and cut vertical,
// so the raw frame would show material the audience never sees.
ZStack {
Color.black
PlayerSurface(player: player, fills: model.isCropping)
.aspectRatio(model.previewAspect, contentMode: .fit)
.clipped()
}
.overlay(alignment: .topTrailing) { framingBadge }
} else {
ZStack {
Color.black.opacity(0.85)
VStack(spacing: 10) {
Image(systemName: "film.stack")
.font(.system(size: 28)).foregroundStyle(.secondary)
Text(model.source.isEmpty
? "A análise de voz não registrou qual mídia foi usada."
: "Não achei \(model.source) na pasta do projeto.")
.font(.callout).foregroundStyle(.secondary)
Text("A revisão funciona igual sem o preview — ele só ajuda a conferir o corte.")
.font(.caption).foregroundStyle(.tertiary)
Button("Localizar a mídia…") { pickMedia() }
.buttonStyle(.bordered)
}
.multilineTextAlignment(.center)
.padding(.horizontal, 24)
}
}
Divider()
summaryBar
}
}
private var summaryBar: some View {
HStack(spacing: 16) {
summaryItem("text.quote", "\(model.phrases.count) frases")
@@ -181,28 +85,6 @@ struct PhraseReviewView: View {
String(format: "%02d:%02d finais", Int(seconds) / 60, Int(seconds) % 60)
}
/// Says which frame is on screen, and lets the editor flip to the raw take.
/// Without it a centred crop looks like the footage itself, and someone
/// would judge framing on an approximation without knowing it.
@ViewBuilder
private var framingBadge: some View {
if model.projectAspect != nil, abs((model.projectAspect ?? 0) - model.videoAspect) > 0.01 {
Button {
model.matchProjectFraming.toggle()
} label: {
Label(model.matchProjectFraming ? "Enquadramento do projeto" : "Mídia original",
systemImage: model.matchProjectFraming ? "crop" : "rectangle.expand.vertical")
.font(.caption2)
}
.buttonStyle(.borderless)
.padding(6)
.background(Capsule().fill(.black.opacity(0.45)))
.foregroundStyle(.white)
.padding(8)
.help("A fonte é horizontal e o projeto é vertical — o preview mostra o corte central aproximado. O enquadramento real de cada clipe vem do Final Cut.")
}
}
private func pickMedia() {
let panel = NSOpenPanel()
panel.canChooseFiles = true
@@ -373,14 +255,17 @@ private struct PhraseRow: View {
FlowWords(words: phrase.words, phrase: phrase) { word, edge in
model.trimToWord(word, edge: edge, in: phrase.id)
}
Text("Clique = começa aqui · ⌥clique = termina aqui")
Text("Clique = começa/desfaz aqui · ⌥clique = termina/desfaz aqui · sublinhado = ênfase da palavra")
.font(.caption2).foregroundStyle(.tertiary)
}
.padding(.top, 2)
}
}
/// The phrase's words as wrapping chips, dimmed where they fall outside the trim.
/// The phrase's words as wrapping chips, dimmed where they fall outside the
/// trim and underlined where the acoustics mark them as an emphasis peak —
/// the same word-level signal `05-zoom.md` picks a punch-in's `start` from,
/// made visible instead of buried in the JSON.
private struct FlowWords: View {
let words: [ReviewWord]
let phrase: ReviewPhrase
@@ -394,16 +279,31 @@ private struct FlowWords: View {
alignment: .leading, spacing: 3) {
ForEach(words) { word in
let kept = word.start >= phrase.trimStart - 0.001 && word.end <= phrase.trimEnd + 0.001
let level = EmphasisPalette.levelFromScore(word.emphasis)
Text(word.text)
.font(.caption2)
.fontWeight(level >= 2 ? .semibold : .regular)
.padding(.horizontal, 4)
.padding(.vertical, 2)
.background(
RoundedRectangle(cornerRadius: 3)
.fill(kept ? Color.accentColor.opacity(0.12) : Color.secondary.opacity(0.08))
)
.overlay(alignment: .bottom) {
if level >= 1 {
Rectangle()
.fill(EmphasisPalette.color(level))
.frame(height: 2)
.padding(.horizontal, 3)
}
}
.foregroundStyle(kept ? .primary : .secondary)
.strikethrough(!kept)
.help(
level >= 1
? "Ênfase \(EmphasisPalette.label(level).lowercased()) (\(Int(word.emphasis * 100))%)"
: "Sem ênfase"
)
.onTapGesture {
onTrim(word, NSEvent.modifierFlags.contains(.option) ? .end : .start)
}