feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão

Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-21 18:26:04 -04:00
co-authored by Claude Sonnet 5
parent 711c397dfe
commit 7b5aed79ee
36 changed files with 2922 additions and 624 deletions
+31 -8
View File
@@ -21,6 +21,20 @@ enum EmphasisPalette {
}
}
/// The same 0–3 tiers a phrase's `emphasis` uses, derived from a raw 0–1
/// acoustic score — the thresholds `10-revisao-humana.md` documents for
/// deriving a phrase's level from `peak_emphasis` when no explicit zoom
/// was set, reused here per WORD so a word chip and a phrase row read as
/// the same scale.
static func levelFromScore(_ score: Double) -> Int {
switch score {
case ..<0.25: return 0
case ..<0.45: return 1
case ..<0.65: return 2
default: return 3
}
}
static func speakerColor(_ speaker: String, among speakers: [String]) -> Color {
let palette: [Color] = [.teal, .purple, .green, .indigo, .brown, .cyan]
guard let index = speakers.firstIndex(of: speaker) else { return .gray }
@@ -47,7 +61,9 @@ struct TimelineTracksView: View {
private let trackSpacing: CGFloat = 4
private var pps: CGFloat { CGFloat(model.pixelsPerSecond) }
private var contentWidth: CGFloat { max(320, CGFloat(model.duration) * pps) }
/// Width follows the *kept* duration, not the raw take's — the timeline
/// draws the cut, so removed stretches take no horizontal space.
private var contentWidth: CGFloat { max(320, CGFloat(model.keptDuration) * pps) }
/// Name, icon and height of each lane, in the order they stack. The gutter
/// and the tracks are built from this one list so a label can never drift
@@ -215,8 +231,8 @@ struct TimelineTracksView: View {
Canvas { context, size in
let step = tickStep()
var time = 0.0
while time <= model.duration {
let position = x(time)
while time <= model.keptDuration {
let position = compactX(time)
context.stroke(
Path { $0.move(to: CGPoint(x: position, y: size.height - 6))
$0.addLine(to: CGPoint(x: position, y: size.height)) },
@@ -301,7 +317,8 @@ struct TimelineTracksView: View {
.gesture(
DragGesture(minimumDistance: 1)
.onChanged { value in
let time = phrase.start + Double((value.location.x) / pps)
let compactOrigin = model.compactTime(phrase.start)
let time = model.rawTime(fromCompact: compactOrigin + Double(value.location.x / pps))
model.trim(phrase.id, edge: edge, to: time)
}
)
@@ -397,8 +414,8 @@ struct TimelineTracksView: View {
private var scrubGesture: some Gesture {
DragGesture(minimumDistance: 0)
.onChanged { value in
let from = Double(value.startLocation.x / pps)
let to = Double(value.location.x / pps)
let from = model.rawTime(fromCompact: Double(value.startLocation.x / pps))
let to = model.rawTime(fromCompact: Double(value.location.x / pps))
if abs(value.translation.width) > 3 {
model.setRange(from: from, to: to)
model.seek(to: min(from, to))
@@ -486,10 +503,16 @@ struct TimelineTracksView: View {
// MARK: - Escala
private func x(_ time: Double) -> CGFloat { CGFloat(time) * pps }
/// Pixel position of a raw source-media time, after collapsing whatever
/// lies between it and the previous kept phrase.
private func x(_ time: Double) -> CGFloat { compactX(model.compactTime(time)) }
/// Pixel position of a time already in the compacted (edited) timeline —
/// used for the ruler and playhead, which think in that space directly.
private func compactX(_ compactTime: Double) -> CGFloat { CGFloat(compactTime) * pps }
private func width(from: Double, to: Double) -> CGFloat {
max(0, CGFloat(to - from) * pps)
max(0, CGFloat(model.compactTime(to) - model.compactTime(from)) * pps)
}
/// Ruler spacing that keeps labels ~80pt apart at any zoom.