feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
711c397dfe
commit
7b5aed79ee
@@ -21,6 +21,20 @@ enum EmphasisPalette {
|
||||
}
|
||||
}
|
||||
|
||||
/// The same 0–3 tiers a phrase's `emphasis` uses, derived from a raw 0–1
|
||||
/// acoustic score — the thresholds `10-revisao-humana.md` documents for
|
||||
/// deriving a phrase's level from `peak_emphasis` when no explicit zoom
|
||||
/// was set, reused here per WORD so a word chip and a phrase row read as
|
||||
/// the same scale.
|
||||
static func levelFromScore(_ score: Double) -> Int {
|
||||
switch score {
|
||||
case ..<0.25: return 0
|
||||
case ..<0.45: return 1
|
||||
case ..<0.65: return 2
|
||||
default: return 3
|
||||
}
|
||||
}
|
||||
|
||||
static func speakerColor(_ speaker: String, among speakers: [String]) -> Color {
|
||||
let palette: [Color] = [.teal, .purple, .green, .indigo, .brown, .cyan]
|
||||
guard let index = speakers.firstIndex(of: speaker) else { return .gray }
|
||||
@@ -47,7 +61,9 @@ struct TimelineTracksView: View {
|
||||
private let trackSpacing: CGFloat = 4
|
||||
|
||||
private var pps: CGFloat { CGFloat(model.pixelsPerSecond) }
|
||||
private var contentWidth: CGFloat { max(320, CGFloat(model.duration) * pps) }
|
||||
/// Width follows the *kept* duration, not the raw take's — the timeline
|
||||
/// draws the cut, so removed stretches take no horizontal space.
|
||||
private var contentWidth: CGFloat { max(320, CGFloat(model.keptDuration) * pps) }
|
||||
|
||||
/// Name, icon and height of each lane, in the order they stack. The gutter
|
||||
/// and the tracks are built from this one list so a label can never drift
|
||||
@@ -215,8 +231,8 @@ struct TimelineTracksView: View {
|
||||
Canvas { context, size in
|
||||
let step = tickStep()
|
||||
var time = 0.0
|
||||
while time <= model.duration {
|
||||
let position = x(time)
|
||||
while time <= model.keptDuration {
|
||||
let position = compactX(time)
|
||||
context.stroke(
|
||||
Path { $0.move(to: CGPoint(x: position, y: size.height - 6))
|
||||
$0.addLine(to: CGPoint(x: position, y: size.height)) },
|
||||
@@ -301,7 +317,8 @@ struct TimelineTracksView: View {
|
||||
.gesture(
|
||||
DragGesture(minimumDistance: 1)
|
||||
.onChanged { value in
|
||||
let time = phrase.start + Double((value.location.x) / pps)
|
||||
let compactOrigin = model.compactTime(phrase.start)
|
||||
let time = model.rawTime(fromCompact: compactOrigin + Double(value.location.x / pps))
|
||||
model.trim(phrase.id, edge: edge, to: time)
|
||||
}
|
||||
)
|
||||
@@ -397,8 +414,8 @@ struct TimelineTracksView: View {
|
||||
private var scrubGesture: some Gesture {
|
||||
DragGesture(minimumDistance: 0)
|
||||
.onChanged { value in
|
||||
let from = Double(value.startLocation.x / pps)
|
||||
let to = Double(value.location.x / pps)
|
||||
let from = model.rawTime(fromCompact: Double(value.startLocation.x / pps))
|
||||
let to = model.rawTime(fromCompact: Double(value.location.x / pps))
|
||||
if abs(value.translation.width) > 3 {
|
||||
model.setRange(from: from, to: to)
|
||||
model.seek(to: min(from, to))
|
||||
@@ -486,10 +503,16 @@ struct TimelineTracksView: View {
|
||||
|
||||
// MARK: - Escala
|
||||
|
||||
private func x(_ time: Double) -> CGFloat { CGFloat(time) * pps }
|
||||
/// Pixel position of a raw source-media time, after collapsing whatever
|
||||
/// lies between it and the previous kept phrase.
|
||||
private func x(_ time: Double) -> CGFloat { compactX(model.compactTime(time)) }
|
||||
|
||||
/// Pixel position of a time already in the compacted (edited) timeline —
|
||||
/// used for the ruler and playhead, which think in that space directly.
|
||||
private func compactX(_ compactTime: Double) -> CGFloat { CGFloat(compactTime) * pps }
|
||||
|
||||
private func width(from: Double, to: Double) -> CGFloat {
|
||||
max(0, CGFloat(to - from) * pps)
|
||||
max(0, CGFloat(model.compactTime(to) - model.compactTime(from)) * pps)
|
||||
}
|
||||
|
||||
/// Ruler spacing that keeps labels ~80pt apart at any zoom.
|
||||
|
||||
Reference in New Issue
Block a user