Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
530 lines
21 KiB
Swift
530 lines
21 KiB
Swift
import SwiftUI
|
||
|
||
/// Colors shared by the timeline and the inspector, so a block and its row in
|
||
/// the list always read as the same thing.
|
||
enum EmphasisPalette {
|
||
static func color(_ level: Int) -> Color {
|
||
switch level {
|
||
case 1: return Color.blue
|
||
case 2: return Color.orange
|
||
case 3: return Color.pink
|
||
default: return Color.secondary
|
||
}
|
||
}
|
||
|
||
static func label(_ level: Int) -> String {
|
||
switch level {
|
||
case 1: return "Leve"
|
||
case 2: return "Média"
|
||
case 3: return "Forte"
|
||
default: return "Sem"
|
||
}
|
||
}
|
||
|
||
/// The same 0–3 tiers a phrase's `emphasis` uses, derived from a raw 0–1
|
||
/// acoustic score — the thresholds `10-revisao-humana.md` documents for
|
||
/// deriving a phrase's level from `peak_emphasis` when no explicit zoom
|
||
/// was set, reused here per WORD so a word chip and a phrase row read as
|
||
/// the same scale.
|
||
static func levelFromScore(_ score: Double) -> Int {
|
||
switch score {
|
||
case ..<0.25: return 0
|
||
case ..<0.45: return 1
|
||
case ..<0.65: return 2
|
||
default: return 3
|
||
}
|
||
}
|
||
|
||
static func speakerColor(_ speaker: String, among speakers: [String]) -> Color {
|
||
let palette: [Color] = [.teal, .purple, .green, .indigo, .brown, .cyan]
|
||
guard let index = speakers.firstIndex(of: speaker) else { return .gray }
|
||
return palette[index % palette.count]
|
||
}
|
||
}
|
||
|
||
/// The timeline strip: four stacked tracks over one shared time axis.
|
||
///
|
||
/// Phrases are laid out as real views rather than drawn into a Canvas, because
|
||
/// every one of them is a target — click to select, drag its edge to trim,
|
||
/// right-click to change emphasis. The dense per-word energy track *is* a
|
||
/// Canvas: it has thousands of bars and nothing to hit.
|
||
struct TimelineTracksView: View {
|
||
@ObservedObject var model: PhraseReviewModel
|
||
|
||
private let rulerHeight: CGFloat = 18
|
||
private let phraseHeight: CGFloat = 46
|
||
private let energyHeight: CGFloat = 34
|
||
private let stripHeight: CGFloat = 12
|
||
private let handleWidth: CGFloat = 8
|
||
|
||
private let gutterWidth: CGFloat = 92
|
||
private let trackSpacing: CGFloat = 4
|
||
|
||
private var pps: CGFloat { CGFloat(model.pixelsPerSecond) }
|
||
/// Width follows the *kept* duration, not the raw take's — the timeline
|
||
/// draws the cut, so removed stretches take no horizontal space.
|
||
private var contentWidth: CGFloat { max(320, CGFloat(model.keptDuration) * pps) }
|
||
|
||
/// Name, icon and height of each lane, in the order they stack. The gutter
|
||
/// and the tracks are built from this one list so a label can never drift
|
||
/// off the lane it names.
|
||
private var lanes: [(label: String, icon: String, height: CGFloat)] {
|
||
[
|
||
("", "", rulerHeight),
|
||
("Zooms", "plus.magnifyingglass", stripHeight + 6),
|
||
("Frases", "text.quote", phraseHeight),
|
||
("Energia", "waveform", energyHeight),
|
||
("Emoção", "face.smiling", stripHeight),
|
||
("Locutor", "person.wave.2", stripHeight),
|
||
("Roteiro", "list.bullet.rectangle", stripHeight),
|
||
]
|
||
}
|
||
|
||
var body: some View {
|
||
VStack(spacing: 0) {
|
||
toolbar
|
||
Divider()
|
||
HStack(alignment: .top, spacing: 0) {
|
||
gutter
|
||
Divider()
|
||
timelineScroller
|
||
}
|
||
}
|
||
.background(Color(nsColor: .underPageBackgroundColor))
|
||
}
|
||
|
||
/// Fixed column naming each lane. Without it the stripes are six colours
|
||
/// with no way to tell which one is emotion and which one is the speaker.
|
||
private var gutter: some View {
|
||
VStack(alignment: .leading, spacing: trackSpacing) {
|
||
ForEach(lanes.indices, id: \.self) { index in
|
||
let lane = lanes[index]
|
||
HStack(spacing: 4) {
|
||
if !lane.icon.isEmpty {
|
||
Image(systemName: lane.icon).font(.system(size: 9))
|
||
}
|
||
Text(lane.label).font(.system(size: 10))
|
||
Spacer(minLength: 0)
|
||
}
|
||
.foregroundStyle(.secondary)
|
||
.frame(height: lane.height, alignment: .center)
|
||
}
|
||
}
|
||
.padding(.horizontal, 8)
|
||
.padding(.vertical, 8)
|
||
.frame(width: gutterWidth, alignment: .leading)
|
||
}
|
||
|
||
private var timelineScroller: some View {
|
||
ScrollViewReader { proxy in
|
||
ScrollView([.horizontal]) {
|
||
ZStack(alignment: .topLeading) {
|
||
VStack(alignment: .leading, spacing: trackSpacing) {
|
||
ruler
|
||
zoomTrack
|
||
phraseTrack
|
||
energyTrack
|
||
emotionTrack
|
||
speakerTrack
|
||
scriptTrack
|
||
}
|
||
.frame(width: contentWidth, alignment: .leading)
|
||
rangeOverlay
|
||
playhead
|
||
// Anchors the auto-scroll: one invisible marker per
|
||
// phrase, so selecting a line off-screen brings it in.
|
||
ForEach(model.phrases) { phrase in
|
||
Color.clear
|
||
.frame(width: 1, height: 1)
|
||
.offset(x: x(phrase.start))
|
||
.id(phrase.id)
|
||
}
|
||
}
|
||
.padding(.vertical, 8)
|
||
.contentShape(Rectangle())
|
||
.gesture(scrubGesture)
|
||
.contextMenu { timelineMenu }
|
||
}
|
||
.onChange(of: model.selection) { _, newValue in
|
||
guard let newValue else { return }
|
||
withAnimation(.easeOut(duration: 0.2)) {
|
||
proxy.scrollTo(newValue, anchor: .center)
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// MARK: - Barra de controles
|
||
|
||
private var toolbar: some View {
|
||
HStack(spacing: 12) {
|
||
Button {
|
||
model.togglePlay()
|
||
} label: {
|
||
Image(systemName: model.isPlaying ? "pause.fill" : "play.fill")
|
||
}
|
||
.buttonStyle(.borderless)
|
||
.help("Reproduzir (espaço)")
|
||
.disabled(model.player == nil)
|
||
|
||
Text(timecode(model.currentTime))
|
||
.font(.system(.caption, design: .monospaced))
|
||
.foregroundStyle(.secondary)
|
||
|
||
Button {
|
||
model.playSelectedPhrase()
|
||
} label: {
|
||
Image(systemName: "play.rectangle")
|
||
}
|
||
.buttonStyle(.borderless)
|
||
.help("Tocar só a frase selecionada (⏎)")
|
||
.disabled(model.player == nil || model.selection == nil)
|
||
|
||
Toggle("Pular removidos", isOn: $model.skipRemoved)
|
||
.toggleStyle(.checkbox)
|
||
.font(.caption)
|
||
.help("Durante a reprodução, salta os trechos desativados — mostra como o corte ficou.")
|
||
|
||
Button {
|
||
model.addZoomForRange()
|
||
} label: {
|
||
Label("Zoom no trecho", systemImage: "plus.magnifyingglass")
|
||
}
|
||
.buttonStyle(.borderless)
|
||
.font(.caption)
|
||
.disabled(!model.hasRange)
|
||
.help("Arraste na timeline para marcar um trecho e crie um zoom nele. A escala vem de Análise de Voz.")
|
||
|
||
Spacer()
|
||
|
||
legend
|
||
|
||
Spacer()
|
||
|
||
Image(systemName: "minus.magnifyingglass").foregroundStyle(.secondary)
|
||
Slider(value: $model.pixelsPerSecond,
|
||
in: model.minPixelsPerSecond...model.maxPixelsPerSecond)
|
||
.frame(width: 130)
|
||
Image(systemName: "plus.magnifyingglass").foregroundStyle(.secondary)
|
||
}
|
||
.padding(.horizontal, 12)
|
||
.padding(.vertical, 8)
|
||
}
|
||
|
||
private var legend: some View {
|
||
HStack(spacing: 10) {
|
||
ForEach(0..<4, id: \.self) { level in
|
||
HStack(spacing: 4) {
|
||
RoundedRectangle(cornerRadius: 2)
|
||
.fill(EmphasisPalette.color(level))
|
||
.frame(width: 10, height: 10)
|
||
Text(EmphasisPalette.label(level)).font(.caption2)
|
||
}
|
||
}
|
||
}
|
||
.foregroundStyle(.secondary)
|
||
}
|
||
|
||
// MARK: - Trilhas
|
||
|
||
private var ruler: some View {
|
||
Canvas { context, size in
|
||
let step = tickStep()
|
||
var time = 0.0
|
||
while time <= model.keptDuration {
|
||
let position = compactX(time)
|
||
context.stroke(
|
||
Path { $0.move(to: CGPoint(x: position, y: size.height - 6))
|
||
$0.addLine(to: CGPoint(x: position, y: size.height)) },
|
||
with: .color(.secondary.opacity(0.5))
|
||
)
|
||
context.draw(
|
||
Text(timecode(time)).font(.system(size: 9, design: .monospaced))
|
||
.foregroundColor(.secondary),
|
||
at: CGPoint(x: position + 18, y: 6)
|
||
)
|
||
time += step
|
||
}
|
||
}
|
||
.frame(width: contentWidth, height: rulerHeight)
|
||
}
|
||
|
||
private var phraseTrack: some View {
|
||
ZStack(alignment: .topLeading) {
|
||
RoundedRectangle(cornerRadius: 4)
|
||
.fill(Color.secondary.opacity(0.06))
|
||
.frame(width: contentWidth, height: phraseHeight)
|
||
ForEach(model.phrases) { phrase in
|
||
phraseBlock(phrase)
|
||
}
|
||
}
|
||
.frame(width: contentWidth, height: phraseHeight, alignment: .topLeading)
|
||
}
|
||
|
||
@ViewBuilder
|
||
private func phraseBlock(_ phrase: ReviewPhrase) -> some View {
|
||
let isSelected = model.selection == phrase.id
|
||
let color = EmphasisPalette.color(phrase.emphasis)
|
||
let fullWidth = max(2, width(from: phrase.start, to: phrase.end))
|
||
let keptWidth = max(1, width(from: phrase.trimStart, to: phrase.trimEnd))
|
||
|
||
ZStack(alignment: .topLeading) {
|
||
// The whole line, dim — what is there before the edit.
|
||
RoundedRectangle(cornerRadius: 4)
|
||
.fill(color.opacity(phrase.active ? 0.15 : 0.10))
|
||
.frame(width: fullWidth, height: phraseHeight)
|
||
|
||
// What survives: the kept span, drawn solid over it.
|
||
RoundedRectangle(cornerRadius: 4)
|
||
.fill(color.opacity(phrase.active ? 0.55 : 0.12))
|
||
.frame(width: keptWidth, height: phraseHeight)
|
||
.offset(x: width(from: phrase.start, to: phrase.trimStart))
|
||
|
||
Text(phrase.text)
|
||
.font(.system(size: 10))
|
||
.lineLimit(2)
|
||
.padding(.horizontal, 4)
|
||
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
|
||
.foregroundStyle(phrase.active ? .primary : .secondary)
|
||
.strikethrough(!phrase.active)
|
||
|
||
RoundedRectangle(cornerRadius: 4)
|
||
.stroke(isSelected ? Color.accentColor : color.opacity(0.4),
|
||
lineWidth: isSelected ? 2 : 1)
|
||
.frame(width: fullWidth, height: phraseHeight)
|
||
|
||
if isSelected && phrase.active {
|
||
trimHandle(phrase, edge: .start)
|
||
trimHandle(phrase, edge: .end)
|
||
}
|
||
}
|
||
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
|
||
.offset(x: x(phrase.start))
|
||
.contentShape(Rectangle())
|
||
.onTapGesture { model.goTo(phraseID: phrase.id) }
|
||
.contextMenu { phraseMenu(phrase) }
|
||
.help(phrase.reason.isEmpty ? phrase.text : "\(phrase.text)\n— \(phrase.reason)")
|
||
}
|
||
|
||
private func trimHandle(_ phrase: ReviewPhrase, edge: TrimEdge) -> some View {
|
||
let offset = edge == .start
|
||
? width(from: phrase.start, to: phrase.trimStart)
|
||
: width(from: phrase.start, to: phrase.trimEnd) - handleWidth
|
||
return RoundedRectangle(cornerRadius: 2)
|
||
.fill(Color.accentColor)
|
||
.frame(width: handleWidth, height: phraseHeight)
|
||
.offset(x: offset)
|
||
.gesture(
|
||
DragGesture(minimumDistance: 1)
|
||
.onChanged { value in
|
||
let compactOrigin = model.compactTime(phrase.start)
|
||
let time = model.rawTime(fromCompact: compactOrigin + Double(value.location.x / pps))
|
||
model.trim(phrase.id, edge: edge, to: time)
|
||
}
|
||
)
|
||
.help(edge == .start ? "Arraste para cortar o começo (pula de palavra em palavra)"
|
||
: "Arraste para cortar o fim (pula de palavra em palavra)")
|
||
}
|
||
|
||
@ViewBuilder
|
||
private func phraseMenu(_ phrase: ReviewPhrase) -> some View {
|
||
Button("Tocar esta frase") {
|
||
model.goTo(phraseID: phrase.id)
|
||
model.playSelectedPhrase()
|
||
}
|
||
Button(phrase.active ? "Remover do corte" : "Trazer de volta") {
|
||
model.toggleActive(phrase.id)
|
||
}
|
||
Button("Adicionar zoom nesta frase") { model.addZoomForPhrase(phrase.id) }
|
||
Divider()
|
||
ForEach(0..<4, id: \.self) { level in
|
||
Button("Ênfase: \(EmphasisPalette.label(level))") {
|
||
model.setEmphasis(level, for: phrase.id)
|
||
}
|
||
}
|
||
Divider()
|
||
Button(phrase.isBackstage ? "Marcar como roteiro" : "Marcar como bastidor") {
|
||
model.setTrack(phrase.isBackstage ? ReviewPhrase.trackScript : ReviewPhrase.trackBackstage,
|
||
for: phrase.id)
|
||
}
|
||
if phrase.isTrimmed {
|
||
Divider()
|
||
Button("Desfazer corte da frase") { model.resetTrim(phrase.id) }
|
||
}
|
||
}
|
||
|
||
/// Per-word energy/emphasis, straight from the voice timeline — the closest
|
||
/// thing to a waveform without opening the audio again.
|
||
private var energyTrack: some View {
|
||
Canvas { context, size in
|
||
for phrase in model.phrases {
|
||
for word in phrase.words {
|
||
let start = x(word.start)
|
||
let barWidth = max(1, width(from: word.start, to: word.end) - 1)
|
||
let height = size.height * CGFloat(max(0.04, word.energy))
|
||
let rect = CGRect(x: start, y: size.height - height,
|
||
width: barWidth, height: height)
|
||
let color = word.emphasis >= 0.65 ? Color.pink
|
||
: word.emphasis >= 0.45 ? Color.orange
|
||
: Color.secondary
|
||
context.fill(Path(rect),
|
||
with: .color(color.opacity(phrase.active ? 0.6 : 0.2)))
|
||
}
|
||
}
|
||
}
|
||
.frame(width: contentWidth, height: energyHeight)
|
||
.background(RoundedRectangle(cornerRadius: 4).fill(Color.secondary.opacity(0.06)))
|
||
}
|
||
|
||
private var speakerTrack: some View {
|
||
stripTrack { phrase in
|
||
EmphasisPalette.speakerColor(phrase.speaker, among: model.speakers)
|
||
}
|
||
}
|
||
|
||
private var scriptTrack: some View {
|
||
stripTrack { phrase in phrase.isBackstage ? Color.gray : Color.mint }
|
||
}
|
||
|
||
private func stripTrack(_ color: @escaping (ReviewPhrase) -> Color) -> some View {
|
||
Canvas { context, size in
|
||
for phrase in model.phrases {
|
||
let rect = CGRect(x: x(phrase.start), y: 0,
|
||
width: max(1, width(from: phrase.start, to: phrase.end)),
|
||
height: size.height)
|
||
context.fill(Path(roundedRect: rect, cornerRadius: 2),
|
||
with: .color(color(phrase).opacity(phrase.active ? 0.7 : 0.2)))
|
||
}
|
||
}
|
||
.frame(width: contentWidth, height: stripHeight)
|
||
}
|
||
|
||
private var playhead: some View {
|
||
Rectangle()
|
||
.fill(Color.red)
|
||
.frame(width: 1.5)
|
||
.offset(x: x(model.currentTime))
|
||
.allowsHitTesting(false)
|
||
}
|
||
|
||
/// One gesture, two meanings, decided by whether the mouse moved: a click
|
||
/// parks the playhead, a drag marks in/out. Splitting them across separate
|
||
/// controls would mean choosing a tool before every action, which is
|
||
/// exactly the ceremony this screen is meant to avoid.
|
||
private var scrubGesture: some Gesture {
|
||
DragGesture(minimumDistance: 0)
|
||
.onChanged { value in
|
||
let from = model.rawTime(fromCompact: Double(value.startLocation.x / pps))
|
||
let to = model.rawTime(fromCompact: Double(value.location.x / pps))
|
||
if abs(value.translation.width) > 3 {
|
||
model.setRange(from: from, to: to)
|
||
model.seek(to: min(from, to))
|
||
} else {
|
||
model.clearRange()
|
||
model.seek(to: to)
|
||
}
|
||
}
|
||
}
|
||
|
||
/// The marked in/out, drawn over every track so the span reads against the
|
||
/// phrases and the energy at once.
|
||
private var rangeOverlay: some View {
|
||
Group {
|
||
if let span = model.rangeSpan {
|
||
Rectangle()
|
||
.fill(Color.accentColor.opacity(0.18))
|
||
.overlay(Rectangle().stroke(Color.accentColor.opacity(0.6), lineWidth: 1))
|
||
.frame(width: max(1, width(from: span.start, to: span.end)))
|
||
.offset(x: x(span.start))
|
||
.allowsHitTesting(false)
|
||
}
|
||
}
|
||
}
|
||
|
||
@ViewBuilder
|
||
private var timelineMenu: some View {
|
||
if model.hasRange, let span = model.rangeSpan {
|
||
Button("Adicionar zoom no trecho (\(secondsLabel(span.end - span.start)))") {
|
||
model.addZoomForRange()
|
||
}
|
||
Button("Tocar o trecho") { model.playRange(from: span.start, to: span.end) }
|
||
Button("Limpar seleção") { model.clearRange() }
|
||
} else {
|
||
Text("Arraste na timeline para marcar um trecho")
|
||
}
|
||
if let zoom = model.zoom(at: model.currentTime) {
|
||
Divider()
|
||
Button("Remover o zoom daqui") { model.removeZoom(zoom.id) }
|
||
}
|
||
}
|
||
|
||
private func secondsLabel(_ seconds: Double) -> String {
|
||
String(format: "%.1fs", seconds)
|
||
}
|
||
|
||
/// Punch-ins, on their own lane above the script: they are a second layer
|
||
/// over the same time, not a property of a phrase.
|
||
private var zoomTrack: some View {
|
||
ZStack(alignment: .topLeading) {
|
||
RoundedRectangle(cornerRadius: 3)
|
||
.fill(Color.secondary.opacity(0.06))
|
||
.frame(width: contentWidth, height: stripHeight + 6)
|
||
ForEach(model.zooms) { zoom in
|
||
RoundedRectangle(cornerRadius: 3)
|
||
.fill(Color.yellow.opacity(0.55))
|
||
.overlay(
|
||
Image(systemName: "plus.magnifyingglass")
|
||
.font(.system(size: 8)).foregroundStyle(.black.opacity(0.6))
|
||
)
|
||
.frame(width: max(6, width(from: zoom.start, to: zoom.end)),
|
||
height: stripHeight + 6)
|
||
.offset(x: x(zoom.start))
|
||
.help("Zoom marcado — \(secondsLabel(zoom.end - zoom.start)). A escala vem de Análise de Voz.")
|
||
.contextMenu {
|
||
Button("Remover este zoom") { model.removeZoom(zoom.id) }
|
||
}
|
||
}
|
||
}
|
||
.frame(width: contentWidth, height: stripHeight + 6, alignment: .topLeading)
|
||
}
|
||
|
||
/// Delivery emotion per phrase — the fourth signal to read against the text.
|
||
private var emotionTrack: some View {
|
||
stripTrack { phrase in
|
||
switch phrase.emotion {
|
||
case "excited": return .orange
|
||
case "tense": return .red
|
||
case "calm": return .blue
|
||
case "reflective": return .purple
|
||
default: return .secondary
|
||
}
|
||
}
|
||
}
|
||
|
||
// MARK: - Escala
|
||
|
||
/// Pixel position of a raw source-media time, after collapsing whatever
|
||
/// lies between it and the previous kept phrase.
|
||
private func x(_ time: Double) -> CGFloat { compactX(model.compactTime(time)) }
|
||
|
||
/// Pixel position of a time already in the compacted (edited) timeline —
|
||
/// used for the ruler and playhead, which think in that space directly.
|
||
private func compactX(_ compactTime: Double) -> CGFloat { CGFloat(compactTime) * pps }
|
||
|
||
private func width(from: Double, to: Double) -> CGFloat {
|
||
max(0, CGFloat(model.compactTime(to) - model.compactTime(from)) * pps)
|
||
}
|
||
|
||
/// Ruler spacing that keeps labels ~80pt apart at any zoom.
|
||
private func tickStep() -> Double {
|
||
let candidates: [Double] = [1, 2, 5, 10, 15, 30, 60, 120, 300, 600]
|
||
let wanted = 80 / Double(pps)
|
||
return candidates.first { $0 >= wanted } ?? 600
|
||
}
|
||
|
||
private func timecode(_ seconds: Double) -> String {
|
||
let total = Int(seconds.rounded(.down))
|
||
return String(format: "%02d:%02d", total / 60, total % 60)
|
||
}
|
||
}
|