feat: etapa 5 do assistente — revisão de ênfases com timeline

Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-19 21:29:27 -04:00
co-authored by Claude Opus 5
parent e7748c2c58
commit 1bebee4359
31 changed files with 4622 additions and 83 deletions
@@ -0,0 +1,506 @@
import SwiftUI
/// Colors shared by the timeline and the inspector, so a block and its row in
/// the list always read as the same thing.
enum EmphasisPalette {
static func color(_ level: Int) -> Color {
switch level {
case 1: return Color.blue
case 2: return Color.orange
case 3: return Color.pink
default: return Color.secondary
}
}
static func label(_ level: Int) -> String {
switch level {
case 1: return "Leve"
case 2: return "Média"
case 3: return "Forte"
default: return "Sem"
}
}
static func speakerColor(_ speaker: String, among speakers: [String]) -> Color {
let palette: [Color] = [.teal, .purple, .green, .indigo, .brown, .cyan]
guard let index = speakers.firstIndex(of: speaker) else { return .gray }
return palette[index % palette.count]
}
}
/// The timeline strip: four stacked tracks over one shared time axis.
///
/// Phrases are laid out as real views rather than drawn into a Canvas, because
/// every one of them is a target — click to select, drag its edge to trim,
/// right-click to change emphasis. The dense per-word energy track *is* a
/// Canvas: it has thousands of bars and nothing to hit.
struct TimelineTracksView: View {
@ObservedObject var model: PhraseReviewModel
private let rulerHeight: CGFloat = 18
private let phraseHeight: CGFloat = 46
private let energyHeight: CGFloat = 34
private let stripHeight: CGFloat = 12
private let handleWidth: CGFloat = 8
private let gutterWidth: CGFloat = 92
private let trackSpacing: CGFloat = 4
private var pps: CGFloat { CGFloat(model.pixelsPerSecond) }
private var contentWidth: CGFloat { max(320, CGFloat(model.duration) * pps) }
/// Name, icon and height of each lane, in the order they stack. The gutter
/// and the tracks are built from this one list so a label can never drift
/// off the lane it names.
private var lanes: [(label: String, icon: String, height: CGFloat)] {
[
("", "", rulerHeight),
("Zooms", "plus.magnifyingglass", stripHeight + 6),
("Frases", "text.quote", phraseHeight),
("Energia", "waveform", energyHeight),
("Emoção", "face.smiling", stripHeight),
("Locutor", "person.wave.2", stripHeight),
("Roteiro", "list.bullet.rectangle", stripHeight),
]
}
var body: some View {
VStack(spacing: 0) {
toolbar
Divider()
HStack(alignment: .top, spacing: 0) {
gutter
Divider()
timelineScroller
}
}
.background(Color(nsColor: .underPageBackgroundColor))
}
/// Fixed column naming each lane. Without it the stripes are six colours
/// with no way to tell which one is emotion and which one is the speaker.
private var gutter: some View {
VStack(alignment: .leading, spacing: trackSpacing) {
ForEach(lanes.indices, id: \.self) { index in
let lane = lanes[index]
HStack(spacing: 4) {
if !lane.icon.isEmpty {
Image(systemName: lane.icon).font(.system(size: 9))
}
Text(lane.label).font(.system(size: 10))
Spacer(minLength: 0)
}
.foregroundStyle(.secondary)
.frame(height: lane.height, alignment: .center)
}
}
.padding(.horizontal, 8)
.padding(.vertical, 8)
.frame(width: gutterWidth, alignment: .leading)
}
private var timelineScroller: some View {
ScrollViewReader { proxy in
ScrollView([.horizontal]) {
ZStack(alignment: .topLeading) {
VStack(alignment: .leading, spacing: trackSpacing) {
ruler
zoomTrack
phraseTrack
energyTrack
emotionTrack
speakerTrack
scriptTrack
}
.frame(width: contentWidth, alignment: .leading)
rangeOverlay
playhead
// Anchors the auto-scroll: one invisible marker per
// phrase, so selecting a line off-screen brings it in.
ForEach(model.phrases) { phrase in
Color.clear
.frame(width: 1, height: 1)
.offset(x: x(phrase.start))
.id(phrase.id)
}
}
.padding(.vertical, 8)
.contentShape(Rectangle())
.gesture(scrubGesture)
.contextMenu { timelineMenu }
}
.onChange(of: model.selection) { _, newValue in
guard let newValue else { return }
withAnimation(.easeOut(duration: 0.2)) {
proxy.scrollTo(newValue, anchor: .center)
}
}
}
}
// MARK: - Barra de controles
private var toolbar: some View {
HStack(spacing: 12) {
Button {
model.togglePlay()
} label: {
Image(systemName: model.isPlaying ? "pause.fill" : "play.fill")
}
.buttonStyle(.borderless)
.help("Reproduzir (espaço)")
.disabled(model.player == nil)
Text(timecode(model.currentTime))
.font(.system(.caption, design: .monospaced))
.foregroundStyle(.secondary)
Button {
model.playSelectedPhrase()
} label: {
Image(systemName: "play.rectangle")
}
.buttonStyle(.borderless)
.help("Tocar só a frase selecionada (⏎)")
.disabled(model.player == nil || model.selection == nil)
Toggle("Pular removidos", isOn: $model.skipRemoved)
.toggleStyle(.checkbox)
.font(.caption)
.help("Durante a reprodução, salta os trechos desativados — mostra como o corte ficou.")
Button {
model.addZoomForRange()
} label: {
Label("Zoom no trecho", systemImage: "plus.magnifyingglass")
}
.buttonStyle(.borderless)
.font(.caption)
.disabled(!model.hasRange)
.help("Arraste na timeline para marcar um trecho e crie um zoom nele. A escala vem de Análise de Voz.")
Spacer()
legend
Spacer()
Image(systemName: "minus.magnifyingglass").foregroundStyle(.secondary)
Slider(value: $model.pixelsPerSecond,
in: model.minPixelsPerSecond...model.maxPixelsPerSecond)
.frame(width: 130)
Image(systemName: "plus.magnifyingglass").foregroundStyle(.secondary)
}
.padding(.horizontal, 12)
.padding(.vertical, 8)
}
private var legend: some View {
HStack(spacing: 10) {
ForEach(0..<4, id: \.self) { level in
HStack(spacing: 4) {
RoundedRectangle(cornerRadius: 2)
.fill(EmphasisPalette.color(level))
.frame(width: 10, height: 10)
Text(EmphasisPalette.label(level)).font(.caption2)
}
}
}
.foregroundStyle(.secondary)
}
// MARK: - Trilhas
private var ruler: some View {
Canvas { context, size in
let step = tickStep()
var time = 0.0
while time <= model.duration {
let position = x(time)
context.stroke(
Path { $0.move(to: CGPoint(x: position, y: size.height - 6))
$0.addLine(to: CGPoint(x: position, y: size.height)) },
with: .color(.secondary.opacity(0.5))
)
context.draw(
Text(timecode(time)).font(.system(size: 9, design: .monospaced))
.foregroundColor(.secondary),
at: CGPoint(x: position + 18, y: 6)
)
time += step
}
}
.frame(width: contentWidth, height: rulerHeight)
}
private var phraseTrack: some View {
ZStack(alignment: .topLeading) {
RoundedRectangle(cornerRadius: 4)
.fill(Color.secondary.opacity(0.06))
.frame(width: contentWidth, height: phraseHeight)
ForEach(model.phrases) { phrase in
phraseBlock(phrase)
}
}
.frame(width: contentWidth, height: phraseHeight, alignment: .topLeading)
}
@ViewBuilder
private func phraseBlock(_ phrase: ReviewPhrase) -> some View {
let isSelected = model.selection == phrase.id
let color = EmphasisPalette.color(phrase.emphasis)
let fullWidth = max(2, width(from: phrase.start, to: phrase.end))
let keptWidth = max(1, width(from: phrase.trimStart, to: phrase.trimEnd))
ZStack(alignment: .topLeading) {
// The whole line, dim — what is there before the edit.
RoundedRectangle(cornerRadius: 4)
.fill(color.opacity(phrase.active ? 0.15 : 0.10))
.frame(width: fullWidth, height: phraseHeight)
// What survives: the kept span, drawn solid over it.
RoundedRectangle(cornerRadius: 4)
.fill(color.opacity(phrase.active ? 0.55 : 0.12))
.frame(width: keptWidth, height: phraseHeight)
.offset(x: width(from: phrase.start, to: phrase.trimStart))
Text(phrase.text)
.font(.system(size: 10))
.lineLimit(2)
.padding(.horizontal, 4)
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
.foregroundStyle(phrase.active ? .primary : .secondary)
.strikethrough(!phrase.active)
RoundedRectangle(cornerRadius: 4)
.stroke(isSelected ? Color.accentColor : color.opacity(0.4),
lineWidth: isSelected ? 2 : 1)
.frame(width: fullWidth, height: phraseHeight)
if isSelected && phrase.active {
trimHandle(phrase, edge: .start)
trimHandle(phrase, edge: .end)
}
}
.frame(width: fullWidth, height: phraseHeight, alignment: .topLeading)
.offset(x: x(phrase.start))
.contentShape(Rectangle())
.onTapGesture { model.goTo(phraseID: phrase.id) }
.contextMenu { phraseMenu(phrase) }
.help(phrase.reason.isEmpty ? phrase.text : "\(phrase.text)\n— \(phrase.reason)")
}
private func trimHandle(_ phrase: ReviewPhrase, edge: TrimEdge) -> some View {
let offset = edge == .start
? width(from: phrase.start, to: phrase.trimStart)
: width(from: phrase.start, to: phrase.trimEnd) - handleWidth
return RoundedRectangle(cornerRadius: 2)
.fill(Color.accentColor)
.frame(width: handleWidth, height: phraseHeight)
.offset(x: offset)
.gesture(
DragGesture(minimumDistance: 1)
.onChanged { value in
let time = phrase.start + Double((value.location.x) / pps)
model.trim(phrase.id, edge: edge, to: time)
}
)
.help(edge == .start ? "Arraste para cortar o começo (pula de palavra em palavra)"
: "Arraste para cortar o fim (pula de palavra em palavra)")
}
@ViewBuilder
private func phraseMenu(_ phrase: ReviewPhrase) -> some View {
Button("Tocar esta frase") {
model.goTo(phraseID: phrase.id)
model.playSelectedPhrase()
}
Button(phrase.active ? "Remover do corte" : "Trazer de volta") {
model.toggleActive(phrase.id)
}
Button("Adicionar zoom nesta frase") { model.addZoomForPhrase(phrase.id) }
Divider()
ForEach(0..<4, id: \.self) { level in
Button("Ênfase: \(EmphasisPalette.label(level))") {
model.setEmphasis(level, for: phrase.id)
}
}
Divider()
Button(phrase.isBackstage ? "Marcar como roteiro" : "Marcar como bastidor") {
model.setTrack(phrase.isBackstage ? ReviewPhrase.trackScript : ReviewPhrase.trackBackstage,
for: phrase.id)
}
if phrase.isTrimmed {
Divider()
Button("Desfazer corte da frase") { model.resetTrim(phrase.id) }
}
}
/// Per-word energy/emphasis, straight from the voice timeline — the closest
/// thing to a waveform without opening the audio again.
private var energyTrack: some View {
Canvas { context, size in
for phrase in model.phrases {
for word in phrase.words {
let start = x(word.start)
let barWidth = max(1, width(from: word.start, to: word.end) - 1)
let height = size.height * CGFloat(max(0.04, word.energy))
let rect = CGRect(x: start, y: size.height - height,
width: barWidth, height: height)
let color = word.emphasis >= 0.65 ? Color.pink
: word.emphasis >= 0.45 ? Color.orange
: Color.secondary
context.fill(Path(rect),
with: .color(color.opacity(phrase.active ? 0.6 : 0.2)))
}
}
}
.frame(width: contentWidth, height: energyHeight)
.background(RoundedRectangle(cornerRadius: 4).fill(Color.secondary.opacity(0.06)))
}
private var speakerTrack: some View {
stripTrack { phrase in
EmphasisPalette.speakerColor(phrase.speaker, among: model.speakers)
}
}
private var scriptTrack: some View {
stripTrack { phrase in phrase.isBackstage ? Color.gray : Color.mint }
}
private func stripTrack(_ color: @escaping (ReviewPhrase) -> Color) -> some View {
Canvas { context, size in
for phrase in model.phrases {
let rect = CGRect(x: x(phrase.start), y: 0,
width: max(1, width(from: phrase.start, to: phrase.end)),
height: size.height)
context.fill(Path(roundedRect: rect, cornerRadius: 2),
with: .color(color(phrase).opacity(phrase.active ? 0.7 : 0.2)))
}
}
.frame(width: contentWidth, height: stripHeight)
}
private var playhead: some View {
Rectangle()
.fill(Color.red)
.frame(width: 1.5)
.offset(x: x(model.currentTime))
.allowsHitTesting(false)
}
/// One gesture, two meanings, decided by whether the mouse moved: a click
/// parks the playhead, a drag marks in/out. Splitting them across separate
/// controls would mean choosing a tool before every action, which is
/// exactly the ceremony this screen is meant to avoid.
private var scrubGesture: some Gesture {
DragGesture(minimumDistance: 0)
.onChanged { value in
let from = Double(value.startLocation.x / pps)
let to = Double(value.location.x / pps)
if abs(value.translation.width) > 3 {
model.setRange(from: from, to: to)
model.seek(to: min(from, to))
} else {
model.clearRange()
model.seek(to: to)
}
}
}
/// The marked in/out, drawn over every track so the span reads against the
/// phrases and the energy at once.
private var rangeOverlay: some View {
Group {
if let span = model.rangeSpan {
Rectangle()
.fill(Color.accentColor.opacity(0.18))
.overlay(Rectangle().stroke(Color.accentColor.opacity(0.6), lineWidth: 1))
.frame(width: max(1, width(from: span.start, to: span.end)))
.offset(x: x(span.start))
.allowsHitTesting(false)
}
}
}
@ViewBuilder
private var timelineMenu: some View {
if model.hasRange, let span = model.rangeSpan {
Button("Adicionar zoom no trecho (\(secondsLabel(span.end - span.start)))") {
model.addZoomForRange()
}
Button("Tocar o trecho") { model.playRange(from: span.start, to: span.end) }
Button("Limpar seleção") { model.clearRange() }
} else {
Text("Arraste na timeline para marcar um trecho")
}
if let zoom = model.zoom(at: model.currentTime) {
Divider()
Button("Remover o zoom daqui") { model.removeZoom(zoom.id) }
}
}
private func secondsLabel(_ seconds: Double) -> String {
String(format: "%.1fs", seconds)
}
/// Punch-ins, on their own lane above the script: they are a second layer
/// over the same time, not a property of a phrase.
private var zoomTrack: some View {
ZStack(alignment: .topLeading) {
RoundedRectangle(cornerRadius: 3)
.fill(Color.secondary.opacity(0.06))
.frame(width: contentWidth, height: stripHeight + 6)
ForEach(model.zooms) { zoom in
RoundedRectangle(cornerRadius: 3)
.fill(Color.yellow.opacity(0.55))
.overlay(
Image(systemName: "plus.magnifyingglass")
.font(.system(size: 8)).foregroundStyle(.black.opacity(0.6))
)
.frame(width: max(6, width(from: zoom.start, to: zoom.end)),
height: stripHeight + 6)
.offset(x: x(zoom.start))
.help("Zoom marcado — \(secondsLabel(zoom.end - zoom.start)). A escala vem de Análise de Voz.")
.contextMenu {
Button("Remover este zoom") { model.removeZoom(zoom.id) }
}
}
}
.frame(width: contentWidth, height: stripHeight + 6, alignment: .topLeading)
}
/// Delivery emotion per phrase — the fourth signal to read against the text.
private var emotionTrack: some View {
stripTrack { phrase in
switch phrase.emotion {
case "excited": return .orange
case "tense": return .red
case "calm": return .blue
case "reflective": return .purple
default: return .secondary
}
}
}
// MARK: - Escala
private func x(_ time: Double) -> CGFloat { CGFloat(time) * pps }
private func width(from: Double, to: Double) -> CGFloat {
max(0, CGFloat(to - from) * pps)
}
/// Ruler spacing that keeps labels ~80pt apart at any zoom.
private func tickStep() -> Double {
let candidates: [Double] = [1, 2, 5, 10, 15, 30, 60, 120, 300, 600]
let wanted = 80 / Double(pps)
return candidates.first { $0 >= wanted } ?? 600
}
private func timecode(_ seconds: Double) -> String {
let total = Int(seconds.rounded(.down))
return String(format: "%02d:%02d", total / 60, total % 60)
}
}