feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e7748c2c58
commit
1bebee4359
@@ -116,6 +116,145 @@ struct ZoomClip: Identifiable {
|
||||
}
|
||||
}
|
||||
|
||||
/// One word inside a phrase, with the acoustics that justify an emphasis.
|
||||
struct ReviewWord: Identifiable {
|
||||
let id: Int
|
||||
let text: String
|
||||
let start: Double
|
||||
let end: Double
|
||||
let energy: Double
|
||||
let emphasis: Double
|
||||
|
||||
init(id: Int, json: [String: Any]) {
|
||||
self.id = id
|
||||
text = json["text"] as? String ?? ""
|
||||
start = json["start"] as? Double ?? 0
|
||||
end = json["end"] as? Double ?? 0
|
||||
energy = json["energy"] as? Double ?? 0
|
||||
emphasis = json["emphasis"] as? Double ?? 0
|
||||
}
|
||||
}
|
||||
|
||||
/// A phrase in the review step — one spoken line plus the decision made about
|
||||
/// it. Mirrors `fcpxml/phrase_review.py`; `emphasis` is 0–3 and everything
|
||||
/// mutable here is what the editor is allowed to change.
|
||||
struct ReviewPhrase: Identifiable {
|
||||
let id: Int
|
||||
let start: Double
|
||||
let end: Double
|
||||
var trimStart: Double
|
||||
var trimEnd: Double
|
||||
var text: String
|
||||
let speaker: String
|
||||
var active: Bool
|
||||
var emphasis: Int
|
||||
var track: String
|
||||
let peakEmphasis: Double
|
||||
let emotion: String
|
||||
let emotionConfidence: Double
|
||||
let takeBoundary: Bool
|
||||
let gapBefore: Double
|
||||
let reason: String
|
||||
let words: [ReviewWord]
|
||||
|
||||
static let trackScript = "roteiro"
|
||||
static let trackBackstage = "bastidor"
|
||||
|
||||
/// Delivery emotion as the analysis names it, in the user's language plus a
|
||||
/// glyph — the label alone is too easy to skim past in a dense list.
|
||||
static func emotionLabel(_ emotion: String) -> (String, String) {
|
||||
switch emotion {
|
||||
case "excited": return ("Empolgado", "flame")
|
||||
case "tense": return ("Tenso", "bolt")
|
||||
case "calm": return ("Calmo", "leaf")
|
||||
case "reflective": return ("Reflexivo", "moon")
|
||||
default: return ("Neutro", "circle")
|
||||
}
|
||||
}
|
||||
|
||||
init(json: [String: Any]) {
|
||||
id = json["index"] as? Int ?? 0
|
||||
start = json["start"] as? Double ?? 0
|
||||
end = json["end"] as? Double ?? 0
|
||||
trimStart = json["trim_start"] as? Double ?? (json["start"] as? Double ?? 0)
|
||||
trimEnd = json["trim_end"] as? Double ?? (json["end"] as? Double ?? 0)
|
||||
text = json["text"] as? String ?? ""
|
||||
speaker = json["speaker"] as? String ?? ""
|
||||
active = json["active"] as? Bool ?? true
|
||||
emphasis = json["emphasis"] as? Int ?? 0
|
||||
track = json["track"] as? String ?? ReviewPhrase.trackScript
|
||||
peakEmphasis = json["peak_emphasis"] as? Double ?? 0
|
||||
emotion = json["emotion"] as? String ?? "neutral"
|
||||
emotionConfidence = json["emotion_confidence"] as? Double ?? 0
|
||||
takeBoundary = json["take_boundary"] as? Bool ?? false
|
||||
gapBefore = json["gap_before"] as? Double ?? 0
|
||||
reason = json["reason"] as? String ?? ""
|
||||
words = (json["words"] as? [[String: Any]] ?? [])
|
||||
.enumerated().map { ReviewWord(id: $0.offset, json: $0.element) }
|
||||
}
|
||||
|
||||
var asJSON: [String: Any] {
|
||||
[
|
||||
"index": id,
|
||||
"start": start,
|
||||
"end": end,
|
||||
"trim_start": trimStart,
|
||||
"trim_end": trimEnd,
|
||||
"text": text,
|
||||
"speaker": speaker,
|
||||
"active": active,
|
||||
"emphasis": emphasis,
|
||||
"track": track,
|
||||
"reason": reason,
|
||||
]
|
||||
}
|
||||
|
||||
var isBackstage: Bool { track == ReviewPhrase.trackBackstage }
|
||||
var isTrimmed: Bool { trimStart > start + 0.001 || trimEnd < end - 0.001 }
|
||||
var timecode: String {
|
||||
String(format: "%02d:%02d", Int(start) / 60, Int(start) % 60)
|
||||
}
|
||||
|
||||
/// The word boundaries a trim handle is allowed to land on.
|
||||
func snap(_ time: Double, edge: TrimEdge) -> Double {
|
||||
let boundaries = words.map { edge == .start ? $0.start : $0.end }.filter { $0 > 0 }
|
||||
guard let nearest = boundaries.min(by: { abs($0 - time) < abs($1 - time) }) else {
|
||||
return time
|
||||
}
|
||||
return nearest
|
||||
}
|
||||
}
|
||||
|
||||
enum TrimEdge { case start, end }
|
||||
|
||||
/// A punch-in the editor placed by hand over an arbitrary range, next to the
|
||||
/// whole-phrase zoom that an emphasis level produces. It stores only *when* —
|
||||
/// the scale and the ramp come from the Voice Analysis settings at render time.
|
||||
struct ManualZoom: Identifiable {
|
||||
let id = UUID()
|
||||
var start: Double
|
||||
var end: Double
|
||||
|
||||
/// Below this a punch-in has no room to ramp in and back out; the writer
|
||||
/// rejects the window, so offering it would place nothing.
|
||||
static let minimumDuration: Double = 0.4
|
||||
|
||||
init(start: Double, end: Double) {
|
||||
self.start = start
|
||||
self.end = end
|
||||
}
|
||||
|
||||
init?(json: [String: Any]) {
|
||||
guard let start = json["start"] as? Double, let end = json["end"] as? Double,
|
||||
end - start >= ManualZoom.minimumDuration
|
||||
else { return nil }
|
||||
self.start = start
|
||||
self.end = end
|
||||
}
|
||||
|
||||
var asJSON: [String: Any] { ["start": start, "end": end] }
|
||||
}
|
||||
|
||||
struct ZoomSegment: Identifiable {
|
||||
let id: Int
|
||||
let start: Double
|
||||
|
||||
Reference in New Issue
Block a user