feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
711c397dfe
commit
7b5aed79ee
@@ -28,30 +28,6 @@ final class PhraseReviewModel: ObservableObject {
|
||||
/// In/out the editor dragged on the timeline, in source seconds.
|
||||
@Published var rangeStart: Double?
|
||||
@Published var rangeEnd: Double?
|
||||
/// Aspect ratio of the footage as recorded.
|
||||
@Published var videoAspect: Double = 16.0 / 9.0
|
||||
/// Aspect ratio the project delivers in, read from the .fcpxml. It is
|
||||
/// routinely *not* the footage's: these takes are shot horizontal and
|
||||
/// delivered vertical, so previewing the raw frame would show a crop the
|
||||
/// audience never sees — and the emphasis decisions are about what lands on
|
||||
/// screen. Nil until the project is known.
|
||||
@Published var projectAspect: Double?
|
||||
/// Whether the preview crops to the delivery frame. On by default whenever
|
||||
/// the two aspects disagree.
|
||||
@Published var matchProjectFraming = true
|
||||
|
||||
/// What the preview should actually draw.
|
||||
var previewAspect: Double {
|
||||
guard matchProjectFraming, let projectAspect else { return videoAspect }
|
||||
return projectAspect
|
||||
}
|
||||
|
||||
/// True when the delivery frame differs enough from the footage that the
|
||||
/// preview is showing a crop rather than the whole take.
|
||||
var isCropping: Bool {
|
||||
guard matchProjectFraming, let projectAspect else { return false }
|
||||
return abs(projectAspect - videoAspect) > 0.01
|
||||
}
|
||||
|
||||
private(set) var source = ""
|
||||
private(set) var sourcePath = ""
|
||||
@@ -76,25 +52,17 @@ final class PhraseReviewModel: ObservableObject {
|
||||
// MARK: - Carregar
|
||||
|
||||
/// Builds the review from the voice timeline plus whatever the AI decided.
|
||||
/// A review saved on a previous visit wins — see `cmd_build_phrase_review`.
|
||||
/// Reads the delivery format from the project so the preview can frame the
|
||||
/// take the way it will actually be seen.
|
||||
func loadProjectFormat(projectPath: String) {
|
||||
PythonBridge.call(command: "inspect", arguments: ["path": projectPath]) { [weak self] result, _ in
|
||||
Task { @MainActor in
|
||||
guard let self,
|
||||
let timelines = result?["timelines"] as? [[String: Any]],
|
||||
let first = timelines.first,
|
||||
let width = first["width"] as? Int, let height = first["height"] as? Int,
|
||||
width > 0, height > 0
|
||||
else { return }
|
||||
self.projectAspect = Double(width) / Double(height)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// A review saved on a previous visit wins — see `cmd_build_phrase_review` —
|
||||
/// UNLESS `fresh` is true, in which case that saved review is ignored and
|
||||
/// `active`/`emphasis`/etc. come straight from this call's `decisionsJSON`.
|
||||
/// Pass `fresh: true` when the decisions themselves changed since the
|
||||
/// review was last built (the caller re-pasted/regenerated the AI's JSON
|
||||
/// and re-ran `apply_voice_actions`) — otherwise the saved review from the
|
||||
/// PREVIOUS decisions silently wins over the fresh cut it should reflect,
|
||||
/// which is exactly the desync the wizard's "active" toggle showed against
|
||||
/// the just-reapplied FCPXML.
|
||||
func load(voiceTimelinePath: String, decisionsJSON: String,
|
||||
outputFolder: String? = nil, mediaFolder: String? = nil) {
|
||||
outputFolder: String? = nil, mediaFolder: String? = nil, fresh: Bool = false) {
|
||||
self.voiceTimelinePath = voiceTimelinePath
|
||||
isLoading = true
|
||||
errorMessage = nil
|
||||
@@ -102,6 +70,7 @@ final class PhraseReviewModel: ObservableObject {
|
||||
var arguments: [String: Any] = ["voice_timeline": voiceTimelinePath]
|
||||
if let outputFolder { arguments["output_dir"] = outputFolder }
|
||||
if let mediaFolder { arguments["media_dir"] = mediaFolder }
|
||||
if fresh { arguments["fresh"] = true }
|
||||
if let data = decisionsJSON.data(using: .utf8),
|
||||
let parsed = try? JSONSerialization.jsonObject(with: data) {
|
||||
arguments["actions"] = parsed
|
||||
@@ -160,7 +129,6 @@ final class PhraseReviewModel: ObservableObject {
|
||||
let asset = AVURLAsset(url: URL(fileURLWithPath: sourcePath))
|
||||
let player = AVPlayer(playerItem: AVPlayerItem(asset: asset))
|
||||
self.player = player
|
||||
readAspect(from: asset)
|
||||
// 60 Hz: the same observer drives the playhead *and* decides when to
|
||||
// jump a removed stretch, so its period is the worst-case amount of cut
|
||||
// material that can be heard before the skip lands. At 20 Hz that was an
|
||||
@@ -173,22 +141,6 @@ final class PhraseReviewModel: ObservableObject {
|
||||
}
|
||||
}
|
||||
|
||||
/// The displayed aspect ratio, honouring the rotation the camera recorded.
|
||||
/// A phone take is stored 1920×1080 with a 90° transform: reading
|
||||
/// `naturalSize` alone would call a vertical video horizontal.
|
||||
private func readAspect(from asset: AVURLAsset) {
|
||||
Task { [weak self] in
|
||||
guard let track = try? await asset.loadTracks(withMediaType: .video).first,
|
||||
let size = try? await track.load(.naturalSize),
|
||||
let transform = try? await track.load(.preferredTransform)
|
||||
else { return }
|
||||
let displayed = size.applying(transform)
|
||||
let width = abs(displayed.width), height = abs(displayed.height)
|
||||
guard width > 0, height > 0 else { return }
|
||||
await MainActor.run { self?.videoAspect = width / height }
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Reprodução
|
||||
|
||||
private func tick(_ time: Double) {
|
||||
@@ -335,8 +287,21 @@ final class PhraseReviewModel: ObservableObject {
|
||||
|
||||
/// Trim everything before/after a given word — the text-first way to cut,
|
||||
/// since the editor reads the line and points at where it should begin.
|
||||
/// Clicking the word that is ALREADY that edge toggles it back off —
|
||||
/// the trim on that side resets to the phrase's own start/end — so the
|
||||
/// same click that sets a boundary also clears it, instead of needing
|
||||
/// the separate "Inteira" button for a one-sided undo.
|
||||
func trimToWord(_ word: ReviewWord, edge: TrimEdge, in id: Int) {
|
||||
trim(id, edge: edge, to: edge == .start ? word.start : word.end)
|
||||
guard let phrase = phrases.first(where: { $0.id == id }) else { return }
|
||||
let epsilon = 0.001
|
||||
switch edge {
|
||||
case .start where abs(word.start - phrase.trimStart) < epsilon:
|
||||
update(id) { $0.trimStart = $0.start }
|
||||
case .end where abs(word.end - phrase.trimEnd) < epsilon:
|
||||
update(id) { $0.trimEnd = $0.end }
|
||||
default:
|
||||
trim(id, edge: edge, to: edge == .start ? word.start : word.end)
|
||||
}
|
||||
}
|
||||
|
||||
// MARK: - Trecho marcado e zooms
|
||||
@@ -400,11 +365,59 @@ final class PhraseReviewModel: ObservableObject {
|
||||
phrases.filter { $0.active }.reduce(0) { $0 + ($1.trimEnd - $1.trimStart) }
|
||||
}
|
||||
|
||||
// MARK: - Tempo compactado (sem os vãos do que foi cortado)
|
||||
|
||||
/// Kept spans of source media, in order, each carrying the position it
|
||||
/// lands at once every removed stretch between phrases is squeezed out.
|
||||
/// The timeline draws and scrubs in this space so it reads like the cut
|
||||
/// itself instead of the raw take with holes in it.
|
||||
private var keptSegments: [(rawStart: Double, rawEnd: Double, compactStart: Double)] {
|
||||
var offset = 0.0
|
||||
var segments: [(Double, Double, Double)] = []
|
||||
for phrase in phrases.sorted(by: { $0.start < $1.start }) where phrase.active {
|
||||
guard phrase.trimEnd > phrase.trimStart else { continue }
|
||||
segments.append((phrase.trimStart, phrase.trimEnd, offset))
|
||||
offset += phrase.trimEnd - phrase.trimStart
|
||||
}
|
||||
return segments
|
||||
}
|
||||
|
||||
/// Maps a raw source-media time to its position on the compacted timeline.
|
||||
/// Time inside removed material collapses to the boundary of the nearest
|
||||
/// kept segment, so cut stretches take up no space at all.
|
||||
func compactTime(_ raw: Double) -> Double {
|
||||
let segments = keptSegments
|
||||
for segment in segments {
|
||||
if raw < segment.rawStart { return segment.compactStart }
|
||||
if raw <= segment.rawEnd { return segment.compactStart + (raw - segment.rawStart) }
|
||||
}
|
||||
guard let last = segments.last else { return 0 }
|
||||
return raw >= last.rawEnd ? last.compactStart + (last.rawEnd - last.rawStart) : 0
|
||||
}
|
||||
|
||||
/// The inverse of `compactTime`: where a click on the compacted timeline
|
||||
/// lands in the raw source media, for seeking and scrubbing.
|
||||
func rawTime(fromCompact compact: Double) -> Double {
|
||||
let segments = keptSegments
|
||||
for segment in segments {
|
||||
let compactEnd = segment.compactStart + (segment.rawEnd - segment.rawStart)
|
||||
if compact <= compactEnd {
|
||||
return segment.rawStart + max(0, compact - segment.compactStart)
|
||||
}
|
||||
}
|
||||
return segments.last?.rawEnd ?? 0
|
||||
}
|
||||
|
||||
/// Persists the edited review plus the actions derived from it. Called when
|
||||
/// the wizard advances — the render itself happens in the next step.
|
||||
func save(completion: @escaping (String?) -> Void) {
|
||||
/// Persists the edited review and hands back BOTH paths it wrote:
|
||||
/// `review_path` (the human-readable `_phrase_review.json`) and
|
||||
/// `actions_path` (`_phrase_actions.json`, the cut/zoom list derived from
|
||||
/// it — what `finalizeProcessing` needs to actually apply the review's
|
||||
/// active/inactive decisions instead of just filing them away).
|
||||
func save(completion: @escaping (_ reviewPath: String?, _ actionsPath: String?) -> Void) {
|
||||
guard !voiceTimelinePath.isEmpty, !phrases.isEmpty else {
|
||||
completion(nil)
|
||||
completion(nil, nil)
|
||||
return
|
||||
}
|
||||
let arguments: [String: Any] = [
|
||||
@@ -418,11 +431,11 @@ final class PhraseReviewModel: ObservableObject {
|
||||
PythonBridge.call(command: "save_phrase_review", arguments: arguments) { result, error in
|
||||
Task { @MainActor in
|
||||
if let error {
|
||||
completion(nil)
|
||||
completion(nil, nil)
|
||||
_ = error
|
||||
return
|
||||
}
|
||||
completion(result?["review_path"] as? String)
|
||||
completion(result?["review_path"] as? String, result?["actions_path"] as? String)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user