chore: atualização geral

This commit is contained in:
João Henrique
2026-08-19 16:35:29 -04:00
parent 8fca456ceb
commit e7748c2c58
66 changed files with 13037 additions and 4237 deletions
+290 -117
View File
@@ -16,6 +16,8 @@ struct TranscriptionView: View {
@State private var processedPath: String?
@State private var isRemovingSilences = false
@State private var silencePadding: Double = 0.05
@State private var silenceNoiseDb: Double = -30
@State private var silenceMinDuration: Double = 0.5
@State private var isRemovingFillers = false
@State private var fillerRemovedPath: String?
@State private var phraseInput = ""
@@ -39,27 +41,19 @@ struct TranscriptionView: View {
@State private var selectedZoomEndID: Int?
@State private var zoomMessage = ""
@State private var outputFolder: String?
@State private var batchVoiceAnalysis = false
@State private var batchVoiceEdit = false
@State private var voiceActionsPath: String?
@State private var batchSilences = true
@State private var batchFillers = false
@State private var batchPhrases = false
@State private var batchMarkers = false
@State private var batchSubtitles = true
@State private var batchDynamicSubtitles = false
@State private var dynamicSubtitlesBandHeight: Double = 0.22
@State private var dynamicSubtitlesBlockCenterY: Double = -167
@State private var dynamicSubtitlesFont = "Helvetica Neue"
@State private var dynamicSubtitlesFontSize: Double = 90
@State private var dynamicSubtitlesActiveColor = Color.white
@State private var dynamicSubtitlesInactiveColor = Color(white: 0.7)
@State private var dynamicSubtitlesPath: String?
@State private var isBatchProcessing = false
@State private var batchStatus = ""
private let dynamicSubtitlesFontChoices = [
"Helvetica Neue", "Helvetica", "Arial", "Avenir Next",
"Futura", "SF Pro Display", "Georgia", "Impact",
]
init(projectPath: String? = nil, embedded: Bool = false) {
self.embedded = embedded
_projectPath = State(initialValue: projectPath)
@@ -133,6 +127,19 @@ struct TranscriptionView: View {
.disabled(isRunning || projectPath == nil || outputFolder == nil || hasNoInstalledModel)
.frame(maxWidth: .infinity)
.buttonStyle(.borderedProminent)
if isRunning {
VStack(alignment: .leading, spacing: 6) {
ProgressView(value: progress)
HStack {
Text(stage.isEmpty ? "Processando o áudio…" : stage)
Spacer()
Text("\(Int(progress * 100))%").monospacedDigit()
}
.font(.caption).foregroundStyle(.secondary)
}
.padding(.top, 4)
}
}
.disabled(outputFolder == nil)
@@ -143,15 +150,6 @@ struct TranscriptionView: View {
}
}
if isRunning {
Section {
VStack(alignment: .leading, spacing: 8) {
ProgressView(value: progress)
Text("\(Int(progress * 100))%").font(.caption).foregroundStyle(.secondary)
}
}
}
if !results.isEmpty {
Section("Transcrição Concluída") {
ForEach(results, id: \.media) { r in
@@ -176,105 +174,174 @@ struct TranscriptionView: View {
if outputFolder == nil {
Text("Selecione a pasta do projeto para habilitar o processamento.")
.font(.caption2).foregroundStyle(.secondary)
} else if results.isEmpty {
Label(
isRunning ? "Aguardando a transcrição terminar…" : "Transcreva o projeto acima para liberar o processamento.",
systemImage: "lock.fill"
)
.font(.caption).foregroundStyle(.orange)
}
GroupBox("Processar em lote") {
VStack(alignment: .leading, spacing: 6) {
Toggle("Remover silêncios do áudio", isOn: $batchSilences)
VStack(alignment: .leading, spacing: 4) {
HStack {
Text("Tolerância do corte")
.foregroundStyle(batchSilences ? .primary : .secondary)
Spacer()
Text(String(format: "%.2fs", silencePadding))
.font(.caption).foregroundStyle(.secondary).monospacedDigit()
}
Slider(value: $silencePadding, in: 0...2, step: 0.01)
.disabled(!batchSilences)
Text("Quanto de silêncio sobra em volta de cada corte.")
.font(.caption2).foregroundStyle(.secondary)
GroupBox {
VStack(alignment: .leading, spacing: 18) {
batchOptionRow(
toggle: Toggle("Analisar voz (transcrição, locutor, ênfase)", isOn: $batchVoiceAnalysis),
expanded: batchVoiceAnalysis
) {
Text("Gera o JSON com transcrição, diarização e intensidade (pitch/energia/ritmo) por palavra — a base que o corte por voz usa para decidir tomadas e zooms sem reabrir o áudio depois. Só análise: não corta nada.")
.font(.caption).foregroundStyle(.secondary)
}
.padding(.leading, 20)
Divider()
batchOptionRow(
toggle: Toggle("Aplicar edição por voz (lista de decisões)", isOn: $batchVoiceEdit),
expanded: batchVoiceEdit
) {
VStack(alignment: .leading, spacing: 6) {
Text("Aplica um JSON de decisões — cortes, zooms, textos e marcadores — gerado a partir da análise de voz. É o passo que descarta bastidor e tomadas repetidas.")
.font(.caption).foregroundStyle(.secondary)
HStack {
Image(systemName: "doc.text")
Text(voiceActionsPath.map { URL(fileURLWithPath: $0).lastPathComponent }
?? "Nenhum arquivo de decisões")
.foregroundStyle(voiceActionsPath == nil ? .secondary : .primary)
.lineLimit(1).truncationMode(.middle)
Spacer()
Button("Escolher…") { pickVoiceActions() }
}
}
}
Divider()
batchOptionRow(
toggle: Toggle("Remover silêncios do áudio", isOn: $batchSilences),
expanded: true
) {
VStack(alignment: .leading, spacing: 6) {
HStack {
Text("Tolerância do corte")
.font(.callout)
.foregroundStyle(.secondary)
Spacer()
Text(String(format: "%.2fs", silencePadding))
.font(.callout).foregroundStyle(.secondary).monospacedDigit()
}
Slider(value: $silencePadding, in: 0...2, step: 0.01)
.disabled(!batchSilences)
.onChange(of: silencePadding) { _, _ in saveSilenceConfig() }
HStack {
Text("Limiar de silêncio")
.font(.callout)
.foregroundStyle(.secondary)
Spacer()
Text(String(format: "%.0f dB", silenceNoiseDb))
.font(.callout).foregroundStyle(.secondary).monospacedDigit()
}
Slider(value: $silenceNoiseDb, in: -60...(-10), step: 1)
.disabled(!batchSilences)
.onChange(of: silenceNoiseDb) { _, _ in saveSilenceConfig() }
HStack {
Text("Duração mínima")
.font(.callout)
.foregroundStyle(.secondary)
Spacer()
Text(String(format: "%.2fs", silenceMinDuration))
.font(.callout).foregroundStyle(.secondary).monospacedDigit()
}
Slider(value: $silenceMinDuration, in: 0.1...5, step: 0.05)
.disabled(!batchSilences)
.onChange(of: silenceMinDuration) { _, _ in saveSilenceConfig() }
Text("Quanto de silêncio sobra em volta de cada corte, abaixo de que volume conta como silêncio, e quanto tempo ele precisa durar. Fica salvo e vale também fora do app.")
.font(.caption).foregroundStyle(.secondary)
}
}
Divider()
Toggle("Remover palavras de preenchimento", isOn: $batchFillers)
Toggle("Cortar frases ditas", isOn: $batchPhrases)
if batchPhrases {
Divider()
batchOptionRow(
toggle: Toggle("Cortar frases ditas", isOn: $batchPhrases),
expanded: batchPhrases
) {
TextField("Frases separadas por vírgula", text: $phraseInput)
.textFieldStyle(.roundedBorder)
.padding(.leading, 20)
}
Divider()
Toggle("Marcar o que foi dito na timeline", isOn: $batchMarkers)
Divider()
Toggle("Exportar legendas SRT", isOn: $batchSubtitles)
Toggle("Gerar legendas dinâmicas (cascata, editáveis no FCP)", isOn: $batchDynamicSubtitles)
if batchDynamicSubtitles {
VStack(alignment: .leading, spacing: 6) {
Picker("Fonte", selection: $dynamicSubtitlesFont) {
ForEach(dynamicSubtitlesFontChoices, id: \.self) { Text($0).tag($0) }
}
HStack {
Text("Tamanho da fonte")
Spacer()
Text("\(Int(dynamicSubtitlesFontSize))pt")
.font(.caption).foregroundStyle(.secondary).monospacedDigit()
}
Slider(value: $dynamicSubtitlesFontSize, in: 20...200, step: 1)
ColorPicker("Cor A", selection: $dynamicSubtitlesActiveColor, supportsOpacity: true)
ColorPicker("Cor B", selection: $dynamicSubtitlesInactiveColor, supportsOpacity: true)
HStack {
Text("Altura do bloco")
Spacer()
Text("\(Int(dynamicSubtitlesBandHeight * 100))%")
.font(.caption).foregroundStyle(.secondary).monospacedDigit()
}
Slider(value: $dynamicSubtitlesBandHeight, in: 0.10...0.50, step: 0.01)
HStack {
Text("Altura na tela")
Spacer()
Text("\(Int(dynamicSubtitlesBlockCenterY))")
.font(.caption).foregroundStyle(.secondary).monospacedDigit()
}
Slider(value: $dynamicSubtitlesBlockCenterY, in: -700...300, step: 1)
Text("Roda por último, depois dos outros passos marcados acima, usando o timing já cortado da timeline.")
.font(.caption2).foregroundStyle(.secondary)
}
.padding(.leading, 20)
}
Button {
processBatch()
} label: {
if isBatchProcessing { ProgressView().controlSize(.small) }
Label("Processar selecionados", systemImage: "play.fill")
}
.disabled(isBatchProcessing || isRunning || projectPath == nil || outputFolder == nil
|| !batchSilences && !batchFillers && !batchPhrases && !batchMarkers
&& !batchSubtitles && !batchDynamicSubtitles)
if !batchStatus.isEmpty {
Text(batchStatus).font(.caption).foregroundStyle(.secondary)
}
HStack {
Button("Abrir pasta selecionada") {
if let outputFolder {
NSWorkspace.shared.open(URL(fileURLWithPath: outputFolder))
}
}
.disabled(outputFolder == nil)
Button("Abrir no Final Cut Pro") {
if let path = dynamicSubtitlesPath ?? processedPath ?? transcriptEditPath ?? transcriptMarkersPath {
NSWorkspace.shared.open(URL(fileURLWithPath: path))
}
}
.disabled(dynamicSubtitlesPath == nil && processedPath == nil
&& transcriptEditPath == nil && transcriptMarkersPath == nil)
Divider()
batchOptionRow(
toggle: Toggle("Gerar legendas dinâmicas (cascata, editáveis no FCP)", isOn: $batchDynamicSubtitles),
expanded: batchDynamicSubtitles
) {
Text("Usa o estilo configurado na aba \"Legendas Dinâmicas\". Roda por último, depois dos outros passos marcados acima, usando o timing já cortado da timeline.")
.font(.caption).foregroundStyle(.secondary)
}
}
.padding(.vertical, 8)
}
if results.isEmpty {
Text("Transcreva primeiro para habilitar os cortes por texto.")
.font(.caption2).foregroundStyle(.secondary)
VStack(alignment: .leading, spacing: 12) {
Button {
processBatch()
} label: {
if isBatchProcessing {
HStack {
ProgressView().controlSize(.small)
Text("Processando…")
}
.frame(maxWidth: .infinity)
} else {
Label("Processar selecionados", systemImage: "play.fill")
.frame(maxWidth: .infinity)
}
}
.buttonStyle(.borderedProminent)
.controlSize(.large)
.disabled(isBatchProcessing || isRunning || projectPath == nil || outputFolder == nil
|| !batchVoiceAnalysis && !batchVoiceEdit && !batchSilences && !batchFillers
&& !batchPhrases && !batchMarkers && !batchSubtitles && !batchDynamicSubtitles
// Ligada sem arquivo, a etapa falharia no meio da
// cadeia e interromperia tudo o que vem depois.
|| batchVoiceEdit && voiceActionsPath == nil)
if !batchStatus.isEmpty {
Text(batchStatus).font(.caption).foregroundStyle(.secondary)
}
HStack(spacing: 12) {
Button("Abrir pasta selecionada") {
if let outputFolder {
NSWorkspace.shared.open(URL(fileURLWithPath: outputFolder))
}
}
.disabled(outputFolder == nil)
Button("Abrir no Final Cut Pro") {
if let path = dynamicSubtitlesPath ?? processedPath ?? transcriptEditPath ?? transcriptMarkersPath {
NSWorkspace.shared.open(URL(fileURLWithPath: path))
}
}
.disabled(dynamicSubtitlesPath == nil && processedPath == nil
&& transcriptEditPath == nil && transcriptMarkersPath == nil)
Spacer()
}
}
.padding(.top, 4)
}
.disabled(outputFolder == nil)
.disabled(outputFolder == nil || results.isEmpty)
Section("Zoom (aproximar em um trecho)") {
if outputFolder != nil && results.isEmpty {
Label(
isRunning ? "Aguardando a transcrição terminar…" : "Transcreva o projeto acima para liberar o zoom.",
systemImage: "lock.fill"
)
.font(.caption).foregroundStyle(.orange)
}
if zoomClips.isEmpty {
Text("Carregando clipes do projeto…")
.font(.caption).foregroundStyle(.secondary)
@@ -347,12 +414,47 @@ struct TranscriptionView: View {
resultRow(zoomResultPath)
}
}
.disabled(outputFolder == nil)
.disabled(outputFolder == nil || results.isEmpty)
}
.formStyle(.grouped)
.task { await loadCatalog(); loadZoomClips() }
.onChange(of: projectPath) { _, _ in loadZoomClips() }
.onChange(of: outputFolder) { _, _ in loadZoomClips() }
.task { await loadCatalog(); loadProjectConfig(); loadZoomClips(); loadSilenceConfig() }
.onChange(of: projectPath) { _, newValue in
loadZoomClips()
saveProjectConfig(file: newValue ?? "")
}
.onChange(of: outputFolder) { _, newValue in
loadZoomClips()
saveProjectConfig(folder: newValue ?? "")
}
}
/// Reabre o último projeto salvo em ~/.fcp-mcp-server/config.json — pasta
/// de saída e arquivo — para a tela não começar vazia a cada abertura.
/// Só preenche o que ainda está vazio, então o projeto aberto pela aba
/// Projeto (embedded) nunca é sobrescrito pelo que ficou salvo. Caminhos
/// que sumiram do disco voltam vazios do Python e são ignorados aqui.
private func loadProjectConfig() {
PythonBridge.call(command: "project_config") { result, _ in
DispatchQueue.main.async {
guard let result, result["ok"] as? Bool == true else { return }
if outputFolder == nil, let folder = result["folder"] as? String, !folder.isEmpty {
outputFolder = folder
}
if projectPath == nil, let file = result["file"] as? String, !file.isEmpty {
projectPath = file
}
}
}
}
/// Grava a pasta e/ou o arquivo do projeto no config compartilhado. Campos
/// omitidos ficam como estão; string vazia limpa o campo.
private func saveProjectConfig(folder: String? = nil, file: String? = nil) {
var arguments: [String: Any] = [:]
if let folder { arguments["folder"] = folder }
if let file { arguments["file"] = file }
guard !arguments.isEmpty else { return }
PythonBridge.call(command: "set_project_config", arguments: arguments) { _, _ in }
}
/// Presents an NSOpenPanel configured to select a single file (not a
@@ -374,6 +476,45 @@ struct TranscriptionView: View {
}
}
/// Lê os limiares de silêncio salvos em ~/.fcp-mcp-server/config.json —
/// os mesmos que remove_media_silence usa por padrão — para os controles
/// abrirem já mostrando o que de fato vai rodar.
private func loadSilenceConfig() {
PythonBridge.call(command: "silence_config") { result, _ in
DispatchQueue.main.async {
guard let result, result["ok"] as? Bool == true else { return }
silencePadding = result["padding"] as? Double ?? silencePadding
silenceNoiseDb = result["noise_db"] as? Double ?? silenceNoiseDb
silenceMinDuration = result["min_silence"] as? Double ?? silenceMinDuration
}
}
}
private func saveSilenceConfig() {
PythonBridge.call(command: "set_silence_config", arguments: [
"padding": silencePadding,
"noise_db": silenceNoiseDb,
"min_silence": silenceMinDuration,
]) { _, _ in }
}
/// Escolhe o JSON de decisões (cortes/zooms/textos/marcadores) que
/// `apply_voice_actions` vai aplicar. Pré-seleciona a pasta do projeto,
/// que é onde o arquivo costuma ser gravado.
private func pickVoiceActions() {
let panel = NSOpenPanel()
panel.canChooseFiles = true
panel.canChooseDirectories = false
panel.allowsMultipleSelection = false
panel.allowedContentTypes = [.json]
panel.prompt = "Usar este arquivo"
panel.message = "Selecione o JSON com a lista de decisões da edição por voz."
if let outputFolder { panel.directoryURL = URL(fileURLWithPath: outputFolder) }
if panel.runModal() == .OK, let url = panel.url {
voiceActionsPath = url.path
}
}
private func pickOutputFolder() {
let panel = NSOpenPanel()
panel.canChooseFiles = false
@@ -438,6 +579,28 @@ struct TranscriptionView: View {
}
}
/// Renders a toggle plus its optional expanded detail block, indented and
/// visually tied together so batch options don't collapse into one wall
/// of controls with no breathing room.
@ViewBuilder
private func batchOptionRow<Toggle: View, Detail: View>(
toggle: Toggle, expanded: Bool, @ViewBuilder detail: () -> Detail
) -> some View {
VStack(alignment: .leading, spacing: 12) {
toggle
if expanded {
detail()
.padding(.leading, 20)
.padding(.vertical, 10)
.padding(.trailing, 8)
.background(
RoundedRectangle(cornerRadius: 8)
.fill(Color.secondary.opacity(0.06))
)
}
}
}
@ViewBuilder
private func resultRow(_ path: String) -> some View {
Text(path).font(.caption).lineLimit(1)
@@ -596,9 +759,11 @@ struct TranscriptionView: View {
guard let projectPath else { return }
isRemovingSilences = true
errorMessage = nil
// Sem repassar limiares: remove_silences lê os valores salvos em
// ~/.fcp-mcp-server/config.json (os mesmos que os controles acima
// gravam), então há uma só fonte da verdade.
PythonBridge.call(command: "remove_silences", arguments: [
"path": projectPath,
"padding": silencePadding,
]) { result, err in
DispatchQueue.main.async {
isRemovingSilences = false
@@ -614,6 +779,16 @@ struct TranscriptionView: View {
private func processBatch() {
guard let projectPath, let outputFolder else { return }
var operations: [String] = []
// Analysis first: it only writes a JSON sidecar next to the source
// media and never touches the project XML, so its position relative
// to the other steps doesn't change what they do — but running it
// before any cut keeps the mental model simple (measure, then edit).
if batchVoiceAnalysis { operations.append("analyze_voice") }
// The edit itself, and it has to run FIRST on the intact timeline:
// every zoom/text/marker it places is positioned by shifting from its
// own cut list, so a step that already rippled the timeline would put
// them on the wrong frame — with no visible error.
if batchVoiceEdit { operations.append("apply_voice_actions") }
if batchSilences { operations.append("remove_silences") }
if batchFillers { operations.append("remove_filler_words") }
if batchPhrases { operations.append("edit_by_transcript") }
@@ -638,19 +813,17 @@ struct TranscriptionView: View {
let operation = operations[index]
batchStatus = "Processando: \(operation)…"
var arguments: [String: Any] = ["path": currentPath, "output_dir": outputFolder]
if operation == "remove_silences" { arguments["padding"] = silencePadding }
if operation == "apply_voice_actions", let voiceActionsPath {
arguments["actions_path"] = voiceActionsPath
}
if operation == "edit_by_transcript" {
let phrases = phraseInput.split(separator: ",").map { $0.trimmingCharacters(in: .whitespaces) }.filter { !$0.isEmpty }
arguments["phrases"] = phrases
}
if operation == "generate_dynamic_subtitles" {
arguments["band_height"] = dynamicSubtitlesBandHeight
arguments["block_center_y"] = dynamicSubtitlesBlockCenterY
arguments["font"] = dynamicSubtitlesFont
arguments["font_size"] = Int(dynamicSubtitlesFontSize)
arguments["active_color"] = dynamicSubtitlesActiveColor.fcpxmlColorString
arguments["inactive_color"] = dynamicSubtitlesInactiveColor.fcpxmlColorString
}
// O estilo das legendas não é repassado aqui: generate_dynamic_subtitles
// lê ~/.fcp-mcp-server/config.json (o mesmo arquivo que a aba "Legendas
// Dinâmicas" grava) como seu próprio padrão, então há uma só fonte da
// verdade em vez de duas cópias podendo divergir.
PythonBridge.call(command: operation, arguments: arguments) { result, err in
DispatchQueue.main.async {
guard result?["ok"] as? Bool == true else {