chore: atualização geral

This commit is contained in:
João Henrique
2026-08-19 16:35:29 -04:00
parent 8fca456ceb
commit e7748c2c58
66 changed files with 13037 additions and 4237 deletions
+258 -3
View File
@@ -41,6 +41,34 @@ Commands:
clips, cuts, connected, markers}]}
or {"ok": false, "error": "..."}
analyze_voice {"path": "...", "output_dir": "...", "model": "...",
"language": "pt"|"auto"|null, "hf_token": "..."|null,
"num_speakers": ""|null}
Build the voice timeline (transcript+diarization+acoustics) for
every unique source media — analysis only, writes _voice_timeline.json
next to each media, `path` passes through unchanged. Meant as one
entry in the batch operations list (see processBatchStep), so
`refine_voice_timeline` never has to reopen the audio later.
-> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."}
dynamic_subtitle_config {}
-> {"ok": true, "band_height", "block_center_y", "line_gap", "font",
"font_size", "emphasis_font", "emphasis_face", "emphasis_size",
"active_color", "emphasis_color", "text_scale"}
set_dynamic_subtitle_config {<any of the fields above>}
Persists only the given fields to ~/.fcp-mcp-server/config.json.
generate_dynamic_subtitles reads this as its own fallback default.
-> {"ok": true, <same shape as dynamic_subtitle_config>}
silence_config {}
-> {"ok": true, "noise_db": -30.0, "min_silence": 0.5, "padding": 0.05}
set_silence_config {"noise_db": -30.0, "min_silence": 0.5, "padding": 0.05}
Persists only the given fields. detect_media_silence and
remove_media_silence read this as their own fallback default.
-> {"ok": true, <same shape as silence_config>}
transcribe {"path": "...", "model": "small", "language": "pt"|null,
"hf_token": "..."|null, "num_speakers": ""|null}
-> JSON-lines:
@@ -76,7 +104,9 @@ Commands:
generate_dynamic_subtitles {"path": "...", "clip_name": "..."|null,
"band_height": 0.22, "block_center_y": -167,
"font": "Helvetica Neue", "font_size": 128,
"active_color": "1 1 1 1", "inactive_color": "0.7 0.7 0.7 1",
"emphasis_font": "Playfair Display",
"emphasis_face": "Medium Italic", "emphasis_size": 265,
"active_color": "1 1 1 1", "emphasis_color": "1 1 1 1",
"model": "small", "language": "pt"|null}
-> {"ok": true, "path": "..._dynamic_subtitles.fcpxml", "message": "..."}
or {"ok": false, "error": "..."}
@@ -89,6 +119,16 @@ Commands:
-> {"ok": true, "diarization": bool, "diarization_message": "...",
"num_speakers": "..."}
voice_analysis
-> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85,
"emphasis_weights": {...}, "emotion_enabled": false,
"emotion_sensitivity": 0.5}
set_voice_analysis {"energy_threshold": 0.6, "emphasis_threshold": 0.9,
"emphasis_weights": {"energy": 0.4}|null,
"emotion_enabled": true, "emotion_sensitivity": 0.5}
-> same shape as voice_analysis (only given fields change)
Exit code 0 on success, 1 on error.
"""
@@ -122,16 +162,24 @@ from fcpxml.model_manager import ( # noqa: E402
is_model_downloaded,
list_installed_models,
load_catalog,
load_dynamic_subtitle_config,
load_hf_token,
load_num_speakers,
load_project_config,
load_selected_model,
load_silence_config,
load_transcript_language,
load_voice_analysis_config,
model_cache_dir,
save_dynamic_subtitle_config,
save_hf_token,
save_models_dir,
save_num_speakers,
save_project_config,
save_selected_model,
save_silence_config,
save_transcript_language,
save_voice_analysis_config,
)
from fcpxml.parser import parse_fcpxml # noqa: E402
from fcpxml.transcribe import transcribe # noqa: E402
@@ -600,14 +648,23 @@ def cmd_transcribe(args: dict) -> int:
total = len(media_paths)
results: list[dict] = []
for i, mp in enumerate(media_paths, 1):
_emit({"type": "progress", "fraction": i / total, "stage": f"Transcrevendo {Path(mp).name} ({i}/{total})…"})
stage = f"Transcrevendo {Path(mp).name} ({i}/{total})…"
_emit({"type": "progress", "fraction": (i - 1) / total, "stage": stage})
json_path = _transcript_json_path(mp, output_dir)
cached = _load_cached_transcript(json_path)
if cached is not None:
_emit({"type": "progress", "fraction": i / total, "stage": stage})
results.append(_result_row(mp, cached))
continue
data = transcribe(mp, model_size=model, language=language)
def _on_progress(file_fraction: float, _i: int = i, _stage: str = stage) -> None:
# Blend this file's own progress into the overall fraction so a
# single-media project doesn't jump straight to 100% before the
# actual (slow) decoding work has even started.
overall = (_i - 1 + file_fraction) / total
_emit({"type": "progress", "fraction": overall, "stage": _stage})
data = transcribe(mp, model_size=model, language=language, progress_cb=_on_progress)
if data is None:
_emit({"type": "error", "message": f"Não foi possível transcrever: {Path(mp).name}"})
return 1
@@ -638,6 +695,65 @@ def cmd_transcribe(args: dict) -> int:
return 0
def cmd_analyze_voice(args: dict) -> int:
"""Build the voice timeline (transcript+diarization+acoustics -> emphasis)
for every unique source media in the project, so `refine_voice_timeline`
and friends have something to read without ever reopening the audio.
Analysis only — writes _voice_timeline.json next to each media, doesn't
touch the project XML. `path` passes through unchanged so it composes
with the other batch steps (silence removal, captions) regardless of
where in the list it runs.
"""
path = str(args.get("path", ""))
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
model = str(args.get("model", "") or load_selected_model() or "")
language = args.get("language")
if language is None:
language = load_transcript_language()
if language == "auto":
language = None
token = str(args.get("hf_token") or load_hf_token() or "")
num_speakers = str(args.get("num_speakers") or load_num_speakers() or "")
try:
proj = parse_fcpxml(path)
except Exception as exc:
_emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
return 1
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
media_paths: list[str] = []
if tl is not None:
for clip in getattr(tl, "clips", []):
mp = media_src_to_path(clip.media_path or "")
if mp and Path(mp).is_file() and mp not in media_paths:
media_paths.append(mp)
if not media_paths:
_emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."})
return 1
from server import handle_build_voice_timeline
messages: list[str] = []
for mp in media_paths:
try:
contents = asyncio.run(handle_build_voice_timeline({
"media_path": mp, "model": model, "language": language,
"hf_token": token, "num_speakers": num_speakers,
"output_dir": args.get("output_dir"),
}))
except Exception as exc:
_emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"})
return 1
messages.append("\n".join(getattr(c, "text", str(c)) for c in contents))
_emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)})
return 0
def cmd_export_srt(args: dict) -> int:
"""Write a captions .srt synced to the edited timeline.
@@ -822,6 +938,135 @@ def cmd_set_diarization(args: dict) -> int:
return 0
def cmd_voice_analysis(args: dict) -> int:
"""Read the persisted voice-analysis settings (energy/emphasis/emotion)."""
_emit({"ok": True, **load_voice_analysis_config()})
return 0
def cmd_set_voice_analysis(args: dict) -> int:
"""Persist voice-analysis settings. Only the given fields change."""
weights = args.get("emphasis_weights")
config = save_voice_analysis_config(
energy_threshold=args.get("energy_threshold"),
emphasis_weights=weights if isinstance(weights, dict) else None,
emphasis_threshold=args.get("emphasis_threshold"),
emotion_enabled=args.get("emotion_enabled"),
emotion_sensitivity=args.get("emotion_sensitivity"),
)
_emit({"ok": True, **config})
return 0
def cmd_dynamic_subtitle_config(args: dict) -> int:
"""Read the persisted dynamic-subtitle style (font, size, color, layout)."""
_emit({"ok": True, **load_dynamic_subtitle_config()})
return 0
def cmd_set_dynamic_subtitle_config(args: dict) -> int:
"""Persist dynamic-subtitle style fields. Only the given fields change."""
config = save_dynamic_subtitle_config(**{
k: args.get(k) for k in (
"band_height", "block_center_y", "line_gap", "font", "font_size",
"emphasis_font", "emphasis_face", "emphasis_size",
"active_color", "emphasis_color", "text_scale",
)
})
_emit({"ok": True, **config})
return 0
def cmd_silence_config(args: dict) -> int:
"""Read the persisted silence thresholds (noise floor, duration, padding)."""
_emit({"ok": True, **load_silence_config()})
return 0
def cmd_set_silence_config(args: dict) -> int:
"""Persist silence thresholds. Only the given fields change."""
config = save_silence_config(
noise_db=args.get("noise_db"),
min_silence=args.get("min_silence"),
padding=args.get("padding"),
)
_emit({"ok": True, **config})
return 0
def cmd_apply_voice_actions(args: dict) -> int:
"""Apply a decision list (cuts/zooms/texts/markers) to the project XML.
The list is produced by a model reading the _voice_timeline.json — this
is the step that turns those decisions into an edit, and the one the
batch chain was missing: without it the app could measure the voice and
caption the result, but never cut by it.
`actions_path` points at the JSON; either a bare list or the
``{"actions": [...]}`` wrapper the skill emits is accepted. Times stay in
ORIGINAL source seconds — the handler resolves cuts first and shifts
everything else itself.
"""
path = str(args.get("path", ""))
if not path or not Path(path).exists():
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
return 1
actions = args.get("actions")
if actions is None:
actions_path = str(args.get("actions_path", ""))
if not actions_path or not Path(actions_path).exists():
_emit({"ok": False, "error": "Arquivo de decisões (JSON) não encontrado."})
return 1
try:
with open(actions_path, encoding="utf-8") as fh:
loaded = json.load(fh)
except (OSError, ValueError) as exc:
_emit({"ok": False, "error": f"Erro ao ler as decisões: {exc}"})
return 1
actions = loaded.get("actions") if isinstance(loaded, dict) else loaded
if not isinstance(actions, list) or not actions:
_emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."})
return 1
from server import handle_apply_voice_actions
try:
contents = asyncio.run(handle_apply_voice_actions({
"filepath": path,
"actions": actions,
"output_dir": args.get("output_dir"),
}))
except Exception as exc:
_emit({"ok": False, "error": f"Falha ao aplicar as decisões: {exc}"})
return 1
message = "\n".join(getattr(c, "text", str(c)) for c in contents)
# The handler reports dropped/rejected actions individually; hand the
# whole report back so the app can surface them instead of only the count.
out_path = path
for line in message.splitlines():
if line.startswith("- **Saved to**:"):
out_path = line.split("`")[1] if "`" in line else path
break
_emit({"ok": True, "path": out_path, "message": message})
return 0
def cmd_project_config(args: dict) -> int:
"""Read the last project folder/file the app was working on."""
_emit({"ok": True, **load_project_config()})
return 0
def cmd_set_project_config(args: dict) -> int:
"""Persist the last project folder/file. Only the given fields change."""
config = save_project_config(folder=args.get("folder"), file=args.get("file"))
_emit({"ok": True, **config})
return 0
def _result_row(mp: str, data: dict) -> dict:
words = data.get("words", [])
preview = (data.get("text", "") or "")[:160]
@@ -875,6 +1120,16 @@ def main() -> int:
"zoom_segments": cmd_zoom_segments,
"rename_speakers": cmd_rename_speakers,
"set_diarization": cmd_set_diarization,
"voice_analysis": cmd_voice_analysis,
"set_voice_analysis": cmd_set_voice_analysis,
"analyze_voice": cmd_analyze_voice,
"dynamic_subtitle_config": cmd_dynamic_subtitle_config,
"set_dynamic_subtitle_config": cmd_set_dynamic_subtitle_config,
"apply_voice_actions": cmd_apply_voice_actions,
"project_config": cmd_project_config,
"set_project_config": cmd_set_project_config,
"silence_config": cmd_silence_config,
"set_silence_config": cmd_set_silence_config,
}
handler = handlers.get(command)
if handler is None: