chore: atualização geral
This commit is contained in:
+258
-3
@@ -41,6 +41,34 @@ Commands:
|
||||
clips, cuts, connected, markers}]}
|
||||
or {"ok": false, "error": "..."}
|
||||
|
||||
analyze_voice {"path": "...", "output_dir": "...", "model": "...",
|
||||
"language": "pt"|"auto"|null, "hf_token": "..."|null,
|
||||
"num_speakers": ""|null}
|
||||
Build the voice timeline (transcript+diarization+acoustics) for
|
||||
every unique source media — analysis only, writes _voice_timeline.json
|
||||
next to each media, `path` passes through unchanged. Meant as one
|
||||
entry in the batch operations list (see processBatchStep), so
|
||||
`refine_voice_timeline` never has to reopen the audio later.
|
||||
-> {"ok": true, "path": "...", "message": "..."} or {"ok": false, "error": "..."}
|
||||
|
||||
dynamic_subtitle_config {}
|
||||
-> {"ok": true, "band_height", "block_center_y", "line_gap", "font",
|
||||
"font_size", "emphasis_font", "emphasis_face", "emphasis_size",
|
||||
"active_color", "emphasis_color", "text_scale"}
|
||||
|
||||
set_dynamic_subtitle_config {<any of the fields above>}
|
||||
Persists only the given fields to ~/.fcp-mcp-server/config.json.
|
||||
generate_dynamic_subtitles reads this as its own fallback default.
|
||||
-> {"ok": true, <same shape as dynamic_subtitle_config>}
|
||||
|
||||
silence_config {}
|
||||
-> {"ok": true, "noise_db": -30.0, "min_silence": 0.5, "padding": 0.05}
|
||||
|
||||
set_silence_config {"noise_db": -30.0, "min_silence": 0.5, "padding": 0.05}
|
||||
Persists only the given fields. detect_media_silence and
|
||||
remove_media_silence read this as their own fallback default.
|
||||
-> {"ok": true, <same shape as silence_config>}
|
||||
|
||||
transcribe {"path": "...", "model": "small", "language": "pt"|null,
|
||||
"hf_token": "..."|null, "num_speakers": ""|null}
|
||||
-> JSON-lines:
|
||||
@@ -76,7 +104,9 @@ Commands:
|
||||
generate_dynamic_subtitles {"path": "...", "clip_name": "..."|null,
|
||||
"band_height": 0.22, "block_center_y": -167,
|
||||
"font": "Helvetica Neue", "font_size": 128,
|
||||
"active_color": "1 1 1 1", "inactive_color": "0.7 0.7 0.7 1",
|
||||
"emphasis_font": "Playfair Display",
|
||||
"emphasis_face": "Medium Italic", "emphasis_size": 265,
|
||||
"active_color": "1 1 1 1", "emphasis_color": "1 1 1 1",
|
||||
"model": "small", "language": "pt"|null}
|
||||
-> {"ok": true, "path": "..._dynamic_subtitles.fcpxml", "message": "..."}
|
||||
or {"ok": false, "error": "..."}
|
||||
@@ -89,6 +119,16 @@ Commands:
|
||||
-> {"ok": true, "diarization": bool, "diarization_message": "...",
|
||||
"num_speakers": "..."}
|
||||
|
||||
voice_analysis
|
||||
-> {"ok": true, "energy_threshold": 0.5, "emphasis_threshold": 0.85,
|
||||
"emphasis_weights": {...}, "emotion_enabled": false,
|
||||
"emotion_sensitivity": 0.5}
|
||||
|
||||
set_voice_analysis {"energy_threshold": 0.6, "emphasis_threshold": 0.9,
|
||||
"emphasis_weights": {"energy": 0.4}|null,
|
||||
"emotion_enabled": true, "emotion_sensitivity": 0.5}
|
||||
-> same shape as voice_analysis (only given fields change)
|
||||
|
||||
Exit code 0 on success, 1 on error.
|
||||
"""
|
||||
|
||||
@@ -122,16 +162,24 @@ from fcpxml.model_manager import ( # noqa: E402
|
||||
is_model_downloaded,
|
||||
list_installed_models,
|
||||
load_catalog,
|
||||
load_dynamic_subtitle_config,
|
||||
load_hf_token,
|
||||
load_num_speakers,
|
||||
load_project_config,
|
||||
load_selected_model,
|
||||
load_silence_config,
|
||||
load_transcript_language,
|
||||
load_voice_analysis_config,
|
||||
model_cache_dir,
|
||||
save_dynamic_subtitle_config,
|
||||
save_hf_token,
|
||||
save_models_dir,
|
||||
save_num_speakers,
|
||||
save_project_config,
|
||||
save_selected_model,
|
||||
save_silence_config,
|
||||
save_transcript_language,
|
||||
save_voice_analysis_config,
|
||||
)
|
||||
from fcpxml.parser import parse_fcpxml # noqa: E402
|
||||
from fcpxml.transcribe import transcribe # noqa: E402
|
||||
@@ -600,14 +648,23 @@ def cmd_transcribe(args: dict) -> int:
|
||||
total = len(media_paths)
|
||||
results: list[dict] = []
|
||||
for i, mp in enumerate(media_paths, 1):
|
||||
_emit({"type": "progress", "fraction": i / total, "stage": f"Transcrevendo {Path(mp).name} ({i}/{total})…"})
|
||||
stage = f"Transcrevendo {Path(mp).name} ({i}/{total})…"
|
||||
_emit({"type": "progress", "fraction": (i - 1) / total, "stage": stage})
|
||||
json_path = _transcript_json_path(mp, output_dir)
|
||||
cached = _load_cached_transcript(json_path)
|
||||
if cached is not None:
|
||||
_emit({"type": "progress", "fraction": i / total, "stage": stage})
|
||||
results.append(_result_row(mp, cached))
|
||||
continue
|
||||
|
||||
data = transcribe(mp, model_size=model, language=language)
|
||||
def _on_progress(file_fraction: float, _i: int = i, _stage: str = stage) -> None:
|
||||
# Blend this file's own progress into the overall fraction so a
|
||||
# single-media project doesn't jump straight to 100% before the
|
||||
# actual (slow) decoding work has even started.
|
||||
overall = (_i - 1 + file_fraction) / total
|
||||
_emit({"type": "progress", "fraction": overall, "stage": _stage})
|
||||
|
||||
data = transcribe(mp, model_size=model, language=language, progress_cb=_on_progress)
|
||||
if data is None:
|
||||
_emit({"type": "error", "message": f"Não foi possível transcrever: {Path(mp).name}"})
|
||||
return 1
|
||||
@@ -638,6 +695,65 @@ def cmd_transcribe(args: dict) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_analyze_voice(args: dict) -> int:
|
||||
"""Build the voice timeline (transcript+diarization+acoustics -> emphasis)
|
||||
for every unique source media in the project, so `refine_voice_timeline`
|
||||
and friends have something to read without ever reopening the audio.
|
||||
|
||||
Analysis only — writes _voice_timeline.json next to each media, doesn't
|
||||
touch the project XML. `path` passes through unchanged so it composes
|
||||
with the other batch steps (silence removal, captions) regardless of
|
||||
where in the list it runs.
|
||||
"""
|
||||
path = str(args.get("path", ""))
|
||||
if not path or not Path(path).exists():
|
||||
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
|
||||
return 1
|
||||
|
||||
model = str(args.get("model", "") or load_selected_model() or "")
|
||||
language = args.get("language")
|
||||
if language is None:
|
||||
language = load_transcript_language()
|
||||
if language == "auto":
|
||||
language = None
|
||||
token = str(args.get("hf_token") or load_hf_token() or "")
|
||||
num_speakers = str(args.get("num_speakers") or load_num_speakers() or "")
|
||||
|
||||
try:
|
||||
proj = parse_fcpxml(path)
|
||||
except Exception as exc:
|
||||
_emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"})
|
||||
return 1
|
||||
tl = proj.primary_timeline or (proj.timelines[0] if proj.timelines else None)
|
||||
media_paths: list[str] = []
|
||||
if tl is not None:
|
||||
for clip in getattr(tl, "clips", []):
|
||||
mp = media_src_to_path(clip.media_path or "")
|
||||
if mp and Path(mp).is_file() and mp not in media_paths:
|
||||
media_paths.append(mp)
|
||||
if not media_paths:
|
||||
_emit({"ok": False, "error": "Nenhum arquivo de mídia acessível encontrado."})
|
||||
return 1
|
||||
|
||||
from server import handle_build_voice_timeline
|
||||
|
||||
messages: list[str] = []
|
||||
for mp in media_paths:
|
||||
try:
|
||||
contents = asyncio.run(handle_build_voice_timeline({
|
||||
"media_path": mp, "model": model, "language": language,
|
||||
"hf_token": token, "num_speakers": num_speakers,
|
||||
"output_dir": args.get("output_dir"),
|
||||
}))
|
||||
except Exception as exc:
|
||||
_emit({"ok": False, "error": f"Falha analisando {Path(mp).name}: {exc}"})
|
||||
return 1
|
||||
messages.append("\n".join(getattr(c, "text", str(c)) for c in contents))
|
||||
|
||||
_emit({"ok": True, "path": path, "message": "\n\n---\n\n".join(messages)})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_export_srt(args: dict) -> int:
|
||||
"""Write a captions .srt synced to the edited timeline.
|
||||
|
||||
@@ -822,6 +938,135 @@ def cmd_set_diarization(args: dict) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_voice_analysis(args: dict) -> int:
|
||||
"""Read the persisted voice-analysis settings (energy/emphasis/emotion)."""
|
||||
_emit({"ok": True, **load_voice_analysis_config()})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_set_voice_analysis(args: dict) -> int:
|
||||
"""Persist voice-analysis settings. Only the given fields change."""
|
||||
weights = args.get("emphasis_weights")
|
||||
config = save_voice_analysis_config(
|
||||
energy_threshold=args.get("energy_threshold"),
|
||||
emphasis_weights=weights if isinstance(weights, dict) else None,
|
||||
emphasis_threshold=args.get("emphasis_threshold"),
|
||||
emotion_enabled=args.get("emotion_enabled"),
|
||||
emotion_sensitivity=args.get("emotion_sensitivity"),
|
||||
)
|
||||
_emit({"ok": True, **config})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_dynamic_subtitle_config(args: dict) -> int:
|
||||
"""Read the persisted dynamic-subtitle style (font, size, color, layout)."""
|
||||
_emit({"ok": True, **load_dynamic_subtitle_config()})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_set_dynamic_subtitle_config(args: dict) -> int:
|
||||
"""Persist dynamic-subtitle style fields. Only the given fields change."""
|
||||
config = save_dynamic_subtitle_config(**{
|
||||
k: args.get(k) for k in (
|
||||
"band_height", "block_center_y", "line_gap", "font", "font_size",
|
||||
"emphasis_font", "emphasis_face", "emphasis_size",
|
||||
"active_color", "emphasis_color", "text_scale",
|
||||
)
|
||||
})
|
||||
_emit({"ok": True, **config})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_silence_config(args: dict) -> int:
|
||||
"""Read the persisted silence thresholds (noise floor, duration, padding)."""
|
||||
_emit({"ok": True, **load_silence_config()})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_set_silence_config(args: dict) -> int:
|
||||
"""Persist silence thresholds. Only the given fields change."""
|
||||
config = save_silence_config(
|
||||
noise_db=args.get("noise_db"),
|
||||
min_silence=args.get("min_silence"),
|
||||
padding=args.get("padding"),
|
||||
)
|
||||
_emit({"ok": True, **config})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_apply_voice_actions(args: dict) -> int:
|
||||
"""Apply a decision list (cuts/zooms/texts/markers) to the project XML.
|
||||
|
||||
The list is produced by a model reading the _voice_timeline.json — this
|
||||
is the step that turns those decisions into an edit, and the one the
|
||||
batch chain was missing: without it the app could measure the voice and
|
||||
caption the result, but never cut by it.
|
||||
|
||||
`actions_path` points at the JSON; either a bare list or the
|
||||
``{"actions": [...]}`` wrapper the skill emits is accepted. Times stay in
|
||||
ORIGINAL source seconds — the handler resolves cuts first and shifts
|
||||
everything else itself.
|
||||
"""
|
||||
path = str(args.get("path", ""))
|
||||
if not path or not Path(path).exists():
|
||||
_emit({"ok": False, "error": "Arquivo de projeto não encontrado."})
|
||||
return 1
|
||||
|
||||
actions = args.get("actions")
|
||||
if actions is None:
|
||||
actions_path = str(args.get("actions_path", ""))
|
||||
if not actions_path or not Path(actions_path).exists():
|
||||
_emit({"ok": False, "error": "Arquivo de decisões (JSON) não encontrado."})
|
||||
return 1
|
||||
try:
|
||||
with open(actions_path, encoding="utf-8") as fh:
|
||||
loaded = json.load(fh)
|
||||
except (OSError, ValueError) as exc:
|
||||
_emit({"ok": False, "error": f"Erro ao ler as decisões: {exc}"})
|
||||
return 1
|
||||
actions = loaded.get("actions") if isinstance(loaded, dict) else loaded
|
||||
|
||||
if not isinstance(actions, list) or not actions:
|
||||
_emit({"ok": False, "error": "A lista de decisões está vazia ou malformada."})
|
||||
return 1
|
||||
|
||||
from server import handle_apply_voice_actions
|
||||
|
||||
try:
|
||||
contents = asyncio.run(handle_apply_voice_actions({
|
||||
"filepath": path,
|
||||
"actions": actions,
|
||||
"output_dir": args.get("output_dir"),
|
||||
}))
|
||||
except Exception as exc:
|
||||
_emit({"ok": False, "error": f"Falha ao aplicar as decisões: {exc}"})
|
||||
return 1
|
||||
|
||||
message = "\n".join(getattr(c, "text", str(c)) for c in contents)
|
||||
# The handler reports dropped/rejected actions individually; hand the
|
||||
# whole report back so the app can surface them instead of only the count.
|
||||
out_path = path
|
||||
for line in message.splitlines():
|
||||
if line.startswith("- **Saved to**:"):
|
||||
out_path = line.split("`")[1] if "`" in line else path
|
||||
break
|
||||
_emit({"ok": True, "path": out_path, "message": message})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_project_config(args: dict) -> int:
|
||||
"""Read the last project folder/file the app was working on."""
|
||||
_emit({"ok": True, **load_project_config()})
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_set_project_config(args: dict) -> int:
|
||||
"""Persist the last project folder/file. Only the given fields change."""
|
||||
config = save_project_config(folder=args.get("folder"), file=args.get("file"))
|
||||
_emit({"ok": True, **config})
|
||||
return 0
|
||||
|
||||
|
||||
def _result_row(mp: str, data: dict) -> dict:
|
||||
words = data.get("words", [])
|
||||
preview = (data.get("text", "") or "")[:160]
|
||||
@@ -875,6 +1120,16 @@ def main() -> int:
|
||||
"zoom_segments": cmd_zoom_segments,
|
||||
"rename_speakers": cmd_rename_speakers,
|
||||
"set_diarization": cmd_set_diarization,
|
||||
"voice_analysis": cmd_voice_analysis,
|
||||
"set_voice_analysis": cmd_set_voice_analysis,
|
||||
"analyze_voice": cmd_analyze_voice,
|
||||
"dynamic_subtitle_config": cmd_dynamic_subtitle_config,
|
||||
"set_dynamic_subtitle_config": cmd_set_dynamic_subtitle_config,
|
||||
"apply_voice_actions": cmd_apply_voice_actions,
|
||||
"project_config": cmd_project_config,
|
||||
"set_project_config": cmd_set_project_config,
|
||||
"silence_config": cmd_silence_config,
|
||||
"set_silence_config": cmd_set_silence_config,
|
||||
}
|
||||
handler = handlers.get(command)
|
||||
if handler is None:
|
||||
|
||||
Reference in New Issue
Block a user