feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
711c397dfe
commit
7b5aed79ee
@@ -13,6 +13,7 @@ from mcp.types import TextContent, Tool
|
||||
|
||||
from fcpxml.diarize import assign_speakers, build_speakers, diarization_capability, diarize
|
||||
from fcpxml.emphasis import EmphasisWeights
|
||||
from fcpxml.llm_local import DEFAULT_BASE_URL, generate_voice_actions
|
||||
from fcpxml.media_intel import media_src_to_path
|
||||
from fcpxml.model_manager import (
|
||||
load_hf_token,
|
||||
@@ -21,6 +22,7 @@ from fcpxml.model_manager import (
|
||||
save_voice_analysis_config,
|
||||
)
|
||||
from fcpxml.models import TimeValue
|
||||
from fcpxml.phrase_review import build_phrase_review, save_phrase_review
|
||||
from fcpxml.voice_actions import parse_actions, resolve_actions, speaker_cut_actions
|
||||
from fcpxml.voice_features import extract_energy, extract_pitch, features_capability
|
||||
from fcpxml.voice_timeline import (
|
||||
@@ -93,6 +95,7 @@ TOOLS = [
|
||||
"hf_token": {"type": "string", "description": "HuggingFace token for speaker diarization (default: the persisted token; omit to skip diarization)"},
|
||||
"num_speakers": {"type": "string", "description": "Known number of speakers, if any (default: the persisted setting, else auto-detect)"},
|
||||
"output_dir": {"type": "string", "description": "Folder to write _voice_timeline.json into (default: next to the media file)"},
|
||||
"rotation": {"type": "number", "description": "Degrees the clip is rotated by in the FCPXML (e.g. a Transform filter straightening a tilted phone shot). Recorded in the timeline JSON so a preview can apply the same correction. Default 0."},
|
||||
},
|
||||
"required": ["media_path"]
|
||||
}
|
||||
@@ -167,6 +170,27 @@ TOOLS = [
|
||||
"required": ["filepath", "actions"]
|
||||
}
|
||||
),
|
||||
Tool(
|
||||
name="generate_voice_script",
|
||||
description="Run the WHOLE voice-edit pass internally, no wizard, no copy-paste: reuse an existing voice timeline (or transcribe + build one) -> hand it to a LOCAL model (Ollama running Gemma 3 / Llama) that directs the edit -> return the readable script (roteiro) AND the action JSON, and optionally apply it to a FCPXML. The model reads the full _voice_timeline.json (the whole file goes with the brief) and emits cut/zoom/text/marker decisions per the editar-por-voz brief; decisions are validated row-by-row so one bad row never discards the edit. Times stay in ORIGINAL source seconds; the applier resolves cuts and shifts everything else. Writes _voice_timeline.json, _phrase_review.json, _phrase_actions.json and (when applying) a _voice_edit FCPXML. Defaults to the local model 'gemma3:12b' at http://localhost:11434 — change via model/base_url.",
|
||||
inputSchema={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"media_path": {"type": "string", "description": "Path to the audio/video file to analyze and direct (.wav, .mp3, .m4a, .aac, .aif, .flac, .mov, .mp4). Required when there is no voice_timeline yet; ignored when voice_timeline is provided."},
|
||||
"voice_timeline": {"type": "string", "description": "Path to an existing _voice_timeline.json (e.g. from the assistant's analysis step). When given, it is reused and transcription/acoustics are skipped — the model gets the whole file to direct the edit."},
|
||||
"filepath": {"type": "string", "description": "Optional FCPXML to apply the decisions to (non-destructive: writes a _voice_edit copy). When omitted, only the script and actions are produced."},
|
||||
"model": {"type": "string", "default": "gemma3:12b", "description": "Local model Ollama serves (e.g. 'gemma3:12b', 'gemma3:4b', 'llama3')"},
|
||||
"base_url": {"type": "string", "default": "http://localhost:11434", "description": "Ollama base URL"},
|
||||
"model_size": {"type": "string", "default": "base", "description": "Whisper model size to use if transcription is needed"},
|
||||
"language": {"type": "string", "description": "ISO language code hint for transcription, if needed"},
|
||||
"hf_token": {"type": "string", "description": "HuggingFace token for speaker diarization (omit to skip)"},
|
||||
"num_speakers": {"type": "string", "description": "Known number of speakers, if any"},
|
||||
"output_dir": {"type": "string", "description": "Folder to write the timeline/review/actions JSON into (default: next to the media file)"},
|
||||
"apply_to_fcpxml": {"type": "boolean", "default": True, "description": "When filepath is given, apply the decisions to it. Set false to only produce the script."},
|
||||
},
|
||||
"required": []
|
||||
}
|
||||
),
|
||||
Tool(
|
||||
name="get_voice_analysis_config",
|
||||
description="Read the persisted Voice Analysis settings: energy threshold, emphasis-index weights (energy/pitch_variation/rate_variation/pause_before/duration), emphasis cutoff for punch-in candidates, and emotion detection toggle/sensitivity. Shared with the MacApp settings screen (~/.fcp-mcp-server/config.json).",
|
||||
@@ -349,6 +373,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
|
||||
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
|
||||
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
|
||||
output_dir = arguments.get("output_dir")
|
||||
rotation = float(arguments.get("rotation") or 0.0)
|
||||
|
||||
transcript, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
||||
if transcript is None:
|
||||
@@ -368,6 +393,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
|
||||
emphasis_floor=config["emphasis_floor"],
|
||||
emotion_enabled=config["emotion_enabled"],
|
||||
emotion_sensitivity=config["emotion_sensitivity"],
|
||||
rotation=rotation,
|
||||
)
|
||||
|
||||
json_path = Path(_validate_output_path(
|
||||
@@ -742,6 +768,145 @@ async def handle_save_voice_analysis_config(arguments: dict) -> Sequence[TextCon
|
||||
return _text_result(_voice_analysis_config_text(config))
|
||||
|
||||
|
||||
async def handle_generate_voice_script(arguments: dict) -> Sequence[TextContent]:
|
||||
"""The whole voice-edit pass, run inside the engine against a local model.
|
||||
|
||||
Transcribe (cached) -> build the voice timeline -> ask the local LLM to
|
||||
direct the edit -> build the readable script (roteiro) + the action JSON ->
|
||||
optionally apply to a FCPXML. No wizard, no copy-paste: the model's JSON is
|
||||
parsed and validated like any other decision source, and the applier turns
|
||||
it into FCPXML the same way it would for the rules engine.
|
||||
"""
|
||||
model = "gemma3:12b" if not arguments.get("model") else str(arguments["model"])
|
||||
base_url = str(arguments.get("base_url") or DEFAULT_BASE_URL)
|
||||
model_size = arguments.get("model_size", "base")
|
||||
language = arguments.get("language")
|
||||
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
|
||||
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
|
||||
output_dir = arguments.get("output_dir")
|
||||
fcpxml_path = arguments.get("filepath")
|
||||
apply = bool(arguments.get("apply_to_fcpxml", True)) and bool(fcpxml_path)
|
||||
|
||||
# Camino 1: já temos uma voice timeline (etapa de análise do assistente) —
|
||||
# reaproveita e pula a transcrição/análise acústica/diarização, que é caro.
|
||||
# Camino 2: só mídia — transcreve e monta a timeline do zero.
|
||||
vt_arg = arguments.get("voice_timeline")
|
||||
timeline = load_voice_timeline(Path(vt_arg)) if vt_arg and Path(vt_arg).is_file() else None
|
||||
media_path = arguments.get("media_path")
|
||||
if timeline is None:
|
||||
media_path = _validate_filepath(
|
||||
media_path, AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
|
||||
)
|
||||
transcript, reason = _load_or_transcribe(media_path, model_size, language, output_dir)
|
||||
if transcript is None:
|
||||
return _text_result(
|
||||
f"# Roteiro por IA Local\n\nNão foi possível obter a transcrição "
|
||||
f"({reason}).{_TRANSCRIBE_INSTALL_HINT}"
|
||||
)
|
||||
config = load_voice_analysis_config()
|
||||
timeline = build_voice_timeline(
|
||||
media_path,
|
||||
transcript,
|
||||
hf_token=token,
|
||||
num_speakers=num_speakers,
|
||||
weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
|
||||
peak_percentile=config["peak_percentile"],
|
||||
emphasis_floor=config["emphasis_floor"],
|
||||
emotion_enabled=config["emotion_enabled"],
|
||||
emotion_sensitivity=config["emotion_sensitivity"],
|
||||
)
|
||||
vt_arg = str(_validate_output_path(
|
||||
str(voice_timeline_path(media_path, output_dir)),
|
||||
anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent),
|
||||
))
|
||||
save_voice_timeline(timeline, Path(vt_arg))
|
||||
else:
|
||||
# A timeline veio pronta; a mídia só é necessária se for aplicar e o
|
||||
# caller não a passou — deriva do próprio campo `source` da timeline.
|
||||
if not media_path:
|
||||
candidate = Path(vt_arg).parent / timeline.get("source", "")
|
||||
media_path = str(candidate) if candidate.is_file() else None
|
||||
|
||||
decision = generate_voice_actions(timeline, model=model, base_url=base_url)
|
||||
actions = decision["actions"]
|
||||
errors = list(decision["errors"])
|
||||
if not actions and errors:
|
||||
# The model produced nothing usable (transport error or unparseable
|
||||
# response) — report it clearly instead of a silent "0 decisions".
|
||||
raise RuntimeError(
|
||||
"O modelo local não devolveu decisões utilizáveis: " + "; ".join(errors)
|
||||
)
|
||||
|
||||
review = build_phrase_review(
|
||||
timeline,
|
||||
[a.as_dict() for a in actions],
|
||||
voice_timeline_path=str(vt_arg),
|
||||
)
|
||||
review_path, actions_path = save_phrase_review(str(vt_arg), review)
|
||||
|
||||
roteiro = _roteiro_markdown(review, timeline.get("source", ""))
|
||||
roteiro_path = Path(vt_arg).with_name(Path(vt_arg).stem.replace("_voice_timeline", "") + "_roteiro.md")
|
||||
roteiro_path.write_text(roteiro, encoding="utf-8")
|
||||
|
||||
applied_text = ""
|
||||
if apply:
|
||||
contents = await handle_apply_voice_actions({
|
||||
"filepath": fcpxml_path,
|
||||
"actions": [a.as_dict() for a in actions],
|
||||
"output_dir": output_dir,
|
||||
})
|
||||
applied_text = "\n\n" + "\n".join(getattr(c, "text", str(c)) for c in contents)
|
||||
|
||||
result = f"""# Roteiro por IA Local ({model})
|
||||
|
||||
## Resumo
|
||||
- **Fonte**: {timeline.get('source', '')}
|
||||
- **Duração**: {format_duration(timeline['summary']['duration'])}
|
||||
- **Decisões do modelo**: {len(actions)} (cortes/zoom/texto/marcador)
|
||||
- **Linha do tempo**: {vt_arg}
|
||||
- **Roteiro (legível)**: {roteiro_path}
|
||||
- **Ações JSON**: {actions_path}
|
||||
- **Revisão de frases**: {review_path}
|
||||
"""
|
||||
if errors:
|
||||
result += "\n## Rejeitado / avisos\n" + "\n".join(f"- {e}" for e in errors) + "\n"
|
||||
result += "\n---\n\n" + roteiro
|
||||
result += applied_text
|
||||
result += "\n\n*Tudo rodou internamente: o modelo local leu a timeline e decidiu a edição; nenhum passo manual foi necessário.*"
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
def _roteiro_markdown(review: dict, source: str) -> str:
|
||||
"""The readable script: kept lines (roteiro) then the cut/bastidor lines."""
|
||||
phrases = review.get("phrases", [])
|
||||
kept = [p for p in phrases if p.get("active")]
|
||||
cut = [p for p in phrases if not p.get("active")]
|
||||
|
||||
lines = [f"# Roteiro — {source}", ""]
|
||||
lines.append(f"**{len(kept)} falas mantidas · {len(cut)} cortadas**")
|
||||
lines.append("")
|
||||
lines.append("## Roteiro (mantido)")
|
||||
if not kept:
|
||||
lines.append("_Nenhuma fala mantida._")
|
||||
for p in kept:
|
||||
tag = ""
|
||||
if p.get("emphasis", 0) >= 1:
|
||||
tag = f" · zoom nível {p['emphasis']}"
|
||||
spk = f"[{p.get('speaker', '')}] " if p.get("speaker") else ""
|
||||
lines.append(f"- {spk}{p.get('text', '')}{tag}")
|
||||
if p.get("reason"):
|
||||
lines.append(f" - _decisão_: {p['reason']}")
|
||||
if cut:
|
||||
lines.append("")
|
||||
lines.append("## Cortado / bastidor")
|
||||
for p in cut:
|
||||
spk = f"[{p.get('speaker', '')}] " if p.get("speaker") else ""
|
||||
lines.append(f"- {spk}{p.get('text', '')}")
|
||||
if p.get("reason"):
|
||||
lines.append(f" - _por que cortou_: {p['reason']}")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
HANDLERS = {
|
||||
"diarize_media": handle_diarize_media,
|
||||
"analyze_voice_features": handle_analyze_voice_features,
|
||||
@@ -749,6 +914,7 @@ HANDLERS = {
|
||||
"remove_speakers": handle_remove_speakers,
|
||||
"refine_voice_timeline": handle_refine_voice_timeline,
|
||||
"apply_voice_actions": handle_apply_voice_actions,
|
||||
"generate_voice_script": handle_generate_voice_script,
|
||||
"get_voice_analysis_config": handle_get_voice_analysis_config,
|
||||
"save_voice_analysis_config": handle_save_voice_analysis_config,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user