feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
711c397dfe
commit
7b5aed79ee
@@ -93,6 +93,25 @@ TOOLS = [
|
||||
"required": ["filepath"]
|
||||
}
|
||||
),
|
||||
Tool(
|
||||
name="generate_subtitles_by_emphasis",
|
||||
description="Generate BOTH subtitle styles over the FULL clip and let them coexist by visibility, not by splitting words: plain static titles (see generate_plain_subtitles) cover every word from start to end; dynamic progressive-composition titles (see generate_dynamic_subtitles) are additionally generated for whichever whole phrases were marked as emphasis in the phrase-review step (etapa 5, zoom applied, level >= 1). Wherever a dynamic phrase is on screen, the plain titles underneath it are set enabled=\"0\" (still present in the FCPXML, editable/re-enable-able in Final Cut, just not rendered) instead of never being generated there — so disabling emphasis later never leaves a silent gap in the plain track. Reads emphasis spans from the media's cached '<media>_phrase_actions.json' (written by save_phrase_review after the app's etapa 5 review) — run the voice-editing wizard through that step first, or nothing is treated as emphasis and every title stays plain and enabled. Style knobs are the saved 'Legendas Dinâmicas'/plain-subtitle configs (~/.fcp-mcp-server/config.json); this tool does not expose per-call style overrides, only the split logic — use generate_dynamic_subtitles/generate_plain_subtitles directly if you need one-off styling.",
|
||||
inputSchema={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"filepath": {"type": "string", "description": "Path to FCPXML file"},
|
||||
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
|
||||
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
|
||||
"language": {"type": "string", "description": "ISO language code hint (e.g. 'pt'); auto-detected if omitted"},
|
||||
"granularity": {"type": "string", "enum": ["phrase", "word"], "default": "phrase", "description": "Passed through to the dynamic half, same meaning as in generate_dynamic_subtitles"},
|
||||
"max_words": {"type": "integer", "description": "Max words per block for the plain half. Falls back to saved plain-subtitle config."},
|
||||
"uppercase": {"type": "boolean", "description": "Uppercase the plain half. Falls back to saved plain-subtitle config."},
|
||||
"keep_punctuation": {"type": "boolean", "description": "Keep punctuation in the plain half. Falls back to saved plain-subtitle config."},
|
||||
"output_path": {"type": "string", "description": "Output path (default: adds _emphasis_subtitles suffix)"},
|
||||
},
|
||||
"required": ["filepath"]
|
||||
}
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
@@ -140,6 +159,85 @@ def _plain_subtitle_blocks(words: Sequence[dict], max_words: int) -> list[list[d
|
||||
return blocks
|
||||
|
||||
|
||||
def _phrase_actions_path(media_path: str) -> Path:
|
||||
"""Where `save_phrase_review` writes emphasis decisions for this media.
|
||||
|
||||
Mirrors `phrase_review.review_paths()`'s naming (stem + "_phrase_actions.json"),
|
||||
without importing that module just for a path — the voice_timeline this would
|
||||
normally derive from is itself named `<media stem>_voice_timeline.json`, so
|
||||
stripping straight from the media stem lands on the same file.
|
||||
"""
|
||||
stem = Path(media_path).stem
|
||||
return Path(media_path).with_name(f"{stem}_phrase_actions.json")
|
||||
|
||||
|
||||
def _load_emphasis_spans(media_path: str) -> list[dict]:
|
||||
"""Load emphasis spans (source-media time) saved by the etapa-5 phrase review.
|
||||
|
||||
Returns [] if the review was never run for this media — callers should treat
|
||||
that as "nothing is emphasis yet", not as an error, since the wizard's later
|
||||
steps are optional.
|
||||
"""
|
||||
path = _phrase_actions_path(media_path)
|
||||
if not path.is_file():
|
||||
return []
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return []
|
||||
spans = data.get("emphasis_spans", [])
|
||||
return [s for s in spans if isinstance(s, dict) and "start" in s and "end" in s]
|
||||
|
||||
|
||||
def _word_in_spans(word_start: float, word_end: float, spans: Sequence[dict]) -> bool:
|
||||
"""A word belongs to an emphasis span if its midpoint falls inside it.
|
||||
|
||||
Midpoint, not start, so a word straddling a span boundary (which can happen
|
||||
since spans come from phrase trims, not word timestamps) lands on whichever
|
||||
side it mostly belongs to instead of always defaulting to one edge.
|
||||
"""
|
||||
mid = (word_start + word_end) / 2.0
|
||||
return any(float(s["start"]) <= mid < float(s["end"]) for s in spans)
|
||||
|
||||
|
||||
def _words_in_spans(words: Sequence[dict], spans: Sequence[dict]) -> list[dict]:
|
||||
"""The subset of source-time transcript words that fall inside a span.
|
||||
|
||||
Feeds only the DYNAMIC half — the plain half always gets every word, full
|
||||
clip, unfiltered; this is not a partition of the word list into two
|
||||
disjoint sets, it is "which words also get the dynamic treatment on top".
|
||||
"""
|
||||
if not spans:
|
||||
return []
|
||||
return [
|
||||
w for w in words
|
||||
if _word_in_spans(float(w.get("start", 0.0)), float(w.get("end", w.get("start", 0.0))), spans)
|
||||
]
|
||||
|
||||
|
||||
def _segments_in_spans(segments: Sequence[dict], spans: Sequence[dict]) -> list[dict]:
|
||||
"""Keep only the sentences that fall inside an emphasis span (by midpoint).
|
||||
|
||||
Feeds the dynamic half's sentence-block builder; segments outside every span
|
||||
would only produce blocks with no words left in them after the word filter.
|
||||
"""
|
||||
if not spans:
|
||||
return []
|
||||
kept = []
|
||||
for seg in segments:
|
||||
start = float(seg.get("start", 0.0))
|
||||
end = float(seg.get("end", start))
|
||||
mid = (start + end) / 2.0
|
||||
if any(float(s["start"]) <= mid < float(s["end"]) for s in spans):
|
||||
kept.append(seg)
|
||||
return kept
|
||||
|
||||
|
||||
def _overlaps_any_span(start: float, end: float, spans: Sequence[tuple[float, float]]) -> bool:
|
||||
"""Half-open interval overlap: a plain title under this window must hide."""
|
||||
return any(start < span_end and end > span_start for span_start, span_end in spans)
|
||||
|
||||
|
||||
async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Validate title/subtitle layout for spatial collisions and safe-area
|
||||
containment (collision.validate_titles over every <title> in the file)."""
|
||||
@@ -442,8 +540,214 @@ async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextConte
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[TextContent]:
|
||||
"""Generate plain titles for the whole clip and dynamic titles for the
|
||||
emphasis phrases on top, then hide (enabled="0") the plain titles that
|
||||
fall under a dynamic phrase — never split the word list between the two.
|
||||
|
||||
Plain always covers every word, so turning emphasis off later (editing
|
||||
the phrase review and re-running) never leaves a silent gap: the plain
|
||||
title was there all along, just disabled.
|
||||
"""
|
||||
model = arguments.get("model", "base")
|
||||
language = arguments.get("language")
|
||||
output_dir = arguments.get("output_dir")
|
||||
clip_filter = arguments.get("clip_name")
|
||||
granularity = arguments.get("granularity", "phrase")
|
||||
|
||||
saved_dynamic = load_dynamic_subtitle_config()
|
||||
body_color = saved_dynamic["active_color"]
|
||||
dynamic_config = DynamicSubtitleConfig(
|
||||
style=WordStyle(
|
||||
font=saved_dynamic["font"],
|
||||
font_size=int(saved_dynamic["font_size"]),
|
||||
active_color=body_color,
|
||||
inactive_color="0.7 0.7 0.7 1",
|
||||
emphasis_look=WordLook(
|
||||
int(saved_dynamic["emphasis_size"]),
|
||||
saved_dynamic["emphasis_color"] or body_color,
|
||||
font=saved_dynamic["emphasis_font"],
|
||||
face=saved_dynamic["emphasis_face"],
|
||||
kerning=0.0,
|
||||
),
|
||||
body_look=WordLook(
|
||||
int(saved_dynamic["font_size"]),
|
||||
body_color,
|
||||
font=saved_dynamic["font"],
|
||||
face="Bold",
|
||||
kerning=1.2,
|
||||
),
|
||||
),
|
||||
band_height=float(saved_dynamic["band_height"]),
|
||||
block_center_y=float(saved_dynamic["block_center_y"]),
|
||||
granularity=granularity,
|
||||
text_scale=float(saved_dynamic["text_scale"]),
|
||||
line_gap=float(saved_dynamic["line_gap"]),
|
||||
)
|
||||
|
||||
saved_plain = load_plain_subtitle_config()
|
||||
plain_font = saved_plain["font"]
|
||||
plain_font_size = int(saved_plain["font_size"])
|
||||
plain_font_color = saved_plain["font_color"]
|
||||
max_words = max(1, int(arguments.get("max_words", saved_plain["max_words"])))
|
||||
position_y = float(saved_plain["position_y"])
|
||||
uppercase = bool(arguments.get("uppercase", saved_plain["uppercase"]))
|
||||
keep_punctuation = bool(arguments.get("keep_punctuation", saved_plain["keep_punctuation"]))
|
||||
|
||||
filepath, output_path, modifier = _setup_modifier(arguments, "_emphasis_subtitles")
|
||||
|
||||
added: list[tuple[str, int, int, int, int]] = []
|
||||
skipped: list[tuple[str, str]] = []
|
||||
no_review: list[str] = []
|
||||
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
||||
for el in spine_clips:
|
||||
name = el.get("name", "")
|
||||
if clip_filter and name != clip_filter:
|
||||
continue
|
||||
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
||||
media_path = media_src_to_path(src)
|
||||
if not media_path or not Path(media_path).is_file():
|
||||
skipped.append((name, "media file missing"))
|
||||
continue
|
||||
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
||||
if data is None:
|
||||
skipped.append((name, reason))
|
||||
continue
|
||||
|
||||
spans = _load_emphasis_spans(media_path)
|
||||
if not spans:
|
||||
no_review.append(name)
|
||||
|
||||
clip_source_start = modifier.source_file_start(el).to_seconds()
|
||||
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
||||
window_end = clip_source_start + clip_duration
|
||||
|
||||
# Clip-relative windows, for deciding which plain titles to hide —
|
||||
# same coordinate space add_text_title's offsets end up in.
|
||||
clip_spans = [
|
||||
(max(0.0, float(s["start"]) - clip_source_start), min(clip_duration, float(s["end"]) - clip_source_start))
|
||||
for s in spans
|
||||
if float(s["end"]) > clip_source_start and float(s["start"]) < window_end
|
||||
]
|
||||
|
||||
all_words = data.get("words", [])
|
||||
|
||||
dynamic_lines = 0
|
||||
dynamic_word_count = 0
|
||||
emphasis_words = _words_in_spans(all_words, spans)
|
||||
clip_emphasis_words = _words_overlapping_clip(emphasis_words, clip_source_start, window_end)
|
||||
if clip_emphasis_words:
|
||||
all_segments = data.get("segments", [])
|
||||
emphasis_segments = _segments_in_spans(all_segments, spans)
|
||||
clip_segments = [
|
||||
{
|
||||
"start": float(s.get("start", 0.0)) - clip_source_start,
|
||||
"end": float(s.get("end", 0.0)) - clip_source_start,
|
||||
}
|
||||
for s in emphasis_segments
|
||||
if float(s.get("end", 0.0)) > clip_source_start
|
||||
and float(s.get("start", 0.0)) < window_end
|
||||
]
|
||||
# Pass the element itself, not `name` — see the same note in
|
||||
# handle_generate_dynamic_subtitles (Engine/docs/05_EXPERIENCIAS.md,
|
||||
# entry 2026-08-17).
|
||||
dynamic_lines = len(
|
||||
modifier.generate_dynamic_subtitles(
|
||||
el, clip_emphasis_words, dynamic_config, segments=clip_segments
|
||||
)
|
||||
)
|
||||
dynamic_word_count = len(clip_emphasis_words)
|
||||
|
||||
# Plain covers EVERY word in the clip — never filtered by emphasis.
|
||||
# Titles landing under a dynamic phrase are disabled below instead of
|
||||
# never being created, so turning emphasis off later never leaves a
|
||||
# silent gap where neither style is on screen.
|
||||
plain_created = 0
|
||||
plain_hidden = 0
|
||||
clip_all_words = _words_overlapping_clip(all_words, clip_source_start, window_end)
|
||||
blocks = _plain_subtitle_blocks(clip_all_words, max_words)
|
||||
for block in blocks:
|
||||
parts = [
|
||||
_plain_word_text(w.get("word", ""), uppercase=uppercase, keep_punctuation=keep_punctuation)
|
||||
for w in block
|
||||
]
|
||||
text = " ".join(p for p in parts if p).strip()
|
||||
if not text:
|
||||
continue
|
||||
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
|
||||
end = max(float(w.get("end", start)) for w in block)
|
||||
duration = max(end - start, modifier.frame_duration_fraction())
|
||||
title = modifier.add_text_title(
|
||||
el,
|
||||
text,
|
||||
offset=f"{start:.6f}s",
|
||||
duration=f"{duration:.6f}s",
|
||||
lane=20,
|
||||
position=f"0 {position_y:g}",
|
||||
font=plain_font,
|
||||
font_size=plain_font_size,
|
||||
font_color=plain_font_color,
|
||||
bold=True,
|
||||
face=None,
|
||||
font_scale=1.0,
|
||||
size_param=plain_font_size,
|
||||
)
|
||||
plain_created += 1
|
||||
if _overlaps_any_span(start, end, clip_spans):
|
||||
title.set("enabled", "0")
|
||||
plain_hidden += 1
|
||||
|
||||
if dynamic_lines or plain_created:
|
||||
added.append(
|
||||
(name, dynamic_lines, plain_created, plain_hidden, dynamic_word_count + len(clip_all_words))
|
||||
)
|
||||
else:
|
||||
skipped.append((name, "no words in clip's source range"))
|
||||
|
||||
if not added:
|
||||
text = "# Subtitles by Emphasis\n\nNo captions generated — file unchanged (nothing saved)."
|
||||
if skipped:
|
||||
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
||||
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
||||
)
|
||||
return _text_result(text)
|
||||
|
||||
modifier.save(output_path)
|
||||
total_dynamic = sum(d for _, d, _, _, _ in added)
|
||||
total_plain = sum(p for _, _, p, _, _ in added)
|
||||
total_hidden = sum(h for _, _, _, h, _ in added)
|
||||
total_words = sum(w for _, _, _, _, w in added)
|
||||
result = "# Subtitles by Emphasis Generated\n\n## Summary\n"
|
||||
result += (
|
||||
f"- **Clips Captioned**: {len(added)}\n"
|
||||
f"- **Dynamic Title Lines (emphasis)**: {total_dynamic}\n"
|
||||
f"- **Plain Title Blocks (full clip)**: {total_plain}\n"
|
||||
f"- **Plain Blocks Hidden Under Emphasis (enabled=\"0\")**: {total_hidden}\n"
|
||||
f"- **Total Words**: {total_words}\n\n"
|
||||
)
|
||||
result += _markdown_table(
|
||||
["Clip", "Dynamic Lines", "Plain Blocks", "Hidden", "Words"],
|
||||
[[n, str(d), str(p), str(h), str(w)] for n, d, p, h, w in added],
|
||||
)
|
||||
if no_review:
|
||||
result += (
|
||||
"\n## Sem revisão de ênfase\n"
|
||||
"Nenhum `_phrase_actions.json` encontrado para: "
|
||||
+ ", ".join(no_review)
|
||||
+ " — todas as frases desses clipes saíram como legenda comum. "
|
||||
"Rode a etapa 5 do Assistente (revisão de frases) antes, se quiser destaque dinâmico.\n"
|
||||
)
|
||||
if skipped:
|
||||
result += "\n## Skipped Clips\n" + _markdown_table(
|
||||
["Clip", "Reason"], [[n, r] for n, r in skipped]
|
||||
)
|
||||
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json; emphasis spans from _phrase_actions.json.*"
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
HANDLERS = {
|
||||
"validate_subtitle_layout": handle_validate_subtitle_layout,
|
||||
"generate_dynamic_subtitles": handle_generate_dynamic_subtitles,
|
||||
"generate_plain_subtitles": handle_generate_plain_subtitles,
|
||||
"generate_subtitles_by_emphasis": handle_generate_subtitles_by_emphasis,
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ from mcp.types import TextContent, Tool
|
||||
|
||||
from fcpxml.diarize import assign_speakers, build_speakers, diarization_capability, diarize
|
||||
from fcpxml.emphasis import EmphasisWeights
|
||||
from fcpxml.llm_local import DEFAULT_BASE_URL, generate_voice_actions
|
||||
from fcpxml.media_intel import media_src_to_path
|
||||
from fcpxml.model_manager import (
|
||||
load_hf_token,
|
||||
@@ -21,6 +22,7 @@ from fcpxml.model_manager import (
|
||||
save_voice_analysis_config,
|
||||
)
|
||||
from fcpxml.models import TimeValue
|
||||
from fcpxml.phrase_review import build_phrase_review, save_phrase_review
|
||||
from fcpxml.voice_actions import parse_actions, resolve_actions, speaker_cut_actions
|
||||
from fcpxml.voice_features import extract_energy, extract_pitch, features_capability
|
||||
from fcpxml.voice_timeline import (
|
||||
@@ -93,6 +95,7 @@ TOOLS = [
|
||||
"hf_token": {"type": "string", "description": "HuggingFace token for speaker diarization (default: the persisted token; omit to skip diarization)"},
|
||||
"num_speakers": {"type": "string", "description": "Known number of speakers, if any (default: the persisted setting, else auto-detect)"},
|
||||
"output_dir": {"type": "string", "description": "Folder to write _voice_timeline.json into (default: next to the media file)"},
|
||||
"rotation": {"type": "number", "description": "Degrees the clip is rotated by in the FCPXML (e.g. a Transform filter straightening a tilted phone shot). Recorded in the timeline JSON so a preview can apply the same correction. Default 0."},
|
||||
},
|
||||
"required": ["media_path"]
|
||||
}
|
||||
@@ -167,6 +170,27 @@ TOOLS = [
|
||||
"required": ["filepath", "actions"]
|
||||
}
|
||||
),
|
||||
Tool(
|
||||
name="generate_voice_script",
|
||||
description="Run the WHOLE voice-edit pass internally, no wizard, no copy-paste: reuse an existing voice timeline (or transcribe + build one) -> hand it to a LOCAL model (Ollama running Gemma 3 / Llama) that directs the edit -> return the readable script (roteiro) AND the action JSON, and optionally apply it to a FCPXML. The model reads the full _voice_timeline.json (the whole file goes with the brief) and emits cut/zoom/text/marker decisions per the editar-por-voz brief; decisions are validated row-by-row so one bad row never discards the edit. Times stay in ORIGINAL source seconds; the applier resolves cuts and shifts everything else. Writes _voice_timeline.json, _phrase_review.json, _phrase_actions.json and (when applying) a _voice_edit FCPXML. Defaults to the local model 'gemma3:12b' at http://localhost:11434 — change via model/base_url.",
|
||||
inputSchema={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"media_path": {"type": "string", "description": "Path to the audio/video file to analyze and direct (.wav, .mp3, .m4a, .aac, .aif, .flac, .mov, .mp4). Required when there is no voice_timeline yet; ignored when voice_timeline is provided."},
|
||||
"voice_timeline": {"type": "string", "description": "Path to an existing _voice_timeline.json (e.g. from the assistant's analysis step). When given, it is reused and transcription/acoustics are skipped — the model gets the whole file to direct the edit."},
|
||||
"filepath": {"type": "string", "description": "Optional FCPXML to apply the decisions to (non-destructive: writes a _voice_edit copy). When omitted, only the script and actions are produced."},
|
||||
"model": {"type": "string", "default": "gemma3:12b", "description": "Local model Ollama serves (e.g. 'gemma3:12b', 'gemma3:4b', 'llama3')"},
|
||||
"base_url": {"type": "string", "default": "http://localhost:11434", "description": "Ollama base URL"},
|
||||
"model_size": {"type": "string", "default": "base", "description": "Whisper model size to use if transcription is needed"},
|
||||
"language": {"type": "string", "description": "ISO language code hint for transcription, if needed"},
|
||||
"hf_token": {"type": "string", "description": "HuggingFace token for speaker diarization (omit to skip)"},
|
||||
"num_speakers": {"type": "string", "description": "Known number of speakers, if any"},
|
||||
"output_dir": {"type": "string", "description": "Folder to write the timeline/review/actions JSON into (default: next to the media file)"},
|
||||
"apply_to_fcpxml": {"type": "boolean", "default": True, "description": "When filepath is given, apply the decisions to it. Set false to only produce the script."},
|
||||
},
|
||||
"required": []
|
||||
}
|
||||
),
|
||||
Tool(
|
||||
name="get_voice_analysis_config",
|
||||
description="Read the persisted Voice Analysis settings: energy threshold, emphasis-index weights (energy/pitch_variation/rate_variation/pause_before/duration), emphasis cutoff for punch-in candidates, and emotion detection toggle/sensitivity. Shared with the MacApp settings screen (~/.fcp-mcp-server/config.json).",
|
||||
@@ -349,6 +373,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
|
||||
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
|
||||
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
|
||||
output_dir = arguments.get("output_dir")
|
||||
rotation = float(arguments.get("rotation") or 0.0)
|
||||
|
||||
transcript, reason = _load_or_transcribe(media_path, model, language, output_dir)
|
||||
if transcript is None:
|
||||
@@ -368,6 +393,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
|
||||
emphasis_floor=config["emphasis_floor"],
|
||||
emotion_enabled=config["emotion_enabled"],
|
||||
emotion_sensitivity=config["emotion_sensitivity"],
|
||||
rotation=rotation,
|
||||
)
|
||||
|
||||
json_path = Path(_validate_output_path(
|
||||
@@ -742,6 +768,145 @@ async def handle_save_voice_analysis_config(arguments: dict) -> Sequence[TextCon
|
||||
return _text_result(_voice_analysis_config_text(config))
|
||||
|
||||
|
||||
async def handle_generate_voice_script(arguments: dict) -> Sequence[TextContent]:
|
||||
"""The whole voice-edit pass, run inside the engine against a local model.
|
||||
|
||||
Transcribe (cached) -> build the voice timeline -> ask the local LLM to
|
||||
direct the edit -> build the readable script (roteiro) + the action JSON ->
|
||||
optionally apply to a FCPXML. No wizard, no copy-paste: the model's JSON is
|
||||
parsed and validated like any other decision source, and the applier turns
|
||||
it into FCPXML the same way it would for the rules engine.
|
||||
"""
|
||||
model = "gemma3:12b" if not arguments.get("model") else str(arguments["model"])
|
||||
base_url = str(arguments.get("base_url") or DEFAULT_BASE_URL)
|
||||
model_size = arguments.get("model_size", "base")
|
||||
language = arguments.get("language")
|
||||
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
|
||||
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
|
||||
output_dir = arguments.get("output_dir")
|
||||
fcpxml_path = arguments.get("filepath")
|
||||
apply = bool(arguments.get("apply_to_fcpxml", True)) and bool(fcpxml_path)
|
||||
|
||||
# Camino 1: já temos uma voice timeline (etapa de análise do assistente) —
|
||||
# reaproveita e pula a transcrição/análise acústica/diarização, que é caro.
|
||||
# Camino 2: só mídia — transcreve e monta a timeline do zero.
|
||||
vt_arg = arguments.get("voice_timeline")
|
||||
timeline = load_voice_timeline(Path(vt_arg)) if vt_arg and Path(vt_arg).is_file() else None
|
||||
media_path = arguments.get("media_path")
|
||||
if timeline is None:
|
||||
media_path = _validate_filepath(
|
||||
media_path, AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
|
||||
)
|
||||
transcript, reason = _load_or_transcribe(media_path, model_size, language, output_dir)
|
||||
if transcript is None:
|
||||
return _text_result(
|
||||
f"# Roteiro por IA Local\n\nNão foi possível obter a transcrição "
|
||||
f"({reason}).{_TRANSCRIBE_INSTALL_HINT}"
|
||||
)
|
||||
config = load_voice_analysis_config()
|
||||
timeline = build_voice_timeline(
|
||||
media_path,
|
||||
transcript,
|
||||
hf_token=token,
|
||||
num_speakers=num_speakers,
|
||||
weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
|
||||
peak_percentile=config["peak_percentile"],
|
||||
emphasis_floor=config["emphasis_floor"],
|
||||
emotion_enabled=config["emotion_enabled"],
|
||||
emotion_sensitivity=config["emotion_sensitivity"],
|
||||
)
|
||||
vt_arg = str(_validate_output_path(
|
||||
str(voice_timeline_path(media_path, output_dir)),
|
||||
anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent),
|
||||
))
|
||||
save_voice_timeline(timeline, Path(vt_arg))
|
||||
else:
|
||||
# A timeline veio pronta; a mídia só é necessária se for aplicar e o
|
||||
# caller não a passou — deriva do próprio campo `source` da timeline.
|
||||
if not media_path:
|
||||
candidate = Path(vt_arg).parent / timeline.get("source", "")
|
||||
media_path = str(candidate) if candidate.is_file() else None
|
||||
|
||||
decision = generate_voice_actions(timeline, model=model, base_url=base_url)
|
||||
actions = decision["actions"]
|
||||
errors = list(decision["errors"])
|
||||
if not actions and errors:
|
||||
# The model produced nothing usable (transport error or unparseable
|
||||
# response) — report it clearly instead of a silent "0 decisions".
|
||||
raise RuntimeError(
|
||||
"O modelo local não devolveu decisões utilizáveis: " + "; ".join(errors)
|
||||
)
|
||||
|
||||
review = build_phrase_review(
|
||||
timeline,
|
||||
[a.as_dict() for a in actions],
|
||||
voice_timeline_path=str(vt_arg),
|
||||
)
|
||||
review_path, actions_path = save_phrase_review(str(vt_arg), review)
|
||||
|
||||
roteiro = _roteiro_markdown(review, timeline.get("source", ""))
|
||||
roteiro_path = Path(vt_arg).with_name(Path(vt_arg).stem.replace("_voice_timeline", "") + "_roteiro.md")
|
||||
roteiro_path.write_text(roteiro, encoding="utf-8")
|
||||
|
||||
applied_text = ""
|
||||
if apply:
|
||||
contents = await handle_apply_voice_actions({
|
||||
"filepath": fcpxml_path,
|
||||
"actions": [a.as_dict() for a in actions],
|
||||
"output_dir": output_dir,
|
||||
})
|
||||
applied_text = "\n\n" + "\n".join(getattr(c, "text", str(c)) for c in contents)
|
||||
|
||||
result = f"""# Roteiro por IA Local ({model})
|
||||
|
||||
## Resumo
|
||||
- **Fonte**: {timeline.get('source', '')}
|
||||
- **Duração**: {format_duration(timeline['summary']['duration'])}
|
||||
- **Decisões do modelo**: {len(actions)} (cortes/zoom/texto/marcador)
|
||||
- **Linha do tempo**: {vt_arg}
|
||||
- **Roteiro (legível)**: {roteiro_path}
|
||||
- **Ações JSON**: {actions_path}
|
||||
- **Revisão de frases**: {review_path}
|
||||
"""
|
||||
if errors:
|
||||
result += "\n## Rejeitado / avisos\n" + "\n".join(f"- {e}" for e in errors) + "\n"
|
||||
result += "\n---\n\n" + roteiro
|
||||
result += applied_text
|
||||
result += "\n\n*Tudo rodou internamente: o modelo local leu a timeline e decidiu a edição; nenhum passo manual foi necessário.*"
|
||||
return _text_result(result)
|
||||
|
||||
|
||||
def _roteiro_markdown(review: dict, source: str) -> str:
|
||||
"""The readable script: kept lines (roteiro) then the cut/bastidor lines."""
|
||||
phrases = review.get("phrases", [])
|
||||
kept = [p for p in phrases if p.get("active")]
|
||||
cut = [p for p in phrases if not p.get("active")]
|
||||
|
||||
lines = [f"# Roteiro — {source}", ""]
|
||||
lines.append(f"**{len(kept)} falas mantidas · {len(cut)} cortadas**")
|
||||
lines.append("")
|
||||
lines.append("## Roteiro (mantido)")
|
||||
if not kept:
|
||||
lines.append("_Nenhuma fala mantida._")
|
||||
for p in kept:
|
||||
tag = ""
|
||||
if p.get("emphasis", 0) >= 1:
|
||||
tag = f" · zoom nível {p['emphasis']}"
|
||||
spk = f"[{p.get('speaker', '')}] " if p.get("speaker") else ""
|
||||
lines.append(f"- {spk}{p.get('text', '')}{tag}")
|
||||
if p.get("reason"):
|
||||
lines.append(f" - _decisão_: {p['reason']}")
|
||||
if cut:
|
||||
lines.append("")
|
||||
lines.append("## Cortado / bastidor")
|
||||
for p in cut:
|
||||
spk = f"[{p.get('speaker', '')}] " if p.get("speaker") else ""
|
||||
lines.append(f"- {spk}{p.get('text', '')}")
|
||||
if p.get("reason"):
|
||||
lines.append(f" - _por que cortou_: {p['reason']}")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
HANDLERS = {
|
||||
"diarize_media": handle_diarize_media,
|
||||
"analyze_voice_features": handle_analyze_voice_features,
|
||||
@@ -749,6 +914,7 @@ HANDLERS = {
|
||||
"remove_speakers": handle_remove_speakers,
|
||||
"refine_voice_timeline": handle_refine_voice_timeline,
|
||||
"apply_voice_actions": handle_apply_voice_actions,
|
||||
"generate_voice_script": handle_generate_voice_script,
|
||||
"get_voice_analysis_config": handle_get_voice_analysis_config,
|
||||
"save_voice_analysis_config": handle_save_voice_analysis_config,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user