feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão

Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-21 18:26:04 -04:00
co-authored by Claude Sonnet 5
parent 711c397dfe
commit 7b5aed79ee
36 changed files with 2922 additions and 624 deletions
+304
View File
@@ -93,6 +93,25 @@ TOOLS = [
"required": ["filepath"]
}
),
Tool(
name="generate_subtitles_by_emphasis",
description="Generate BOTH subtitle styles over the FULL clip and let them coexist by visibility, not by splitting words: plain static titles (see generate_plain_subtitles) cover every word from start to end; dynamic progressive-composition titles (see generate_dynamic_subtitles) are additionally generated for whichever whole phrases were marked as emphasis in the phrase-review step (etapa 5, zoom applied, level >= 1). Wherever a dynamic phrase is on screen, the plain titles underneath it are set enabled=\"0\" (still present in the FCPXML, editable/re-enable-able in Final Cut, just not rendered) instead of never being generated there — so disabling emphasis later never leaves a silent gap in the plain track. Reads emphasis spans from the media's cached '<media>_phrase_actions.json' (written by save_phrase_review after the app's etapa 5 review) — run the voice-editing wizard through that step first, or nothing is treated as emphasis and every title stays plain and enabled. Style knobs are the saved 'Legendas Dinâmicas'/plain-subtitle configs (~/.fcp-mcp-server/config.json); this tool does not expose per-call style overrides, only the split logic — use generate_dynamic_subtitles/generate_plain_subtitles directly if you need one-off styling.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"clip_name": {"type": "string", "description": "Only caption the clip with this name (default: all spine clips with matched source media)"},
"model": {"type": "string", "default": "base", "description": "Whisper model size if transcription is needed"},
"language": {"type": "string", "description": "ISO language code hint (e.g. 'pt'); auto-detected if omitted"},
"granularity": {"type": "string", "enum": ["phrase", "word"], "default": "phrase", "description": "Passed through to the dynamic half, same meaning as in generate_dynamic_subtitles"},
"max_words": {"type": "integer", "description": "Max words per block for the plain half. Falls back to saved plain-subtitle config."},
"uppercase": {"type": "boolean", "description": "Uppercase the plain half. Falls back to saved plain-subtitle config."},
"keep_punctuation": {"type": "boolean", "description": "Keep punctuation in the plain half. Falls back to saved plain-subtitle config."},
"output_path": {"type": "string", "description": "Output path (default: adds _emphasis_subtitles suffix)"},
},
"required": ["filepath"]
}
),
]
@@ -140,6 +159,85 @@ def _plain_subtitle_blocks(words: Sequence[dict], max_words: int) -> list[list[d
return blocks
def _phrase_actions_path(media_path: str) -> Path:
"""Where `save_phrase_review` writes emphasis decisions for this media.
Mirrors `phrase_review.review_paths()`'s naming (stem + "_phrase_actions.json"),
without importing that module just for a path — the voice_timeline this would
normally derive from is itself named `<media stem>_voice_timeline.json`, so
stripping straight from the media stem lands on the same file.
"""
stem = Path(media_path).stem
return Path(media_path).with_name(f"{stem}_phrase_actions.json")
def _load_emphasis_spans(media_path: str) -> list[dict]:
"""Load emphasis spans (source-media time) saved by the etapa-5 phrase review.
Returns [] if the review was never run for this media — callers should treat
that as "nothing is emphasis yet", not as an error, since the wizard's later
steps are optional.
"""
path = _phrase_actions_path(media_path)
if not path.is_file():
return []
try:
data = json.loads(path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
return []
spans = data.get("emphasis_spans", [])
return [s for s in spans if isinstance(s, dict) and "start" in s and "end" in s]
def _word_in_spans(word_start: float, word_end: float, spans: Sequence[dict]) -> bool:
"""A word belongs to an emphasis span if its midpoint falls inside it.
Midpoint, not start, so a word straddling a span boundary (which can happen
since spans come from phrase trims, not word timestamps) lands on whichever
side it mostly belongs to instead of always defaulting to one edge.
"""
mid = (word_start + word_end) / 2.0
return any(float(s["start"]) <= mid < float(s["end"]) for s in spans)
def _words_in_spans(words: Sequence[dict], spans: Sequence[dict]) -> list[dict]:
"""The subset of source-time transcript words that fall inside a span.
Feeds only the DYNAMIC half — the plain half always gets every word, full
clip, unfiltered; this is not a partition of the word list into two
disjoint sets, it is "which words also get the dynamic treatment on top".
"""
if not spans:
return []
return [
w for w in words
if _word_in_spans(float(w.get("start", 0.0)), float(w.get("end", w.get("start", 0.0))), spans)
]
def _segments_in_spans(segments: Sequence[dict], spans: Sequence[dict]) -> list[dict]:
"""Keep only the sentences that fall inside an emphasis span (by midpoint).
Feeds the dynamic half's sentence-block builder; segments outside every span
would only produce blocks with no words left in them after the word filter.
"""
if not spans:
return []
kept = []
for seg in segments:
start = float(seg.get("start", 0.0))
end = float(seg.get("end", start))
mid = (start + end) / 2.0
if any(float(s["start"]) <= mid < float(s["end"]) for s in spans):
kept.append(seg)
return kept
def _overlaps_any_span(start: float, end: float, spans: Sequence[tuple[float, float]]) -> bool:
"""Half-open interval overlap: a plain title under this window must hide."""
return any(start < span_end and end > span_start for span_start, span_end in spans)
async def handle_validate_subtitle_layout(arguments: dict) -> Sequence[TextContent]:
"""Validate title/subtitle layout for spatial collisions and safe-area
containment (collision.validate_titles over every <title> in the file)."""
@@ -442,8 +540,214 @@ async def handle_generate_plain_subtitles(arguments: dict) -> Sequence[TextConte
return _text_result(result)
async def handle_generate_subtitles_by_emphasis(arguments: dict) -> Sequence[TextContent]:
"""Generate plain titles for the whole clip and dynamic titles for the
emphasis phrases on top, then hide (enabled="0") the plain titles that
fall under a dynamic phrase — never split the word list between the two.
Plain always covers every word, so turning emphasis off later (editing
the phrase review and re-running) never leaves a silent gap: the plain
title was there all along, just disabled.
"""
model = arguments.get("model", "base")
language = arguments.get("language")
output_dir = arguments.get("output_dir")
clip_filter = arguments.get("clip_name")
granularity = arguments.get("granularity", "phrase")
saved_dynamic = load_dynamic_subtitle_config()
body_color = saved_dynamic["active_color"]
dynamic_config = DynamicSubtitleConfig(
style=WordStyle(
font=saved_dynamic["font"],
font_size=int(saved_dynamic["font_size"]),
active_color=body_color,
inactive_color="0.7 0.7 0.7 1",
emphasis_look=WordLook(
int(saved_dynamic["emphasis_size"]),
saved_dynamic["emphasis_color"] or body_color,
font=saved_dynamic["emphasis_font"],
face=saved_dynamic["emphasis_face"],
kerning=0.0,
),
body_look=WordLook(
int(saved_dynamic["font_size"]),
body_color,
font=saved_dynamic["font"],
face="Bold",
kerning=1.2,
),
),
band_height=float(saved_dynamic["band_height"]),
block_center_y=float(saved_dynamic["block_center_y"]),
granularity=granularity,
text_scale=float(saved_dynamic["text_scale"]),
line_gap=float(saved_dynamic["line_gap"]),
)
saved_plain = load_plain_subtitle_config()
plain_font = saved_plain["font"]
plain_font_size = int(saved_plain["font_size"])
plain_font_color = saved_plain["font_color"]
max_words = max(1, int(arguments.get("max_words", saved_plain["max_words"])))
position_y = float(saved_plain["position_y"])
uppercase = bool(arguments.get("uppercase", saved_plain["uppercase"]))
keep_punctuation = bool(arguments.get("keep_punctuation", saved_plain["keep_punctuation"]))
filepath, output_path, modifier = _setup_modifier(arguments, "_emphasis_subtitles")
added: list[tuple[str, int, int, int, int]] = []
skipped: list[tuple[str, str]] = []
no_review: list[str] = []
spine_clips = [el for _, el in modifier._iter_spine_clips()]
for el in spine_clips:
name = el.get("name", "")
if clip_filter and name != clip_filter:
continue
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
media_path = media_src_to_path(src)
if not media_path or not Path(media_path).is_file():
skipped.append((name, "media file missing"))
continue
data, reason = _load_or_transcribe(media_path, model, language, output_dir)
if data is None:
skipped.append((name, reason))
continue
spans = _load_emphasis_spans(media_path)
if not spans:
no_review.append(name)
clip_source_start = modifier.source_file_start(el).to_seconds()
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
window_end = clip_source_start + clip_duration
# Clip-relative windows, for deciding which plain titles to hide —
# same coordinate space add_text_title's offsets end up in.
clip_spans = [
(max(0.0, float(s["start"]) - clip_source_start), min(clip_duration, float(s["end"]) - clip_source_start))
for s in spans
if float(s["end"]) > clip_source_start and float(s["start"]) < window_end
]
all_words = data.get("words", [])
dynamic_lines = 0
dynamic_word_count = 0
emphasis_words = _words_in_spans(all_words, spans)
clip_emphasis_words = _words_overlapping_clip(emphasis_words, clip_source_start, window_end)
if clip_emphasis_words:
all_segments = data.get("segments", [])
emphasis_segments = _segments_in_spans(all_segments, spans)
clip_segments = [
{
"start": float(s.get("start", 0.0)) - clip_source_start,
"end": float(s.get("end", 0.0)) - clip_source_start,
}
for s in emphasis_segments
if float(s.get("end", 0.0)) > clip_source_start
and float(s.get("start", 0.0)) < window_end
]
# Pass the element itself, not `name` — see the same note in
# handle_generate_dynamic_subtitles (Engine/docs/05_EXPERIENCIAS.md,
# entry 2026-08-17).
dynamic_lines = len(
modifier.generate_dynamic_subtitles(
el, clip_emphasis_words, dynamic_config, segments=clip_segments
)
)
dynamic_word_count = len(clip_emphasis_words)
# Plain covers EVERY word in the clip — never filtered by emphasis.
# Titles landing under a dynamic phrase are disabled below instead of
# never being created, so turning emphasis off later never leaves a
# silent gap where neither style is on screen.
plain_created = 0
plain_hidden = 0
clip_all_words = _words_overlapping_clip(all_words, clip_source_start, window_end)
blocks = _plain_subtitle_blocks(clip_all_words, max_words)
for block in blocks:
parts = [
_plain_word_text(w.get("word", ""), uppercase=uppercase, keep_punctuation=keep_punctuation)
for w in block
]
text = " ".join(p for p in parts if p).strip()
if not text:
continue
start = max(0.0, min(float(w.get("start", 0.0)) for w in block))
end = max(float(w.get("end", start)) for w in block)
duration = max(end - start, modifier.frame_duration_fraction())
title = modifier.add_text_title(
el,
text,
offset=f"{start:.6f}s",
duration=f"{duration:.6f}s",
lane=20,
position=f"0 {position_y:g}",
font=plain_font,
font_size=plain_font_size,
font_color=plain_font_color,
bold=True,
face=None,
font_scale=1.0,
size_param=plain_font_size,
)
plain_created += 1
if _overlaps_any_span(start, end, clip_spans):
title.set("enabled", "0")
plain_hidden += 1
if dynamic_lines or plain_created:
added.append(
(name, dynamic_lines, plain_created, plain_hidden, dynamic_word_count + len(clip_all_words))
)
else:
skipped.append((name, "no words in clip's source range"))
if not added:
text = "# Subtitles by Emphasis\n\nNo captions generated — file unchanged (nothing saved)."
if skipped:
text += "\n\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
return _text_result(text)
modifier.save(output_path)
total_dynamic = sum(d for _, d, _, _, _ in added)
total_plain = sum(p for _, _, p, _, _ in added)
total_hidden = sum(h for _, _, _, h, _ in added)
total_words = sum(w for _, _, _, _, w in added)
result = "# Subtitles by Emphasis Generated\n\n## Summary\n"
result += (
f"- **Clips Captioned**: {len(added)}\n"
f"- **Dynamic Title Lines (emphasis)**: {total_dynamic}\n"
f"- **Plain Title Blocks (full clip)**: {total_plain}\n"
f"- **Plain Blocks Hidden Under Emphasis (enabled=\"0\")**: {total_hidden}\n"
f"- **Total Words**: {total_words}\n\n"
)
result += _markdown_table(
["Clip", "Dynamic Lines", "Plain Blocks", "Hidden", "Words"],
[[n, str(d), str(p), str(h), str(w)] for n, d, p, h, w in added],
)
if no_review:
result += (
"\n## Sem revisão de ênfase\n"
"Nenhum `_phrase_actions.json` encontrado para: "
+ ", ".join(no_review)
+ " — todas as frases desses clipes saíram como legenda comum. "
"Rode a etapa 5 do Assistente (revisão de frases) antes, se quiser destaque dinâmico.\n"
)
if skipped:
result += "\n## Skipped Clips\n" + _markdown_table(
["Clip", "Reason"], [[n, r] for n, r in skipped]
)
result += f"\n\nSaved to: `{output_path}`\n\n*Transcripts are cached as _transcript.json; emphasis spans from _phrase_actions.json.*"
return _text_result(result)
HANDLERS = {
"validate_subtitle_layout": handle_validate_subtitle_layout,
"generate_dynamic_subtitles": handle_generate_dynamic_subtitles,
"generate_plain_subtitles": handle_generate_plain_subtitles,
"generate_subtitles_by_emphasis": handle_generate_subtitles_by_emphasis,
}
+166
View File
@@ -13,6 +13,7 @@ from mcp.types import TextContent, Tool
from fcpxml.diarize import assign_speakers, build_speakers, diarization_capability, diarize
from fcpxml.emphasis import EmphasisWeights
from fcpxml.llm_local import DEFAULT_BASE_URL, generate_voice_actions
from fcpxml.media_intel import media_src_to_path
from fcpxml.model_manager import (
load_hf_token,
@@ -21,6 +22,7 @@ from fcpxml.model_manager import (
save_voice_analysis_config,
)
from fcpxml.models import TimeValue
from fcpxml.phrase_review import build_phrase_review, save_phrase_review
from fcpxml.voice_actions import parse_actions, resolve_actions, speaker_cut_actions
from fcpxml.voice_features import extract_energy, extract_pitch, features_capability
from fcpxml.voice_timeline import (
@@ -93,6 +95,7 @@ TOOLS = [
"hf_token": {"type": "string", "description": "HuggingFace token for speaker diarization (default: the persisted token; omit to skip diarization)"},
"num_speakers": {"type": "string", "description": "Known number of speakers, if any (default: the persisted setting, else auto-detect)"},
"output_dir": {"type": "string", "description": "Folder to write _voice_timeline.json into (default: next to the media file)"},
"rotation": {"type": "number", "description": "Degrees the clip is rotated by in the FCPXML (e.g. a Transform filter straightening a tilted phone shot). Recorded in the timeline JSON so a preview can apply the same correction. Default 0."},
},
"required": ["media_path"]
}
@@ -167,6 +170,27 @@ TOOLS = [
"required": ["filepath", "actions"]
}
),
Tool(
name="generate_voice_script",
description="Run the WHOLE voice-edit pass internally, no wizard, no copy-paste: reuse an existing voice timeline (or transcribe + build one) -> hand it to a LOCAL model (Ollama running Gemma 3 / Llama) that directs the edit -> return the readable script (roteiro) AND the action JSON, and optionally apply it to a FCPXML. The model reads the full _voice_timeline.json (the whole file goes with the brief) and emits cut/zoom/text/marker decisions per the editar-por-voz brief; decisions are validated row-by-row so one bad row never discards the edit. Times stay in ORIGINAL source seconds; the applier resolves cuts and shifts everything else. Writes _voice_timeline.json, _phrase_review.json, _phrase_actions.json and (when applying) a _voice_edit FCPXML. Defaults to the local model 'gemma3:12b' at http://localhost:11434 — change via model/base_url.",
inputSchema={
"type": "object",
"properties": {
"media_path": {"type": "string", "description": "Path to the audio/video file to analyze and direct (.wav, .mp3, .m4a, .aac, .aif, .flac, .mov, .mp4). Required when there is no voice_timeline yet; ignored when voice_timeline is provided."},
"voice_timeline": {"type": "string", "description": "Path to an existing _voice_timeline.json (e.g. from the assistant's analysis step). When given, it is reused and transcription/acoustics are skipped — the model gets the whole file to direct the edit."},
"filepath": {"type": "string", "description": "Optional FCPXML to apply the decisions to (non-destructive: writes a _voice_edit copy). When omitted, only the script and actions are produced."},
"model": {"type": "string", "default": "gemma3:12b", "description": "Local model Ollama serves (e.g. 'gemma3:12b', 'gemma3:4b', 'llama3')"},
"base_url": {"type": "string", "default": "http://localhost:11434", "description": "Ollama base URL"},
"model_size": {"type": "string", "default": "base", "description": "Whisper model size to use if transcription is needed"},
"language": {"type": "string", "description": "ISO language code hint for transcription, if needed"},
"hf_token": {"type": "string", "description": "HuggingFace token for speaker diarization (omit to skip)"},
"num_speakers": {"type": "string", "description": "Known number of speakers, if any"},
"output_dir": {"type": "string", "description": "Folder to write the timeline/review/actions JSON into (default: next to the media file)"},
"apply_to_fcpxml": {"type": "boolean", "default": True, "description": "When filepath is given, apply the decisions to it. Set false to only produce the script."},
},
"required": []
}
),
Tool(
name="get_voice_analysis_config",
description="Read the persisted Voice Analysis settings: energy threshold, emphasis-index weights (energy/pitch_variation/rate_variation/pause_before/duration), emphasis cutoff for punch-in candidates, and emotion detection toggle/sensitivity. Shared with the MacApp settings screen (~/.fcp-mcp-server/config.json).",
@@ -349,6 +373,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
output_dir = arguments.get("output_dir")
rotation = float(arguments.get("rotation") or 0.0)
transcript, reason = _load_or_transcribe(media_path, model, language, output_dir)
if transcript is None:
@@ -368,6 +393,7 @@ async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
emphasis_floor=config["emphasis_floor"],
emotion_enabled=config["emotion_enabled"],
emotion_sensitivity=config["emotion_sensitivity"],
rotation=rotation,
)
json_path = Path(_validate_output_path(
@@ -742,6 +768,145 @@ async def handle_save_voice_analysis_config(arguments: dict) -> Sequence[TextCon
return _text_result(_voice_analysis_config_text(config))
async def handle_generate_voice_script(arguments: dict) -> Sequence[TextContent]:
"""The whole voice-edit pass, run inside the engine against a local model.
Transcribe (cached) -> build the voice timeline -> ask the local LLM to
direct the edit -> build the readable script (roteiro) + the action JSON ->
optionally apply to a FCPXML. No wizard, no copy-paste: the model's JSON is
parsed and validated like any other decision source, and the applier turns
it into FCPXML the same way it would for the rules engine.
"""
model = "gemma3:12b" if not arguments.get("model") else str(arguments["model"])
base_url = str(arguments.get("base_url") or DEFAULT_BASE_URL)
model_size = arguments.get("model_size", "base")
language = arguments.get("language")
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
output_dir = arguments.get("output_dir")
fcpxml_path = arguments.get("filepath")
apply = bool(arguments.get("apply_to_fcpxml", True)) and bool(fcpxml_path)
# Camino 1: já temos uma voice timeline (etapa de análise do assistente) —
# reaproveita e pula a transcrição/análise acústica/diarização, que é caro.
# Camino 2: só mídia — transcreve e monta a timeline do zero.
vt_arg = arguments.get("voice_timeline")
timeline = load_voice_timeline(Path(vt_arg)) if vt_arg and Path(vt_arg).is_file() else None
media_path = arguments.get("media_path")
if timeline is None:
media_path = _validate_filepath(
media_path, AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
)
transcript, reason = _load_or_transcribe(media_path, model_size, language, output_dir)
if transcript is None:
return _text_result(
f"# Roteiro por IA Local\n\nNão foi possível obter a transcrição "
f"({reason}).{_TRANSCRIBE_INSTALL_HINT}"
)
config = load_voice_analysis_config()
timeline = build_voice_timeline(
media_path,
transcript,
hf_token=token,
num_speakers=num_speakers,
weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
peak_percentile=config["peak_percentile"],
emphasis_floor=config["emphasis_floor"],
emotion_enabled=config["emotion_enabled"],
emotion_sensitivity=config["emotion_sensitivity"],
)
vt_arg = str(_validate_output_path(
str(voice_timeline_path(media_path, output_dir)),
anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent),
))
save_voice_timeline(timeline, Path(vt_arg))
else:
# A timeline veio pronta; a mídia só é necessária se for aplicar e o
# caller não a passou — deriva do próprio campo `source` da timeline.
if not media_path:
candidate = Path(vt_arg).parent / timeline.get("source", "")
media_path = str(candidate) if candidate.is_file() else None
decision = generate_voice_actions(timeline, model=model, base_url=base_url)
actions = decision["actions"]
errors = list(decision["errors"])
if not actions and errors:
# The model produced nothing usable (transport error or unparseable
# response) — report it clearly instead of a silent "0 decisions".
raise RuntimeError(
"O modelo local não devolveu decisões utilizáveis: " + "; ".join(errors)
)
review = build_phrase_review(
timeline,
[a.as_dict() for a in actions],
voice_timeline_path=str(vt_arg),
)
review_path, actions_path = save_phrase_review(str(vt_arg), review)
roteiro = _roteiro_markdown(review, timeline.get("source", ""))
roteiro_path = Path(vt_arg).with_name(Path(vt_arg).stem.replace("_voice_timeline", "") + "_roteiro.md")
roteiro_path.write_text(roteiro, encoding="utf-8")
applied_text = ""
if apply:
contents = await handle_apply_voice_actions({
"filepath": fcpxml_path,
"actions": [a.as_dict() for a in actions],
"output_dir": output_dir,
})
applied_text = "\n\n" + "\n".join(getattr(c, "text", str(c)) for c in contents)
result = f"""# Roteiro por IA Local ({model})
## Resumo
- **Fonte**: {timeline.get('source', '')}
- **Duração**: {format_duration(timeline['summary']['duration'])}
- **Decisões do modelo**: {len(actions)} (cortes/zoom/texto/marcador)
- **Linha do tempo**: {vt_arg}
- **Roteiro (legível)**: {roteiro_path}
- **Ações JSON**: {actions_path}
- **Revisão de frases**: {review_path}
"""
if errors:
result += "\n## Rejeitado / avisos\n" + "\n".join(f"- {e}" for e in errors) + "\n"
result += "\n---\n\n" + roteiro
result += applied_text
result += "\n\n*Tudo rodou internamente: o modelo local leu a timeline e decidiu a edição; nenhum passo manual foi necessário.*"
return _text_result(result)
def _roteiro_markdown(review: dict, source: str) -> str:
"""The readable script: kept lines (roteiro) then the cut/bastidor lines."""
phrases = review.get("phrases", [])
kept = [p for p in phrases if p.get("active")]
cut = [p for p in phrases if not p.get("active")]
lines = [f"# Roteiro — {source}", ""]
lines.append(f"**{len(kept)} falas mantidas · {len(cut)} cortadas**")
lines.append("")
lines.append("## Roteiro (mantido)")
if not kept:
lines.append("_Nenhuma fala mantida._")
for p in kept:
tag = ""
if p.get("emphasis", 0) >= 1:
tag = f" · zoom nível {p['emphasis']}"
spk = f"[{p.get('speaker', '')}] " if p.get("speaker") else ""
lines.append(f"- {spk}{p.get('text', '')}{tag}")
if p.get("reason"):
lines.append(f" - _decisão_: {p['reason']}")
if cut:
lines.append("")
lines.append("## Cortado / bastidor")
for p in cut:
spk = f"[{p.get('speaker', '')}] " if p.get("speaker") else ""
lines.append(f"- {spk}{p.get('text', '')}")
if p.get("reason"):
lines.append(f" - _por que cortou_: {p['reason']}")
return "\n".join(lines) + "\n"
HANDLERS = {
"diarize_media": handle_diarize_media,
"analyze_voice_features": handle_analyze_voice_features,
@@ -749,6 +914,7 @@ HANDLERS = {
"remove_speakers": handle_remove_speakers,
"refine_voice_timeline": handle_refine_voice_timeline,
"apply_voice_actions": handle_apply_voice_actions,
"generate_voice_script": handle_generate_voice_script,
"get_voice_analysis_config": handle_get_voice_analysis_config,
"save_voice_analysis_config": handle_save_voice_analysis_config,
}