Files
gart/code/server_tools/voice.py
T
João HenriqueandClaude Opus 5 1bebee4359 feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da
IA chega carregada e o editor afina frase a frase o que é ênfase e o que
fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase
recebem zoom e legenda dinâmica; as demais ficam com legenda comum.

O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas
não muda e a etapa 6 segue intacta.

Backend (fcpxml/phrase_review.py):
- build_phrase_review funde o _voice_timeline.json com as actions da IA
- trim por frase que anda em fronteira de palavra; corte parcial da IA
  chega como trim em vez de ser arredondado fora
- phrase_review_to_actions volta a cuts/zooms + emphasis_spans
- merge_saved_decisions reaplica só as decisões salvas sobre uma revisão
  remontada da análise atual, para reprocessar a voz não ficar mascarado
- resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo

App (SwiftUI):
- layout de sala de edição: preview em cima, inspector à direita, timeline
  atravessando embaixo com seis trilhas rotuladas
- preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal,
  projeto vertical), com alternância para a mídia original
- reprodução pula os trechos removidos e para no fim do trecho
- zoom manual por trecho marcado, sem guardar escala: a forma vem das
  configurações de Análise de Voz no render
- emoção da fala exposta por frase

Correções encontradas no caminho:
- VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc;
  trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22)
- teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21)

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-19 21:29:27 -04:00

755 lines
36 KiB
Python

"""Voz (análise → decisão → aplicação) — tool schemas and handlers.
Extracted from server.py; see Engine/docs/03_SERVER_TOOLS.md for the tool catalog.
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import List, Optional, Sequence, Tuple
from mcp.types import TextContent, Tool
from fcpxml.diarize import assign_speakers, build_speakers, diarization_capability, diarize
from fcpxml.emphasis import EmphasisWeights
from fcpxml.media_intel import media_src_to_path
from fcpxml.model_manager import (
load_hf_token,
load_num_speakers,
load_voice_analysis_config,
save_voice_analysis_config,
)
from fcpxml.models import TimeValue
from fcpxml.voice_actions import parse_actions, resolve_actions, speaker_cut_actions
from fcpxml.voice_features import extract_energy, extract_pitch, features_capability
from fcpxml.voice_timeline import (
build_voice_timeline,
enrich_words,
load_voice_timeline,
restrict_to_kept,
save_voice_timeline,
select_peaks,
suggest_zoom_windows,
voice_timeline_path,
)
from fcpxml.writer import FCPXMLModifier
from server_tools._shared import (
_DIARIZATION_INSTALL_HINT,
_FEATURES_INSTALL_HINT,
_TRANSCRIBE_INSTALL_HINT,
AUDIO_MEDIA_EXTENSIONS,
MAX_MEDIA_FILE_SIZE,
_apply_placed_action,
_load_or_transcribe,
_markdown_table,
_setup_modifier,
_speaker_table,
_text_result,
_validate_filepath,
_validate_output_path,
_voice_analysis_config_text,
format_duration,
)
TOOLS = [
Tool(
name="diarize_media",
description="Identify WHO is speaking (speaker diarization) in an audio/video file using pyannote.audio, and assign SPEAKER_NN labels to each word/segment of its cached transcript. Writes a _diarization.json next to the media file. Requires the optional [diarization] extra (pyannote.audio) and a HuggingFace token with access to pyannote/speaker-diarization-3.1 (set once via save_hf_token or the HF_TOKEN argument); degrades to an install/token hint without them. Transcribes first if no _transcript.json is cached yet.",
inputSchema={
"type": "object",
"properties": {
"media_path": {"type": "string", "description": "Path to audio/video file (.wav, .mp3, .m4a, .aac, .aif, .flac, .mov, .mp4)"},
"hf_token": {"type": "string", "description": "HuggingFace token with pyannote/speaker-diarization-3.1 access (default: the persisted token from save_hf_token, if any)"},
"num_speakers": {"type": "string", "description": "Known number of speakers, if you know it (speeds up and improves accuracy). Leave empty to auto-detect."},
"model": {"type": "string", "default": "base", "description": "Whisper model size to use if transcription is needed (default base)"},
"language": {"type": "string", "description": "ISO language code hint for transcription, if needed"},
},
"required": ["media_path"]
}
),
Tool(
name="analyze_voice_features",
description="Analyze HOW a voice is speaking: pitch, energy, local speech rate, pauses, and a combined emphasis index (0-1) per transcribed word, using the persisted Voice Analysis settings (energy threshold, emphasis weights, emphasis cutoff — see save_voice_analysis_config). Writes a _voice_features.json next to the media file. Requires the optional [intelligence] extra (librosa); degrades to an install hint without it. Transcribes first if no _transcript.json is cached yet.",
inputSchema={
"type": "object",
"properties": {
"media_path": {"type": "string", "description": "Path to audio/video file (.wav, .mp3, .m4a, .aac, .aif, .flac, .mov, .mp4)"},
"model": {"type": "string", "default": "base", "description": "Whisper model size to use if transcription is needed (default base)"},
"language": {"type": "string", "description": "ISO language code hint for transcription, if needed"},
},
"required": ["media_path"]
}
),
Tool(
name="build_voice_timeline",
description="Build the consolidated voice timeline: WHAT was said (transcript), WHO said it (diarization), and HOW it was said (pitch/energy/rate/pauses -> emphasis index), merged into one AI-readable JSON written next to the media as _voice_timeline.json. This is the source of truth for automated editing — layered as summary -> segments -> words, with all acoustic values normalized 0-1 and documented inline, so a model can read the narrative shape and decide how to direct the edit. Every layer degrades independently: no librosa means acoustic values are 0, no HuggingFace token means a single default speaker; the document shape never changes.",
inputSchema={
"type": "object",
"properties": {
"media_path": {"type": "string", "description": "Path to audio/video file (.wav, .mp3, .m4a, .aac, .aif, .flac, .mov, .mp4)"},
"model": {"type": "string", "default": "base", "description": "Whisper model size to use if transcription is needed (default base)"},
"language": {"type": "string", "description": "ISO language code hint for transcription, if needed"},
"hf_token": {"type": "string", "description": "HuggingFace token for speaker diarization (default: the persisted token; omit to skip diarization)"},
"num_speakers": {"type": "string", "description": "Known number of speakers, if any (default: the persisted setting, else auto-detect)"},
"output_dir": {"type": "string", "description": "Folder to write _voice_timeline.json into (default: next to the media file)"},
},
"required": ["media_path"]
}
),
Tool(
name="remove_speakers",
description="Cut everything one or more speakers say out of the timeline — the standard cleanup on an interview shoot, where the interviewer or a crew member talks during the take and only the subject should survive. Reads the media's _voice_timeline.json (build it first with build_voice_timeline, which must have run with diarization so speakers are separated). Call without speaker_ids to just LIST who was detected, with speaking share and sample lines, so you can tell who is who before cutting anything. Non-destructive: writes a _voice_edit copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"media_path": {"type": "string", "description": "Media whose _voice_timeline.json holds the speakers (default: the timeline's first clip media)"},
"speaker_ids": {"type": "array", "items": {"type": "string"}, "description": "Speakers to REMOVE (e.g. [\"SPEAKER_01\"]). Omit to only list the detected speakers without editing."},
"padding": {"type": "number", "default": 0.15, "description": "Seconds trimmed inside each cut so the kept speaker's first syllable is never clipped (default 0.15)"},
"output_path": {"type": "string", "description": "Output path (default: adds _voice_edit suffix)"},
},
"required": ["filepath"]
}
),
Tool(
name="refine_voice_timeline",
description="Re-analyze a voice timeline over only the material that survives a set of cuts, then propose punch-in windows over it. Emphasis is RELATIVE — energy is scored against the loudest word of the recording — so once the loudest moment is cut (a laugh, an aside to the crew), every remaining score is measured against something the viewer will never see and the ranking points at the wrong words. Run this after deciding cuts and before deciding zooms. Cheap: it re-normalizes the already-measured numbers, never re-reads the audio. Times stay in ORIGINAL source seconds, so the result feeds straight back into apply_voice_actions.",
inputSchema={
"type": "object",
"properties": {
"media_path": {"type": "string", "description": "Media whose _voice_timeline.json will be refined (build it first with build_voice_timeline)"},
"cuts": {
"type": "array",
"description": "The ranges being REMOVED, in original source seconds. Pass the cut actions you already decided; anything overlapping them is excluded from the re-analysis.",
"items": {
"type": "object",
"properties": {
"start": {"type": "number", "description": "Start in original source seconds"},
"end": {"type": "number", "description": "End in original source seconds"},
},
"required": ["start", "end"],
},
},
"min_gap": {"type": "number", "default": 8.0, "description": "Minimum seconds between two proposed zooms — effects stacked close together read as nervous editing (default 8.0)"},
"max_zooms": {"type": "integer", "description": "Cap on how many zoom candidates to return (default: no cap — cut the list by rhythm yourself)"},
"save": {"type": "boolean", "default": False, "description": "Also write the refined timeline as _voice_timeline_refined.json next to the media"},
"output_dir": {"type": "string", "description": "Folder holding _voice_timeline.json (default: next to the media file)"},
},
"required": ["media_path", "cuts"]
}
),
Tool(
name="apply_voice_actions",
description="Apply a list of editing decisions (from the rules engine, or from a model that read the _voice_timeline.json) to a timeline, producing FCPXML. Actions are validated first and reported per row, so one malformed decision never discards the edit. All action times are in ORIGINAL source seconds: cuts are resolved first and every other action is moved onto its post-cut position automatically, so decisions never land on the wrong frame. Actions pointing into removed material are dropped and reported, not silently slid. Non-destructive: writes a _voice_edit copy.",
inputSchema={
"type": "object",
"properties": {
"filepath": {"type": "string", "description": "Path to FCPXML file"},
"actions": {
"type": "array",
"description": "The decision list. Each item: {kind, start, end, params, reason, speaker}. kind is cut | zoom | text | marker. Times in original source seconds. zoom takes params.scale (1.0-3.0, default 1.3); text requires params.content.",
"items": {
"type": "object",
"properties": {
"kind": {"type": "string", "enum": ["cut", "zoom", "text", "marker"]},
"start": {"type": "number", "description": "Start in original source seconds"},
"end": {"type": "number", "description": "End in original source seconds"},
"params": {"type": "object", "description": "kind-specific: {scale} for zoom, {content} for text"},
"reason": {"type": "string", "description": "Why this decision was made — kept for review"},
"speaker": {"type": "string", "description": "Speaker id this decision relates to, if any"},
},
"required": ["kind", "start", "end"],
},
},
"output_path": {"type": "string", "description": "Output path (default: adds _voice_edit suffix)"},
},
"required": ["filepath", "actions"]
}
),
Tool(
name="get_voice_analysis_config",
description="Read the persisted Voice Analysis settings: energy threshold, emphasis-index weights (energy/pitch_variation/rate_variation/pause_before/duration), emphasis cutoff for punch-in candidates, and emotion detection toggle/sensitivity. Shared with the MacApp settings screen (~/.fcp-mcp-server/config.json).",
inputSchema={"type": "object", "properties": {}}
),
Tool(
name="save_voice_analysis_config",
description="Persist Voice Analysis settings. Only the fields you pass are changed; omitted fields keep their current value. emphasis_weights don't need to sum to 1 (normalized internally). Shared with the MacApp settings screen (~/.fcp-mcp-server/config.json).",
inputSchema={
"type": "object",
"properties": {
"energy_threshold": {"type": "number", "description": "0-1, how loud (normalized RMS) counts as 'high energy' (default 0.5)"},
"emphasis_weights": {
"type": "object",
"description": "Any subset of {energy, pitch_variation, rate_variation, pause_before, duration} weights for the emphasis index",
"properties": {
"energy": {"type": "number"},
"pitch_variation": {"type": "number"},
"rate_variation": {"type": "number"},
"pause_before": {"type": "number"},
"duration": {"type": "number"},
},
},
"peak_percentile": {"type": "number", "description": "Fraction of words selected as peaks, 0-1 (default 0.02 = top 2%). Selection is relative because the emphasis index's real range depends on the material — measured on a real interview it never passed 0.55."},
"emphasis_floor": {"type": "number", "description": "0-1 minimum emphasis for a peak, guarding genuinely flat audio (default 0.25)"},
"emotion_enabled": {"type": "boolean", "description": "Whether emotion detection runs as part of voice analysis (default false)"},
"emotion_sensitivity": {"type": "number", "description": "0-1 confidence threshold to accept an emotion label (default 0.5)"},
},
}
),
]
async def handle_diarize_media(arguments: dict) -> Sequence[TextContent]:
media_path = _validate_filepath(
arguments["media_path"], AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
)
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
num_speakers = str(arguments.get("num_speakers") or "").strip()
model = arguments.get("model", "base")
language = arguments.get("language")
ok, message = diarization_capability(token)
if not ok:
return _text_result(f"# Speaker Diarization\n\n{message}{_DIARIZATION_INSTALL_HINT}")
transcript, reason = _load_or_transcribe(media_path, model, language)
if transcript is None:
return _text_result(
f"# Speaker Diarization\n\nCould not obtain a transcript to diarize "
f"({reason}).{_TRANSCRIBE_INSTALL_HINT}"
)
tracks = diarize(media_path, token, num_speakers)
if tracks is None:
return _text_result(
"# Speaker Diarization\n\nDiarization failed — check the HuggingFace "
"token has accepted the pyannote/speaker-diarization-3.1 model terms, "
"and that the media file is readable."
)
segments, words = assign_speakers(
transcript.get("segments", []), transcript.get("words", []), tracks
)
speakers = build_speakers(segments)
diarization_data = {
"source": Path(media_path).name,
"speakers": speakers,
"segments": segments,
"words": words,
}
json_path = _validate_output_path(
str(Path(media_path).with_name(Path(media_path).stem + "_diarization.json")),
anchor_dir=str(Path(media_path).parent),
)
with open(json_path, "w") as f:
json.dump(diarization_data, f, indent=2)
result_text = f"""# Speaker Diarization
## Summary
- **Source**: {Path(media_path).name}
- **Speakers Detected**: {len(speakers)}
- **Segments**: {len(segments)}
- **Diarization JSON**: {json_path}
## Speakers
"""
result_text += _markdown_table(
["ID", "Name"], [[s["id"], s["name"]] for s in speakers]
) + "\n"
result_text += (
"\n*Next: `build_voice_timeline` to cross this with acoustic features, "
"or use the segments/words directly for speaker-aware editing.*"
)
return _text_result(result_text)
async def handle_analyze_voice_features(arguments: dict) -> Sequence[TextContent]:
media_path = _validate_filepath(
arguments["media_path"], AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
)
model = arguments.get("model", "base")
language = arguments.get("language")
ok, message = features_capability()
if not ok:
return _text_result(f"# Voice Feature Analysis\n\n{message}{_FEATURES_INSTALL_HINT}")
transcript, reason = _load_or_transcribe(media_path, model, language)
if transcript is None:
return _text_result(
f"# Voice Feature Analysis\n\nCould not obtain a transcript to analyze "
f"({reason}).{_TRANSCRIBE_INSTALL_HINT}"
)
words = transcript.get("words", [])
if not words:
return _text_result("# Voice Feature Analysis\n\nNo words in transcript — nothing to analyze.")
config = load_voice_analysis_config()
weights = EmphasisWeights.from_dict(config["emphasis_weights"])
enriched = enrich_words(
words, extract_pitch(media_path), extract_energy(media_path), weights
)
energy_threshold = config["energy_threshold"]
high_energy_words = [w for w in enriched if w["energy_norm"] >= energy_threshold]
high_emphasis_words = select_peaks(
enriched, config["peak_percentile"], config["emphasis_floor"]
)
features_data = {"source": Path(media_path).name, "config": config, "words": enriched}
json_path = _validate_output_path(
str(Path(media_path).with_name(Path(media_path).stem + "_voice_features.json")),
anchor_dir=str(Path(media_path).parent),
)
with open(json_path, "w") as f:
json.dump(features_data, f, indent=2)
result_text = f"""# Voice Feature Analysis
## Summary
- **Source**: {Path(media_path).name}
- **Words Analyzed**: {len(enriched)}
- **High-Energy Words** (>= {energy_threshold:.2f}): {len(high_energy_words)}
- **Peak Words** (top {config["peak_percentile"]:.1%}): {len(high_emphasis_words)}
- **Features JSON**: {json_path}
## Top Emphasis Words
"""
top = sorted(enriched, key=lambda w: w["emphasis"], reverse=True)[:10]
result_text += _markdown_table(
["Word", "Time", "Emphasis", "Energy", "Pitch Δ"],
[
[
w.get("word", ""),
f"{w.get('start', 0):.2f}s",
f"{w['emphasis']:.2f}",
f"{w['energy_norm']:.2f}",
f"{w['pitch_delta']:.2f}",
]
for w in top
],
) + "\n"
result_text += (
"\n*Thresholds and emphasis weights are configurable in Voice Analysis "
"settings (`save_voice_analysis_config`). Next: `diarize_media` to add "
"speaker labels.*"
)
return _text_result(result_text)
async def handle_build_voice_timeline(arguments: dict) -> Sequence[TextContent]:
media_path = _validate_filepath(
arguments["media_path"], AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
)
model = arguments.get("model", "base")
language = arguments.get("language")
token = str(arguments.get("hf_token") or "").strip() or load_hf_token() or None
num_speakers = str(arguments.get("num_speakers") or "").strip() or load_num_speakers()
output_dir = arguments.get("output_dir")
transcript, reason = _load_or_transcribe(media_path, model, language, output_dir)
if transcript is None:
return _text_result(
f"# Voice Timeline\n\nCould not obtain a transcript "
f"({reason}).{_TRANSCRIBE_INSTALL_HINT}"
)
config = load_voice_analysis_config()
timeline = build_voice_timeline(
media_path,
transcript,
hf_token=token,
num_speakers=num_speakers,
weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
peak_percentile=config["peak_percentile"],
emphasis_floor=config["emphasis_floor"],
emotion_enabled=config["emotion_enabled"],
emotion_sensitivity=config["emotion_sensitivity"],
)
json_path = Path(_validate_output_path(
str(voice_timeline_path(media_path, output_dir)),
anchor_dir=str(Path(output_dir) if output_dir else Path(media_path).parent),
))
save_voice_timeline(timeline, json_path)
summary = timeline["summary"]
layers = timeline["layers"]
result_text = f"""# Voice Timeline
## Summary
- **Source**: {timeline["source"]}
- **Duration**: {format_duration(summary["duration"])}
- **Speakers**: {summary["speaker_count"]}
- **Segments**: {summary["segment_count"]} ({summary["word_count"]} words)
- **Average Emphasis**: {summary["avg_emphasis"]:.2f}
- **Peak Moments** ({summary["peak_selection"]}): {summary["peak_count"]}
- **Timeline JSON**: {json_path}
## Analysis Layers
"""
result_text += _markdown_table(
["Layer", "Status"],
[
["Transcript", "yes" if layers["transcript"] else "empty"],
[
"Acoustics (pitch/energy)",
"yes" if layers["acoustics"] else "FAILED — every acoustic value is 0",
],
["Speakers", "yes" if layers["speakers"] else "not run — single default speaker"],
["Emotion", "yes" if layers.get("emotion") else "not run"],
],
) + "\n"
if summary["peak_moments"]:
result_text += "\n## Peak Moments\n"
result_text += _markdown_table(
["Time", "Word", "Speaker", "Emphasis"],
[
[
f"{m['time']:.2f}s",
m["text"],
m["speaker"],
f"{m['emphasis']:.2f}",
]
for m in summary["peak_moments"][:10]
],
) + "\n"
result_text += (
"\n*The JSON is layered summary -> segments -> words with normalized "
"0-1 values, ready to hand to a model for edit direction.*"
)
return _text_result(result_text)
async def handle_remove_speakers(arguments: dict) -> Sequence[TextContent]:
filepath = _validate_filepath(arguments["filepath"], ('.fcpxml', '.fcpxmld'))
speaker_ids = arguments.get("speaker_ids") or []
media_path = arguments.get("media_path")
if not media_path:
modifier = FCPXMLModifier(filepath)
for _, clip_el in modifier._iter_spine_clips():
src = modifier.resources.get(clip_el.get("ref", ""), {}).get("src", "")
candidate = media_src_to_path(src)
if candidate and Path(candidate).is_file():
media_path = candidate
break
if not media_path:
return _text_result(
"# Speakers\n\nNo source media found for this timeline — pass `media_path` explicitly."
)
timeline = _read_voice_timeline(media_path, arguments.get("output_dir"))
if timeline is None:
return _text_result(
f"# Speakers\n\nNo voice timeline for `{Path(media_path).name}` yet.\n\n"
"Run `build_voice_timeline` on it first."
)
profiles = timeline.get("speakers", [])
header = f"# Speakers in {timeline.get('source', '')}\n\n" + _speaker_table(profiles) + "\n"
if len(profiles) < 2:
header += (
"\n> Only one speaker is present. Either the recording really has one "
"voice, or diarization did not run — check the Models tab for the "
"HuggingFace token.\n"
)
for p in profiles:
samples = [s for s in p.get("samples", []) if s]
if samples:
header += f"\n**{p['id']}** ({p.get('name', '')}) says things like:\n"
header += "".join(f"> {s}\n" for s in samples[:2])
if not speaker_ids:
return _text_result(
header
+ "\n*Nothing was edited. Re-run with `speaker_ids` naming who to REMOVE — "
"typically the interviewer or crew, keeping the subject.*"
)
known = {p["id"] for p in profiles}
unknown = [s for s in speaker_ids if s not in known]
if unknown:
return _text_result(
header + f"\n**Unknown speaker(s): {', '.join(unknown)}** — nothing was edited."
)
if set(speaker_ids) >= known:
return _text_result(
header + "\n**That would remove every speaker**, leaving nothing — nothing was edited."
)
actions = speaker_cut_actions(
timeline, speaker_ids, padding=float(arguments.get("padding", 0.15))
)
if not actions:
return _text_result(header + "\n No speech found for those speakers — nothing was edited.")
result = await handle_apply_voice_actions({
**arguments,
"actions": [a.as_dict() for a in actions],
})
removed = sum(a.duration for a in actions)
return _text_result(
header
+ f"\n## Removed\n- **Speakers cut**: {', '.join(speaker_ids)}\n"
+ f"- **Speech removed**: {format_duration(removed)} across {len(actions)} segments\n\n"
+ result[0].text
)
def _read_voice_timeline(media_path: str, output_dir: Optional[str] = None) -> Optional[dict]:
"""Load the cached voice timeline, project folder first.
``build_voice_timeline`` writes to the chosen project folder when one is
set and beside the media otherwise, so a reader that only checks one of
the two reports "no voice timeline yet" for a file that exists. Checking
both also keeps timelines built before the project folder existed
readable.
"""
if output_dir:
timeline = load_voice_timeline(voice_timeline_path(media_path, output_dir))
if timeline is not None:
return timeline
return load_voice_timeline(voice_timeline_path(media_path))
async def handle_refine_voice_timeline(arguments: dict) -> Sequence[TextContent]:
media_path = _validate_filepath(
arguments["media_path"], AUDIO_MEDIA_EXTENSIONS, max_size=MAX_MEDIA_FILE_SIZE
)
timeline = _read_voice_timeline(media_path, arguments.get("output_dir"))
if timeline is None:
return _text_result(
f"# Refined Voice Timeline\n\nNo voice timeline for "
f"`{Path(media_path).name}` yet.\n\nRun `build_voice_timeline` on it first."
)
cut_ranges: List[Tuple[float, float]] = []
rejected: List[str] = []
for i, raw in enumerate(arguments.get("cuts") or []):
try:
start, end = float(raw["start"]), float(raw["end"])
except (TypeError, ValueError, KeyError):
rejected.append(f"cut #{i}: start/end must be numbers")
continue
if end <= start:
rejected.append(f"cut #{i}: end ({end}) must be after start ({start})")
continue
cut_ranges.append((start, end))
config = load_voice_analysis_config()
refined = restrict_to_kept(
timeline,
cut_ranges,
weights=EmphasisWeights.from_dict(config["emphasis_weights"]),
peak_percentile=config["peak_percentile"],
emphasis_floor=config["emphasis_floor"],
)
before, after = timeline["summary"], refined["summary"]
zooms = suggest_zoom_windows(
refined,
min_gap=float(arguments.get("min_gap", 8.0)),
max_zooms=arguments.get("max_zooms"),
)
result = f"""# Refined Voice Timeline
Re-normalized over the material that survives {len(cut_ranges)} cut(s).
"""
result += _markdown_table(
["Measure", "Raw recording", "Survivors only"],
[
["Duration", format_duration(before["duration"]), format_duration(after["duration"])],
["Segments", str(before["segment_count"]), str(after["segment_count"])],
["Words", str(before["word_count"]), str(after["word_count"])],
["Average emphasis", f"{before['avg_emphasis']:.3f}", f"{after['avg_emphasis']:.3f}"],
["Peak moments", str(before["peak_count"]), str(after["peak_count"])],
],
) + "\n"
if after["word_count"] == 0:
result += "\n> The cuts removed every word — nothing left to analyze.\n"
if after["peak_moments"]:
result += "\n## Peak Moments (re-ranked)\n"
result += _markdown_table(
["Time", "Word", "Speaker", "Emphasis"],
[
[f"{m['time']:.2f}s", m["text"], m["speaker"], f"{m['emphasis']:.2f}"]
for m in after["peak_moments"][:10]
],
) + "\n"
if zooms:
result += "\n## Zoom Candidates\n"
result += _markdown_table(
["Start", "End", "Word", "Emphasis", "Line"],
[
[f"{z['start']:.2f}s", f"{z['end']:.2f}s", z["word"],
f"{z['emphasis']:.2f}", z["line"][:60]]
for z in zooms
],
) + "\n"
else:
result += "\n## Zoom Candidates\n\nNone — no content words survived the cuts.\n"
if arguments.get("save"):
p = Path(media_path)
json_path = Path(_validate_output_path(
str(p.with_name(p.stem + "_voice_timeline_refined.json")),
anchor_dir=str(p.parent),
))
save_voice_timeline(refined, json_path)
result += f"\n**Refined JSON**: {json_path}\n"
if rejected:
result += "\n## Rejected cuts\n" + "\n".join(f"- {r}" for r in rejected) + "\n"
result += (
"\n*Candidates, not obligations — cut the list by rhythm. Times are in "
"original source seconds, ready for `apply_voice_actions`.*"
)
return _text_result(result)
async def handle_apply_voice_actions(arguments: dict) -> Sequence[TextContent]:
raw_actions = arguments.get("actions")
if raw_actions is None:
return _text_result("# Voice Actions\n\nNo `actions` provided — nothing to apply.")
actions, errors = parse_actions(raw_actions)
if not actions:
text = "# Voice Actions\n\nNo valid actions to apply."
if errors:
text += "\n\n## Rejected\n" + "\n".join(f"- {e}" for e in errors)
return _text_result(text)
cut_ranges, placed, dropped = resolve_actions(actions)
filepath, output_path, modifier = _setup_modifier(arguments, "_voice_edit")
def source_window(clip_el) -> tuple[float, float]:
"""The span of source media a spine clip actually uses."""
start = modifier.source_file_start(clip_el).to_seconds()
duration = modifier._parse_time(clip_el.get("duration", "0s")).to_seconds()
return start, start + duration
applied: list[list[str]] = []
unplaced: list[str] = []
# Placements go on before cuts: they are anchored in source coordinates,
# and cutting afterwards ripples the spine around them.
# Cuts go FIRST. Cutting splits a clip into pieces and rewrites the
# spine around them, which would duplicate a zoom onto every piece and
# lose markers entirely. Cutting first means placements land on final,
# stable clips — and `resolve_actions` already moved their times onto
# the post-cut timeline, so they still point at the same moment.
cuts_made = 0
for _, clip_el in modifier._iter_spine_clips():
clip_start, clip_end = source_window(clip_el)
to_frame = modifier.snap_seconds_to_frame
ranges = [
(to_frame(max(s, clip_start) - clip_start), to_frame(min(e, clip_end) - clip_start))
for s, e in cut_ranges
if min(e, clip_end) > max(s, clip_start)
]
ranges = [(a, b) for a, b in ranges if b > a]
if ranges and modifier.cut_clip_ranges(clip_el, ranges) > TimeValue.zero():
cuts_made += len(ranges)
def timeline_window(clip_el) -> tuple[float, float]:
"""Where a spine clip sits on the timeline, in seconds."""
offset = modifier._parse_time(clip_el.get("offset", "0s")).to_seconds()
duration = modifier._parse_time(clip_el.get("duration", "0s")).to_seconds()
return offset, offset + duration
for action in placed:
host = next(
(
clip_el
for _, clip_el in modifier._iter_spine_clips()
if timeline_window(clip_el)[0] <= action.start < timeline_window(clip_el)[1]
),
None,
)
if host is None:
unplaced.append(
f"{action.kind} @ {action.start:.2f}s — falls outside the edited timeline"
)
continue
try:
what = _apply_placed_action(modifier, host, action, timeline_window(host)[0])
applied.append([f"{action.start:.2f}s", what, host.get("name", ""), action.reason])
except (ValueError, KeyError) as exc:
unplaced.append(f"{action.kind} @ {action.start:.2f}s — {exc}")
# Rename the project so it does not land in the library indistinguishable
# from the original. FCP imports by the name in the XML, so an untouched
# name puts two same-named projects in the same event — and the edit looks
# like it did nothing, because the original is what gets opened.
project_name = ""
project_el = modifier.root.find(".//project")
if project_el is not None:
project_name = f"{project_el.get('name', 'Projeto')} — corte por voz"
project_el.set("name", project_name)
modifier.save(output_path)
result = f"""# Voice Actions Applied
## Summary
- **Actions received**: {len(actions)}
- **Placed** (zoom/text/marker): {len(applied)}
- **Cuts applied**: {cuts_made}
- **Project name**: {project_name or '(unchanged)'}
- **Saved to**: `{output_path}`
"""
if applied:
result += "## Applied\n" + _markdown_table(
["Time", "Action", "Clip", "Reason"], applied[:40]
) + "\n"
if dropped:
result += "\n## Dropped (pointed into removed material)\n" + "\n".join(
f"- {a.kind} @ {a.start:.2f}s — {a.reason or 'no reason given'}" for a in dropped
) + "\n"
if unplaced:
result += "\n## Not placed\n" + "\n".join(f"- {u}" for u in unplaced) + "\n"
if errors:
result += "\n## Rejected\n" + "\n".join(f"- {e}" for e in errors) + "\n"
result += "\n*Non-destructive: the original file is untouched.*"
return _text_result(result)
async def handle_get_voice_analysis_config(arguments: dict) -> Sequence[TextContent]:
return _text_result(_voice_analysis_config_text(load_voice_analysis_config()))
async def handle_save_voice_analysis_config(arguments: dict) -> Sequence[TextContent]:
config = save_voice_analysis_config(
energy_threshold=arguments.get("energy_threshold"),
emphasis_weights=arguments.get("emphasis_weights"),
peak_percentile=arguments.get("peak_percentile"),
emphasis_floor=arguments.get("emphasis_floor"),
emotion_enabled=arguments.get("emotion_enabled"),
emotion_sensitivity=arguments.get("emotion_sensitivity"),
)
return _text_result(_voice_analysis_config_text(config))
HANDLERS = {
"diarize_media": handle_diarize_media,
"analyze_voice_features": handle_analyze_voice_features,
"build_voice_timeline": handle_build_voice_timeline,
"remove_speakers": handle_remove_speakers,
"refine_voice_timeline": handle_refine_voice_timeline,
"apply_voice_actions": handle_apply_voice_actions,
"get_voice_analysis_config": handle_get_voice_analysis_config,
"save_voice_analysis_config": handle_save_voice_analysis_config,
}