Eram 882 linhas de seis papéis sem relação, sob um nome que só dizia
"compartilhado" — o depósito onde tudo que servia a mais de um handler
acabava caindo.
media 316 transcrição em cache, corte por fala, relatório
paths 206 sandbox, limites, caminho de saída
project 116 abrir projeto, preparar modifier/generator
captions 112 SRT, VTT, listas com timestamp
detection 99 flash frames, buracos, duplicados
formatting 86 tabelas e relatórios dos handlers
O __init__ reexporta os 46 nomes, então os treze pontos que importam daqui
não mudaram.
_transcript_cut_report saiu de formatting para media: ele precisa do hint de
instalação e do _text_result, ou seja, é relatório de transcrição e não
formatação genérica — mover foi mais honesto que cruzar imports entre os
dois módulos.
Quatro testes patchavam `server_tools._shared.transcribe`; o nome agora é
ligado por _shared/media.py, então o patch passou a apontar para lá — mesmo
padrão da experiência #23.
Lint zerado, 1454 testes passando.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
275 lines
12 KiB
Python
275 lines
12 KiB
Python
"""Mídia e transcrição: cache, corte por trecho falado e ações posicionadas.
|
|
|
|
Extraído de _shared.py — ver server_tools/_shared/__init__.py.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from pathlib import Path
|
|
|
|
from fcpxml.media_intel import media_src_to_path
|
|
from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config
|
|
from fcpxml.models import (
|
|
TimeValue,
|
|
)
|
|
from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text
|
|
from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe
|
|
|
|
from .formatting import _markdown_table, format_duration
|
|
from .paths import _validate_output_path
|
|
from .project import _text_result
|
|
|
|
AUDIO_MEDIA_EXTENSIONS = (
|
|
'.wav', '.aif', '.aiff', '.mp3', '.m4a', '.aac', '.flac', '.mov', '.mp4',
|
|
)
|
|
|
|
_DIARIZATION_INSTALL_HINT = (
|
|
"\n\nInstall the optional diarization extra:\n\n"
|
|
" pip install 'fcp-mcp-server[diarization]'\n\n"
|
|
"and set a HuggingFace token with access to "
|
|
"pyannote/speaker-diarization-3.1 (pass hf_token= or persist one via "
|
|
"save_hf_token)."
|
|
)
|
|
|
|
_FEATURES_INSTALL_HINT = (
|
|
"\n\nInstall the optional media-intelligence extra:\n\n"
|
|
" pip install 'fcp-mcp-server[intelligence]'"
|
|
)
|
|
|
|
def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str:
|
|
"""Apply one non-cut action to the clip that hosts it.
|
|
|
|
``clip_start`` is where that clip begins on the timeline; the writer
|
|
wants times relative to the clip's own head, so the rebase happens here
|
|
— the single place that knows about the conversion. The clip *element*
|
|
is passed through rather than its name: after a cut the pieces share a
|
|
name, and a name lookup would land every edit on the first piece.
|
|
"""
|
|
rel_start = action.start - clip_start
|
|
rel_end = action.end - clip_start
|
|
|
|
if action.kind == "zoom":
|
|
config = load_voice_analysis_config()
|
|
# Only forward an explicit ease — otherwise add_zoom's own default
|
|
# (a fast ramp in, instant snap back out) is what should apply.
|
|
zoom_args = {
|
|
"ease": float(action.params.get("ease", config["zoom_ease_in"])),
|
|
"ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])),
|
|
}
|
|
mode = str(action.params.get("mode", config["zoom_mode"]))
|
|
if mode == "in":
|
|
zoom_args["hold_at_end"] = True
|
|
zoom_args["start_at_peak"] = False
|
|
elif mode == "out":
|
|
zoom_args["hold_at_end"] = False
|
|
zoom_args["start_at_peak"] = True
|
|
elif mode == "in_out":
|
|
zoom_args["hold_at_end"] = False
|
|
zoom_args["start_at_peak"] = False
|
|
modifier.add_zoom(
|
|
clip_id=clip_el,
|
|
start=rel_start,
|
|
end=rel_end,
|
|
scale=float(action.params.get("scale", config["zoom_scale"])),
|
|
**zoom_args,
|
|
)
|
|
return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x"
|
|
|
|
if action.kind == "text":
|
|
# Default to the "Legendas Dinâmicas" emphasis style (the font used
|
|
# to highlight a word in the captions) rather than a hardcoded
|
|
# Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of
|
|
# the video's on-screen text instead of looking like a stray default
|
|
# title. Any of these the action itself specifies still wins.
|
|
subtitle_cfg = load_dynamic_subtitle_config()
|
|
font = action.params.get("font", subtitle_cfg["emphasis_font"])
|
|
face = action.params.get("face", subtitle_cfg["emphasis_face"])
|
|
font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0)
|
|
requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"]))
|
|
requested_kerning = float(action.params.get("kerning", 0.0) or 0.0)
|
|
|
|
# Voice-action callouts are not part of the dynamic subtitle block.
|
|
# When omitted, put them above the subtitle band and shrink wide
|
|
# phrases to the title-safe width. The previous default (Position 0 0,
|
|
# full emphasis size) made long callouts like "PRÓTESES DE SILICONE"
|
|
# collide with captions and run off both sides of a vertical frame.
|
|
emitted_size = requested_size * font_scale
|
|
emitted_kerning = requested_kerning * font_scale
|
|
safe_width = modifier.frame_width() * 0.90
|
|
width = measure_text(
|
|
action.params["content"],
|
|
emitted_size,
|
|
bold=bool(action.params.get("bold", False)),
|
|
kerning=emitted_kerning,
|
|
font=font,
|
|
face=face,
|
|
)
|
|
font_size = requested_size
|
|
if width > safe_width and width > 0:
|
|
font_size = max(32, int(requested_size * safe_width / width))
|
|
position = action.params.get("position")
|
|
if not position:
|
|
position = f"0 {modifier.frame_height() * 0.23:g}"
|
|
|
|
modifier.add_text_title(
|
|
clip_el,
|
|
action.params["content"],
|
|
offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
|
|
duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(),
|
|
position=position,
|
|
font=font,
|
|
font_size=font_size,
|
|
font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]),
|
|
face=face,
|
|
bold=action.params.get("bold", False),
|
|
)
|
|
return f"text \"{action.params['content'][:24]}\""
|
|
|
|
# marker
|
|
modifier.add_marker(
|
|
clip_id=clip_el,
|
|
timecode=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(),
|
|
name=action.params.get("content") or action.reason or "Voice action",
|
|
note=action.reason or None,
|
|
)
|
|
return "marker"
|
|
|
|
TRANSCRIBE_MAX_MEDIA = 10
|
|
|
|
_TRANSCRIBE_INSTALL_HINT = (
|
|
"\n\nInstall the optional transcription extra:\n\n"
|
|
" pip install 'fcp-mcp-server[transcribe]'\n\n"
|
|
"or run via uvx:\n\n"
|
|
" uvx --from \"fcp-mcp-server[transcribe]\" fcp-mcp-server"
|
|
)
|
|
|
|
def _transcript_json_path(media_path: str, output_dir: str | None = None) -> Path:
|
|
"""Where the ``_transcript.json`` for ``media_path`` lives.
|
|
|
|
When ``output_dir`` (the user-selected project folder) is set, the
|
|
transcript is saved/read there instead of next to the source media.
|
|
"""
|
|
p = Path(media_path)
|
|
if output_dir:
|
|
directory = Path(output_dir).expanduser()
|
|
directory.mkdir(parents=True, exist_ok=True)
|
|
return directory / f"{p.stem}_transcript.json"
|
|
return p.with_name(p.stem + "_transcript.json")
|
|
|
|
def _load_or_transcribe(
|
|
media_path: str, model: str, language: str | None, output_dir: str | None = None
|
|
) -> tuple[dict | None, str]:
|
|
"""Load a cached ``_transcript.json`` for a media file, else transcribe and cache it.
|
|
|
|
Returns ``(transcript, "")`` or ``(None, reason)``. The cache makes
|
|
transcription a one-time cost per media file across all transcript tools.
|
|
"""
|
|
json_path = _transcript_json_path(media_path, output_dir)
|
|
if json_path.is_file():
|
|
try:
|
|
with open(json_path) as f:
|
|
data = json.load(f)
|
|
if isinstance(data, dict) and isinstance(data.get("words"), list):
|
|
return data, ""
|
|
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
|
pass # unreadable cache falls through to re-transcribe
|
|
result = transcribe(media_path, model_size=model, language=language)
|
|
if result is None:
|
|
return None, "untranscribable (faster-whisper not installed or media unreadable)"
|
|
anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent)
|
|
out_path = _validate_output_path(str(json_path), anchor_dir=anchor)
|
|
with open(out_path, "w") as f:
|
|
json.dump({"source": Path(media_path).name, **result}, f, indent=2)
|
|
return result, ""
|
|
|
|
def _cut_transcript_spans(modifier, clip_filter, model, language, padding, spans_fn, keep_only=False, output_dir=None):
|
|
"""Shared cut engine for transcript-driven editing.
|
|
|
|
``spans_fn(words) -> [(start, end), ...]`` in source seconds. Spans are
|
|
padded, clamped to each clip's used source window, optionally inverted
|
|
(keep_only), snapped to the frame grid, and cut with ripple.
|
|
"""
|
|
to_frame = modifier.snap_seconds_to_frame
|
|
|
|
cache: dict[str, tuple] = {}
|
|
cuts_made: list[tuple[str, int, float]] = []
|
|
skipped: list[tuple[str, str]] = []
|
|
spine_clips = [el for _, el in modifier._iter_spine_clips()]
|
|
for el in spine_clips:
|
|
name = el.get("name", "")
|
|
if clip_filter and name != clip_filter:
|
|
continue
|
|
src = modifier.resources.get(el.get("ref", ""), {}).get("src", "")
|
|
media_path = media_src_to_path(src)
|
|
if not media_path or not Path(media_path).is_file():
|
|
skipped.append((name, "media file missing"))
|
|
continue
|
|
if media_path not in cache:
|
|
if len(cache) >= TRANSCRIBE_MAX_MEDIA:
|
|
skipped.append((name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)"))
|
|
continue
|
|
cache[media_path] = _load_or_transcribe(media_path, model, language, output_dir)
|
|
data, reason = cache[media_path]
|
|
if data is None:
|
|
skipped.append((name, reason))
|
|
continue
|
|
|
|
clip_source_start = modifier.source_file_start(el).to_seconds()
|
|
clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds()
|
|
window_start = clip_source_start
|
|
window_end = clip_source_start + clip_duration
|
|
|
|
spans = spans_fn(data.get("words", []))
|
|
padded = merge_ranges([(s - padding, e + padding) for s, e in spans])
|
|
clamped = [
|
|
(max(s, window_start), min(e, window_end))
|
|
for s, e in padded
|
|
if min(e, window_end) > max(s, window_start)
|
|
]
|
|
if keep_only:
|
|
if not clamped:
|
|
# Never delete a whole clip just because nothing matched in it.
|
|
skipped.append((name, "no phrase matches — left untouched (keep_only)"))
|
|
continue
|
|
cut_source = invert_ranges(clamped, window_start, window_end)
|
|
else:
|
|
cut_source = clamped
|
|
cut_ranges = [
|
|
(to_frame(s - clip_source_start), to_frame(e - clip_source_start))
|
|
for s, e in cut_source
|
|
]
|
|
cut_ranges = [(a, b) for a, b in cut_ranges if b > a]
|
|
if not cut_ranges:
|
|
continue
|
|
removed = modifier.cut_clip_ranges(el, cut_ranges)
|
|
if removed > TimeValue.zero():
|
|
cuts_made.append((name, len(cut_ranges), removed.to_seconds()))
|
|
return cuts_made, skipped
|
|
|
|
def _transcript_cut_report(title, summary_lines, cuts_made, skipped, output_path, footer):
|
|
if not cuts_made:
|
|
text = f"# {title}\n\nNo cuts to make — file unchanged (nothing saved)."
|
|
if skipped:
|
|
text += "\n\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
|
|
)
|
|
if any("faster-whisper" in reason for _, reason in skipped):
|
|
text += _TRANSCRIBE_INSTALL_HINT
|
|
return _text_result(text)
|
|
total_removed = sum(seconds for _, _, seconds in cuts_made)
|
|
result = f"# {title}\n\n## Summary\n"
|
|
result += "\n".join(summary_lines) + "\n"
|
|
result += f"- **Clips Cut**: {len(cuts_made)}\n- **Total Removed**: {format_duration(total_removed)}\n"
|
|
result += "\n## Cuts\n"
|
|
result += _markdown_table(
|
|
["Clip", "Ranges Cut", "Removed"],
|
|
[[name, str(count), f"{seconds:.2f}s"] for name, count, seconds in cuts_made],
|
|
) + "\n"
|
|
if skipped:
|
|
result += "\n## Skipped Clips\n" + _markdown_table(
|
|
["Clip", "Reason"], [[name, reason] for name, reason in skipped]
|
|
) + "\n"
|
|
result += f"\nSaved to: {output_path}\n\n{footer}"
|
|
return _text_result(result)
|