"""Mídia e transcrição: cache, corte por trecho falado e ações posicionadas. Extraído de _shared.py — ver server_tools/_shared/__init__.py. """ from __future__ import annotations import json from pathlib import Path from fcpxml.media_intel import media_src_to_path from fcpxml.model_manager import load_dynamic_subtitle_config, load_voice_analysis_config from fcpxml.models import ( TimeValue, ) from fcpxml.text_layout import TEXT_TEMPLATE_FONT_SCALE, measure_text from fcpxml.transcribe import invert_ranges, merge_ranges, transcribe from .formatting import _markdown_table, format_duration from .paths import _validate_output_path from .project import _text_result AUDIO_MEDIA_EXTENSIONS = ( '.wav', '.aif', '.aiff', '.mp3', '.m4a', '.aac', '.flac', '.mov', '.mp4', ) _DIARIZATION_INSTALL_HINT = ( "\n\nInstall the optional diarization extra:\n\n" " pip install 'fcp-mcp-server[diarization]'\n\n" "and set a HuggingFace token with access to " "pyannote/speaker-diarization-3.1 (pass hf_token= or persist one via " "save_hf_token)." ) _FEATURES_INSTALL_HINT = ( "\n\nInstall the optional media-intelligence extra:\n\n" " pip install 'fcp-mcp-server[intelligence]'" ) def _apply_placed_action(modifier, clip_el, action, clip_start: float) -> str: """Apply one non-cut action to the clip that hosts it. ``clip_start`` is where that clip begins on the timeline; the writer wants times relative to the clip's own head, so the rebase happens here — the single place that knows about the conversion. The clip *element* is passed through rather than its name: after a cut the pieces share a name, and a name lookup would land every edit on the first piece. """ rel_start = action.start - clip_start rel_end = action.end - clip_start if action.kind == "zoom": config = load_voice_analysis_config() # Only forward an explicit ease — otherwise add_zoom's own default # (a fast ramp in, instant snap back out) is what should apply. zoom_args = { "ease": float(action.params.get("ease", config["zoom_ease_in"])), "ease_out": float(action.params.get("ease_out", config["zoom_ease_out"])), } mode = str(action.params.get("mode", config["zoom_mode"])) if mode == "in": zoom_args["hold_at_end"] = True zoom_args["start_at_peak"] = False elif mode == "out": zoom_args["hold_at_end"] = False zoom_args["start_at_peak"] = True elif mode == "in_out": zoom_args["hold_at_end"] = False zoom_args["start_at_peak"] = False modifier.add_zoom( clip_id=clip_el, start=rel_start, end=rel_end, scale=float(action.params.get("scale", config["zoom_scale"])), **zoom_args, ) return f"zoom {float(action.params.get('scale', config['zoom_scale'])):.2f}x" if action.kind == "text": # Default to the "Legendas Dinâmicas" emphasis style (the font used # to highlight a word in the captions) rather than a hardcoded # Helvetica Neue, so a callout like "MASTOPEXIA" matches the rest of # the video's on-screen text instead of looking like a stray default # title. Any of these the action itself specifies still wins. subtitle_cfg = load_dynamic_subtitle_config() font = action.params.get("font", subtitle_cfg["emphasis_font"]) face = action.params.get("face", subtitle_cfg["emphasis_face"]) font_scale = float(subtitle_cfg.get("text_scale", TEXT_TEMPLATE_FONT_SCALE) or 1.0) requested_size = int(action.params.get("font_size", subtitle_cfg["emphasis_size"])) requested_kerning = float(action.params.get("kerning", 0.0) or 0.0) # Voice-action callouts are not part of the dynamic subtitle block. # When omitted, put them above the subtitle band and shrink wide # phrases to the title-safe width. The previous default (Position 0 0, # full emphasis size) made long callouts like "PRÓTESES DE SILICONE" # collide with captions and run off both sides of a vertical frame. emitted_size = requested_size * font_scale emitted_kerning = requested_kerning * font_scale safe_width = modifier.frame_width() * 0.90 width = measure_text( action.params["content"], emitted_size, bold=bool(action.params.get("bold", False)), kerning=emitted_kerning, font=font, face=face, ) font_size = requested_size if width > safe_width and width > 0: font_size = max(32, int(requested_size * safe_width / width)) position = action.params.get("position") if not position: position = f"0 {modifier.frame_height() * 0.23:g}" modifier.add_text_title( clip_el, action.params["content"], offset=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(), duration=modifier.snap_seconds_to_frame(action.duration).to_fcpxml(), position=position, font=font, font_size=font_size, font_color=action.params.get("font_color", subtitle_cfg["emphasis_color"]), face=face, bold=action.params.get("bold", False), ) return f"text \"{action.params['content'][:24]}\"" # marker modifier.add_marker( clip_id=clip_el, timecode=modifier.snap_seconds_to_frame(rel_start).to_fcpxml(), name=action.params.get("content") or action.reason or "Voice action", note=action.reason or None, ) return "marker" TRANSCRIBE_MAX_MEDIA = 10 _TRANSCRIBE_INSTALL_HINT = ( "\n\nInstall the optional transcription extra:\n\n" " pip install 'fcp-mcp-server[transcribe]'\n\n" "or run via uvx:\n\n" " uvx --from \"fcp-mcp-server[transcribe]\" fcp-mcp-server" ) def _transcript_json_path(media_path: str, output_dir: str | None = None) -> Path: """Where the ``_transcript.json`` for ``media_path`` lives. When ``output_dir`` (the user-selected project folder) is set, the transcript is saved/read there instead of next to the source media. """ p = Path(media_path) if output_dir: directory = Path(output_dir).expanduser() directory.mkdir(parents=True, exist_ok=True) return directory / f"{p.stem}_transcript.json" return p.with_name(p.stem + "_transcript.json") def _load_or_transcribe( media_path: str, model: str, language: str | None, output_dir: str | None = None ) -> tuple[dict | None, str]: """Load a cached ``_transcript.json`` for a media file, else transcribe and cache it. Returns ``(transcript, "")`` or ``(None, reason)``. The cache makes transcription a one-time cost per media file across all transcript tools. """ json_path = _transcript_json_path(media_path, output_dir) if json_path.is_file(): try: with open(json_path) as f: data = json.load(f) if isinstance(data, dict) and isinstance(data.get("words"), list): return data, "" except (OSError, json.JSONDecodeError, UnicodeDecodeError): pass # unreadable cache falls through to re-transcribe result = transcribe(media_path, model_size=model, language=language) if result is None: return None, "untranscribable (faster-whisper not installed or media unreadable)" anchor = str(Path(output_dir).expanduser()) if output_dir else str(Path(media_path).parent) out_path = _validate_output_path(str(json_path), anchor_dir=anchor) with open(out_path, "w") as f: json.dump({"source": Path(media_path).name, **result}, f, indent=2) return result, "" def _cut_transcript_spans(modifier, clip_filter, model, language, padding, spans_fn, keep_only=False, output_dir=None): """Shared cut engine for transcript-driven editing. ``spans_fn(words) -> [(start, end), ...]`` in source seconds. Spans are padded, clamped to each clip's used source window, optionally inverted (keep_only), snapped to the frame grid, and cut with ripple. """ to_frame = modifier.snap_seconds_to_frame cache: dict[str, tuple] = {} cuts_made: list[tuple[str, int, float]] = [] skipped: list[tuple[str, str]] = [] spine_clips = [el for _, el in modifier._iter_spine_clips()] for el in spine_clips: name = el.get("name", "") if clip_filter and name != clip_filter: continue src = modifier.resources.get(el.get("ref", ""), {}).get("src", "") media_path = media_src_to_path(src) if not media_path or not Path(media_path).is_file(): skipped.append((name, "media file missing")) continue if media_path not in cache: if len(cache) >= TRANSCRIBE_MAX_MEDIA: skipped.append((name, f"transcription cap reached ({TRANSCRIBE_MAX_MEDIA} media files)")) continue cache[media_path] = _load_or_transcribe(media_path, model, language, output_dir) data, reason = cache[media_path] if data is None: skipped.append((name, reason)) continue clip_source_start = modifier.source_file_start(el).to_seconds() clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() window_start = clip_source_start window_end = clip_source_start + clip_duration spans = spans_fn(data.get("words", [])) padded = merge_ranges([(s - padding, e + padding) for s, e in spans]) clamped = [ (max(s, window_start), min(e, window_end)) for s, e in padded if min(e, window_end) > max(s, window_start) ] if keep_only: if not clamped: # Never delete a whole clip just because nothing matched in it. skipped.append((name, "no phrase matches — left untouched (keep_only)")) continue cut_source = invert_ranges(clamped, window_start, window_end) else: cut_source = clamped cut_ranges = [ (to_frame(s - clip_source_start), to_frame(e - clip_source_start)) for s, e in cut_source ] cut_ranges = [(a, b) for a, b in cut_ranges if b > a] if not cut_ranges: continue removed = modifier.cut_clip_ranges(el, cut_ranges) if removed > TimeValue.zero(): cuts_made.append((name, len(cut_ranges), removed.to_seconds())) return cuts_made, skipped def _transcript_cut_report(title, summary_lines, cuts_made, skipped, output_path, footer): if not cuts_made: text = f"# {title}\n\nNo cuts to make — file unchanged (nothing saved)." if skipped: text += "\n\n## Skipped Clips\n" + _markdown_table( ["Clip", "Reason"], [[name, reason] for name, reason in skipped] ) if any("faster-whisper" in reason for _, reason in skipped): text += _TRANSCRIBE_INSTALL_HINT return _text_result(text) total_removed = sum(seconds for _, _, seconds in cuts_made) result = f"# {title}\n\n## Summary\n" result += "\n".join(summary_lines) + "\n" result += f"- **Clips Cut**: {len(cuts_made)}\n- **Total Removed**: {format_duration(total_removed)}\n" result += "\n## Cuts\n" result += _markdown_table( ["Clip", "Ranges Cut", "Removed"], [[name, str(count), f"{seconds:.2f}s"] for name, count, seconds in cuts_made], ) + "\n" if skipped: result += "\n## Skipped Clips\n" + _markdown_table( ["Clip", "Reason"], [[name, reason] for name, reason in skipped] ) + "\n" result += f"\nSaved to: {output_path}\n\n{footer}" return _text_result(result)