"""Legendas: dinâmicas, comuns, SRT e as configurações de estilo. Extraído de models_api.py — a tabela de comandos segue lá. """ from __future__ import annotations import asyncio import sys from pathlib import Path # code/ is the package root for fcpxml and server modules. _CODE_DIR = str(Path(__file__).resolve().parent.parent / "code") if _CODE_DIR not in sys.path: sys.path.insert(0, _CODE_DIR) from fcpxml.media_intel import media_src_to_path from fcpxml.model_manager import ( load_dynamic_subtitle_config, load_plain_subtitle_config, save_dynamic_subtitle_config, save_plain_subtitle_config, ) from fcpxml.writer import FCPXMLModifier from . import shared from .shared import ( _derived_output, _emit_no_change_or_error, _transcript_json_path, ) from .shared import _load_cached_transcript # noqa: E402 def cmd_generate_dynamic_subtitles(args: dict) -> int: """Generate word-by-word ("karaoke") caption compound clips, one per line, using each media's cached transcript.""" path = str(args.get("path", "")) if not path or not Path(path).exists(): shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) return 1 try: from server import handle_generate_dynamic_subtitles output = _derived_output(path, "_dynamic_subtitles", args) contents = asyncio.run( handle_generate_dynamic_subtitles({**args, "filepath": path, "output_path": output}) ) message = "\n".join(getattr(content, "text", str(content)) for content in contents) if not Path(output).exists(): shared.emit({"ok": False, "error": message}) return 1 shared.emit({"ok": True, "path": output, "message": message}) return 0 except Exception as exc: shared.emit({"ok": False, "error": str(exc)}) return 1 def cmd_generate_plain_subtitles(args: dict) -> int: """Generate simple static editable subtitle title clips.""" path = str(args.get("path", "")) if not path or not Path(path).exists(): shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) return 1 try: from server import handle_generate_plain_subtitles output = _derived_output(path, "_plain_subtitles", args) contents = asyncio.run( handle_generate_plain_subtitles({**args, "filepath": path, "output_path": output}) ) message = "\n".join(getattr(content, "text", str(content)) for content in contents) if not Path(output).exists(): return _emit_no_change_or_error(path, message) shared.emit({"ok": True, "path": output, "message": message}) return 0 except Exception as exc: shared.emit({"ok": False, "error": str(exc)}) return 1 def cmd_export_srt(args: dict) -> int: """Write a captions .srt synced to the edited timeline. Each transcribed segment is mapped from its SOURCE-media timestamp to its real TIMELINE position (``clip_offset + (seg_start - clip_source_start)``), so captions only cover the frames that remain after cuts/silence removal — not the whole source file. One .srt is produced per media, in timeline order. """ path = str(args.get("path", "")) output_dir = str(args.get("output_dir", "")).strip() if not path or not Path(path).exists(): shared.emit({"ok": False, "error": "Arquivo de projeto não encontrado."}) return 1 try: modifier = FCPXMLModifier(path) except Exception as exc: shared.emit({"ok": False, "error": f"Erro ao ler o projeto: {exc}"}) return 1 # Group spine clips by media so each transcript is loaded once. by_media: dict[str, list] = {} for _, el in modifier._iter_spine_clips(): src = modifier.resources.get(el.get("ref", ""), {}).get("src", "") mp = media_src_to_path(src) if not mp or not Path(mp).is_file(): continue by_media.setdefault(mp, []).append(el) # Never emit a caption past the end of the project — Final Cut rejects an # SRT whose last cue overruns the timeline ("subtitle extends beyond project # duration"). Clamp every mapped cue end to this ceiling. timeline_total = modifier._timeline_duration().to_seconds() srt_paths: list[str] = [] for mp, clips in by_media.items(): cached = _load_cached_transcript(_transcript_json_path(mp, output_dir)) if cached is None: continue segments = cached.get("segments") or [] if not segments: continue rows: list[tuple[float, float, str, int]] = [] for el in clips: clip_source_start = modifier.source_file_start(el).to_seconds() clip_duration = modifier._parse_time(el.get("duration", "0s")).to_seconds() clip_offset = modifier._parse_time(el.get("offset", "0s")).to_seconds() window_end = clip_source_start + clip_duration for seg_index, seg in enumerate(segments): seg_start = float(seg.get("start", 0.0)) seg_end = float(seg.get("end", seg_start)) text = seg.get("text", "").strip() if not text or seg_end <= seg_start: continue # Intersect the complete source segment with this kept clip. # Testing only seg_start loses speech whose first words fall in # a removed range; interval intersection preserves the part # that remains and avoids duplicating a segment wholesale. source_start = max(seg_start, clip_source_start) source_end = min(seg_end, window_end) if source_end <= source_start: continue tl_start = clip_offset + (source_start - clip_source_start) tl_end = clip_offset + (source_end - clip_source_start) tl_start = max(0.0, min(tl_start, timeline_total)) tl_end = max(0.0, min(tl_end, timeline_total)) if tl_end > tl_start: rows.append((tl_start, tl_end, text, seg_index)) if not rows: continue rows.sort(key=lambda r: (r[0], r[1], r[3])) # Merge only pieces from the same original Whisper segment when their # mapped intervals touch. Never merge unrelated speech or invent time. merged: list[tuple[float, float, str, int]] = [] for row in rows: if merged and row[3] == merged[-1][3] and row[0] <= merged[-1][1] + 0.001: prev = merged[-1] merged[-1] = (prev[0], max(prev[1], row[1]), prev[2], prev[3]) else: merged.append(row) blocks = [] for index, (s, e, text, _) in enumerate(merged, 1): start_stamp = srt_stamp(s) end_stamp = srt_stamp(e) # Millisecond SRT precision can collapse a sub-millisecond span; # omit it rather than emit an invalid zero-duration cue. if start_stamp == end_stamp: continue blocks.append(f"{index}\n{start_stamp} --> {end_stamp}\n{text}\n") if not blocks: continue out = ( Path(output_dir).expanduser() / f"{Path(mp).stem}_captions.srt" if output_dir else Path(mp).with_name(Path(mp).stem + "_captions.srt") ) if output_dir: out.parent.mkdir(parents=True, exist_ok=True) try: out.write_text("\n".join(blocks), encoding="utf-8") except OSError as exc: shared.emit({"ok": False, "error": f"Não foi possível salvar a legenda: {exc}"}) return 1 srt_paths.append(str(out)) if not srt_paths: shared.emit({"ok": False, "error": "Nenhuma transcrição encontrada. Transcreva o projeto primeiro."}) return 1 shared.emit({"ok": True, "paths": srt_paths, "message": f"{len(srt_paths)} legenda(s) .srt sincronizada(s) com o corte."}) return 0 def srt_stamp(seconds: float) -> str: """Format float seconds as ``HH:MM:SS,mmm`` (SRT uses a comma). Uses ``floor`` (not ``round``) so a timestamp never rounds up past a frame boundary — an SRT cue ending on the last frame must not overrun the project duration, or Final Cut flags it as extending beyond the project. """ ms = int((seconds if seconds > 0 else 0.0) * 1000) h, rem = divmod(ms, 3600000) m, rem = divmod(rem, 60000) s, ms = divmod(rem, 1000) return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}" def cmd_dynamic_subtitle_config(args: dict) -> int: """Read the persisted dynamic-subtitle style (font, size, color, layout).""" shared.emit({"ok": True, **load_dynamic_subtitle_config()}) return 0 def cmd_set_dynamic_subtitle_config(args: dict) -> int: """Persist dynamic-subtitle style fields. Only the given fields change.""" config = save_dynamic_subtitle_config(**{ k: args.get(k) for k in ( "band_height", "block_center_y", "line_gap", "font", "font_size", "emphasis_font", "emphasis_face", "emphasis_size", "active_color", "emphasis_color", "text_scale", ) }) shared.emit({"ok": True, **config}) return 0 def cmd_plain_subtitle_config(args: dict) -> int: """Read the persisted simple subtitle style.""" shared.emit({"ok": True, **load_plain_subtitle_config()}) return 0 def cmd_set_plain_subtitle_config(args: dict) -> int: """Persist simple subtitle style fields. Only the given fields change.""" config = save_plain_subtitle_config(**{ k: args.get(k) for k in ( "font", "font_size", "font_color", "max_words", "position_y", "uppercase", "keep_punctuation", "text_scale", ) }) shared.emit({"ok": True, **config}) return 0