"""Leitura de legendas e listas com timestamp (SRT, VTT, texto colado). Extraído de _shared.py — ver server_tools/_shared/__init__.py. """ from __future__ import annotations import re def _parse_timestamp_parts( parts: list[str], *, frame_rate: float = 24.0 ) -> float | None: """Convert colon-separated timestamp parts to total seconds. Handles 2-part (M:SS), 3-part (H:MM:SS / HH:MM:SS.ms), and 4-part (HH:MM:SS:FF SMPTE) formats. Returns ``None`` when the part count is unrecognised so callers can skip. Args: parts: Colon-split timestamp components. frame_rate: FPS used to convert the frame component of SMPTE timecodes into fractional seconds (default 24.0). """ if len(parts) == 2: return int(parts[0]) * 60 + float(parts[1]) elif len(parts) == 3: return int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2]) elif len(parts) == 4: # SMPTE: HH:MM:SS:FF — convert frames to fractional seconds base = int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2]) frames = int(parts[3]) return base + (frames / frame_rate) if frame_rate > 0 else base return None def _raw_markers_to_batch( raw_markers: list[dict], marker_type: str = "chapter", max_label: int | None = None, ) -> list[dict]: """Convert raw {seconds, text} marker dicts to batch_add_markers format. Shared by import_srt_markers and import_transcript_markers. """ batch = [] for m in raw_markers: label = m["text"] if max_label and len(label) > max_label: label = label[:max_label] batch.append({ "timecode": f"{m['seconds']}s", "name": label, "marker_type": marker_type.upper(), }) return batch def _extract_subtitle_blocks(text: str, *, strip_vtt_tags: bool = False) -> list[dict]: """Extract timestamp/text pairs from subtitle cue blocks (SRT or VTT). Both SRT and VTT use the same ``start --> end`` cue syntax with text lines underneath; only header stripping and tag cleaning differ. """ markers = [] blocks = re.split(r'\n\s*\n', text.strip()) for block in blocks: lines = block.strip().split('\n') if len(lines) < 2: continue ts_line = None text_lines = [] for line in lines: if '-->' in line: ts_line = line elif ts_line is not None: if strip_vtt_tags: line = re.sub(r'<[^>]+>', '', line) cleaned = line.strip() if cleaned: text_lines.append(cleaned) if not ts_line or not text_lines: continue start_str = ts_line.split('-->')[0].strip().replace(',', '.') seconds = _parse_timestamp_parts(start_str.split(':')) if seconds is not None: markers.append({'seconds': seconds, 'text': ' '.join(text_lines)}) return markers def parse_srt(text: str) -> list[dict]: """Parse SRT subtitle format into timestamp/text pairs.""" return _extract_subtitle_blocks(text) def parse_vtt(text: str) -> list[dict]: """Parse WebVTT subtitle format into timestamp/text pairs.""" text = re.sub(r'^WEBVTT.*?\n', '', text, flags=re.MULTILINE) text = re.sub(r'NOTE\n.*?\n\n', '', text, flags=re.DOTALL) return _extract_subtitle_blocks(text, strip_vtt_tags=True) def parse_transcript_timestamps(text: str) -> list[dict]: """Parse timestamped text (YouTube description format) into markers. Supports formats like: 0:00 Introduction 00:01:30 Main Topic 1:05:30 Conclusion 00:00:00:00 SMPTE timecode """ markers = [] for line in text.strip().split('\n'): line = line.strip() if not line: continue match = re.match(r'^(\d{1,2}:\d{2}(?::\d{2}){0,2})\s+(.+)$', line) if match: seconds = _parse_timestamp_parts(match.group(1).split(':')) if seconds is not None: markers.append({'seconds': seconds, 'text': match.group(2).strip()}) return markers