refactor: _shared.py vira subpacote, um módulo por papel
Eram 882 linhas de seis papéis sem relação, sob um nome que só dizia
"compartilhado" — o depósito onde tudo que servia a mais de um handler
acabava caindo.
media 316 transcrição em cache, corte por fala, relatório
paths 206 sandbox, limites, caminho de saída
project 116 abrir projeto, preparar modifier/generator
captions 112 SRT, VTT, listas com timestamp
detection 99 flash frames, buracos, duplicados
formatting 86 tabelas e relatórios dos handlers
O __init__ reexporta os 46 nomes, então os treze pontos que importam daqui
não mudaram.
_transcript_cut_report saiu de formatting para media: ele precisa do hint de
instalação e do _text_result, ou seja, é relatório de transcrição e não
formatação genérica — mover foi mais honesto que cruzar imports entre os
dois módulos.
Quatro testes patchavam `server_tools._shared.transcribe`; o nome agora é
ligado por _shared/media.py, então o patch passou a apontar para lá — mesmo
padrão da experiência #23.
Lint zerado, 1454 testes passando.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
368bb62706
commit
ffaebb3f72
@@ -0,0 +1,117 @@
|
||||
"""Leitura de legendas e listas com timestamp (SRT, VTT, texto colado).
|
||||
|
||||
Extraído de _shared.py — ver server_tools/_shared/__init__.py.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
|
||||
def _parse_timestamp_parts(
|
||||
parts: list[str], *, frame_rate: float = 24.0
|
||||
) -> float | None:
|
||||
"""Convert colon-separated timestamp parts to total seconds.
|
||||
|
||||
Handles 2-part (M:SS), 3-part (H:MM:SS / HH:MM:SS.ms), and
|
||||
4-part (HH:MM:SS:FF SMPTE) formats. Returns ``None`` when the
|
||||
part count is unrecognised so callers can skip.
|
||||
|
||||
Args:
|
||||
parts: Colon-split timestamp components.
|
||||
frame_rate: FPS used to convert the frame component of SMPTE
|
||||
timecodes into fractional seconds (default 24.0).
|
||||
"""
|
||||
if len(parts) == 2:
|
||||
return int(parts[0]) * 60 + float(parts[1])
|
||||
elif len(parts) == 3:
|
||||
return int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2])
|
||||
elif len(parts) == 4:
|
||||
# SMPTE: HH:MM:SS:FF — convert frames to fractional seconds
|
||||
base = int(parts[0]) * 3600 + int(parts[1]) * 60 + float(parts[2])
|
||||
frames = int(parts[3])
|
||||
return base + (frames / frame_rate) if frame_rate > 0 else base
|
||||
return None
|
||||
|
||||
def _raw_markers_to_batch(
|
||||
raw_markers: list[dict],
|
||||
marker_type: str = "chapter",
|
||||
max_label: int | None = None,
|
||||
) -> list[dict]:
|
||||
"""Convert raw {seconds, text} marker dicts to batch_add_markers format.
|
||||
|
||||
Shared by import_srt_markers and import_transcript_markers.
|
||||
"""
|
||||
batch = []
|
||||
for m in raw_markers:
|
||||
label = m["text"]
|
||||
if max_label and len(label) > max_label:
|
||||
label = label[:max_label]
|
||||
batch.append({
|
||||
"timecode": f"{m['seconds']}s",
|
||||
"name": label,
|
||||
"marker_type": marker_type.upper(),
|
||||
})
|
||||
return batch
|
||||
|
||||
def _extract_subtitle_blocks(text: str, *, strip_vtt_tags: bool = False) -> list[dict]:
|
||||
"""Extract timestamp/text pairs from subtitle cue blocks (SRT or VTT).
|
||||
|
||||
Both SRT and VTT use the same ``start --> end`` cue syntax with
|
||||
text lines underneath; only header stripping and tag cleaning differ.
|
||||
"""
|
||||
markers = []
|
||||
blocks = re.split(r'\n\s*\n', text.strip())
|
||||
for block in blocks:
|
||||
lines = block.strip().split('\n')
|
||||
if len(lines) < 2:
|
||||
continue
|
||||
ts_line = None
|
||||
text_lines = []
|
||||
for line in lines:
|
||||
if '-->' in line:
|
||||
ts_line = line
|
||||
elif ts_line is not None:
|
||||
if strip_vtt_tags:
|
||||
line = re.sub(r'<[^>]+>', '', line)
|
||||
cleaned = line.strip()
|
||||
if cleaned:
|
||||
text_lines.append(cleaned)
|
||||
if not ts_line or not text_lines:
|
||||
continue
|
||||
start_str = ts_line.split('-->')[0].strip().replace(',', '.')
|
||||
seconds = _parse_timestamp_parts(start_str.split(':'))
|
||||
if seconds is not None:
|
||||
markers.append({'seconds': seconds, 'text': ' '.join(text_lines)})
|
||||
return markers
|
||||
|
||||
def parse_srt(text: str) -> list[dict]:
|
||||
"""Parse SRT subtitle format into timestamp/text pairs."""
|
||||
return _extract_subtitle_blocks(text)
|
||||
|
||||
def parse_vtt(text: str) -> list[dict]:
|
||||
"""Parse WebVTT subtitle format into timestamp/text pairs."""
|
||||
text = re.sub(r'^WEBVTT.*?\n', '', text, flags=re.MULTILINE)
|
||||
text = re.sub(r'NOTE\n.*?\n\n', '', text, flags=re.DOTALL)
|
||||
return _extract_subtitle_blocks(text, strip_vtt_tags=True)
|
||||
|
||||
def parse_transcript_timestamps(text: str) -> list[dict]:
|
||||
"""Parse timestamped text (YouTube description format) into markers.
|
||||
|
||||
Supports formats like:
|
||||
0:00 Introduction
|
||||
00:01:30 Main Topic
|
||||
1:05:30 Conclusion
|
||||
00:00:00:00 SMPTE timecode
|
||||
"""
|
||||
markers = []
|
||||
for line in text.strip().split('\n'):
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
match = re.match(r'^(\d{1,2}:\d{2}(?::\d{2}){0,2})\s+(.+)$', line)
|
||||
if match:
|
||||
seconds = _parse_timestamp_parts(match.group(1).split(':'))
|
||||
if seconds is not None:
|
||||
markers.append({'seconds': seconds, 'text': match.group(2).strip()})
|
||||
return markers
|
||||
Reference in New Issue
Block a user