Files
gart/code/fcpxml/transcribe.py
T
João HenriqueandClaude Sonnet 5 7b5aed79ee feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-21 18:26:04 -04:00

333 lines
12 KiB
Python
Executable File

"""Transcript intelligence — local Whisper transcription + text-driven editing.
v0.13 slice 1: transcript-based editing. Transcription runs locally via
faster-whisper (optional ``[transcribe]`` extra) with word-level timestamps;
without it, or when media is missing/unreadable, ``transcribe`` returns
``None`` so callers degrade to an install hint instead of crashing — the
same contract as ``media_intel.detect_beats``.
The matching helpers below are pure functions over word lists so they are
fully testable without any model installed, and so tools can accept
pre-computed transcripts (from a previous ``transcribe_media`` run) instead
of re-transcribing.
"""
import logging
import os
import re
from pathlib import Path
from typing import Callable, List, Optional, Sequence, Tuple
logger = logging.getLogger(__name__)
# Model names are used to resolve (and download) model weights, so they are
# validated against an allowlist, not trusted.
ALLOWED_MODELS = (
"tiny", "tiny.en", "base", "base.en", "small", "small.en",
"medium", "medium.en", "large-v2", "large-v3", "distil-large-v3",
)
# Conservative by default: interjections that are near-universally filler.
# Portuguese "um"/"uma" are usually articles/numerals inside real phrases
# ("de um jeito") rather than discardable hesitations, so only cut them when
# the caller explicitly opts in through the fillers argument.
# "like" / "so" / "actually" are speech, not noise, unless the user opts in.
DEFAULT_FILLERS = ("uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm")
_NORM_RE = re.compile(r"[^\w']+")
def normalize_word(word: str) -> str:
"""Lowercase a word and strip punctuation so matching survives Whisper's
tokenization quirks (leading spaces, trailing commas, case)."""
return _NORM_RE.sub("", word.lower())
def find_phrase_spans(
words: Sequence[dict], phrase: str
) -> List[Tuple[float, float]]:
"""Find every occurrence of ``phrase`` in a word-level transcript.
``words`` is a sequence of ``{"word", "start", "end"}`` dicts in source
seconds. Matching is case- and punctuation-insensitive. Returns
``(start, end)`` source-second ranges spanning first to last matched word.
"""
target = [normalize_word(w) for w in phrase.split()]
target = [t for t in target if t]
if not target:
return []
normed = [normalize_word(w.get("word", "")) for w in words]
spans: List[Tuple[float, float]] = []
i = 0
n, m = len(normed), len(target)
while i <= n - m:
if normed[i:i + m] == target:
spans.append((float(words[i]["start"]), float(words[i + m - 1]["end"])))
i += m
else:
i += 1
return spans
def find_filler_spans(
words: Sequence[dict], fillers: Sequence[str] = DEFAULT_FILLERS
) -> List[Tuple[float, float]]:
"""Find filler-word occurrences (single- or multi-word fillers)."""
spans: List[Tuple[float, float]] = []
for filler in fillers:
spans.extend(find_phrase_spans(words, filler))
spans.sort()
return spans
def merge_ranges(
ranges: Sequence[Tuple[float, float]], min_gap: float = 0.0
) -> List[Tuple[float, float]]:
"""Merge overlapping (or nearly touching, within ``min_gap``) ranges."""
if not ranges:
return []
ordered = sorted(ranges)
merged = [list(ordered[0])]
for start, end in ordered[1:]:
if start <= merged[-1][1] + min_gap:
merged[-1][1] = max(merged[-1][1], end)
else:
merged.append([start, end])
return [(s, e) for s, e in merged]
def invert_ranges(
ranges: Sequence[Tuple[float, float]], window_start: float, window_end: float
) -> List[Tuple[float, float]]:
"""Complement of ``ranges`` within ``[window_start, window_end]`` —
turns keep-ranges into cut-ranges for keep_only mode."""
if window_end <= window_start:
return []
kept = merge_ranges(
[(max(s, window_start), min(e, window_end)) for s, e in ranges
if min(e, window_end) > max(s, window_start)]
)
if not kept:
return [(window_start, window_end)]
out: List[Tuple[float, float]] = []
cursor = window_start
for start, end in kept:
if start > cursor:
out.append((cursor, start))
cursor = max(cursor, end)
if cursor < window_end:
out.append((cursor, window_end))
return out
def transcribe(
path: str,
model_size: str = "base",
language: Optional[str] = None,
progress_cb: Optional[Callable[[float], None]] = None,
align: bool = True,
) -> Optional[dict]:
"""Transcribe an audio/video file locally with word-level timestamps.
Requires the optional ``[transcribe]`` extra (faster-whisper). Returns
``None`` when the model is unavailable or the file is missing/unreadable.
When ``align`` is true (default) and the optional ``whisperx`` dependency is
present, word timestamps are refined by phonetic forced alignment, which
corrects faster-whisper's systematic ~0.3-0.5s early bias on word *starts*
(see ``Engine/docs/05_EXPERIENCIAS.md`` #14). The transcript reports
whether this ran via the ``alignment`` flag, so downstream consumers can
rely on the times without re-measuring.
The model weights are resolved from the configured models directory (see
``model_manager.get_models_dir``), so a model selected/downloaded through
the app is found without an implicit download to the default HF cache.
Returns:
``{"language": str, "duration": float, "text": str,
"alignment": bool,
"segments": [{"text", "start", "end", "start_fmt", "end_fmt"}, ...],
"words": [{"word", "start", "end", "confidence"}, ...]}``
"""
if model_size not in ALLOWED_MODELS:
raise ValueError(
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
)
file_path = Path(path)
if not file_path.is_file():
return None
try:
# Resolve the configured models root and point HF at it *before* the
# first faster_whisper import, so downloads/loads go to our folder.
from .model_manager import get_models_dir
models_dir = get_models_dir()
hf_home = models_dir / "hf_home"
hf_home.mkdir(parents=True, exist_ok=True)
os.environ["HF_HOME"] = str(hf_home)
os.environ["HUGGINGFACE_HUB_CACHE"] = str(hf_home / "hub")
from faster_whisper import WhisperModel
except ImportError:
logger.info("faster-whisper not installed; transcription unavailable")
return None
try:
model = WhisperModel(
model_size,
compute_type="int8",
download_root=str(models_dir),
)
segments_iter, info = model.transcribe(
str(file_path),
language=language,
word_timestamps=True,
vad_filter=True,
)
segments: List[dict] = []
raw_segments: List[dict] = []
words: List[dict] = []
# `info.duration` is known upfront (from the container), so each
# segment's end time — yielded lazily as faster-whisper decodes —
# gives real, granular progress instead of a single before/after step.
total_duration = float(info.duration) if info.duration else 0.0
for seg in segments_iter:
start = float(seg.start)
end = float(seg.end)
seg_words: List[dict] = []
if progress_cb is not None and total_duration > 0:
progress_cb(min(end / total_duration, 1.0))
for w in seg.words or []:
ws = float(w.start)
we = float(w.end)
word = {
"word": w.word.strip(),
"start": ws,
"end": we,
"confidence": float(w.probability),
}
words.append(word)
seg_words.append(word)
segments.append(
{
"text": seg.text.strip(),
"start": start,
"end": end,
"start_fmt": format_timestamp(start),
"end_fmt": format_timestamp(end),
}
)
raw_segments.append(
{
"text": seg.text.strip(),
"start": start,
"end": end,
"words": seg_words,
}
)
alignment_ran = False
if align and raw_segments:
from .forced_align import ForcedAligner
try:
words = ForcedAligner().align(
words, raw_segments, str(file_path), info.language, str(models_dir)
)
alignment_ran = True
except Exception:
logger.warning("forced alignment step failed; keeping raw timestamps")
except Exception:
logger.warning("whisper transcription failed for %s", file_path)
return None
return {
"language": info.language,
"duration": float(info.duration),
"text": " ".join(s["text"] for s in segments),
"alignment": alignment_ran,
"segments": segments,
"words": words,
}
def format_timestamp(seconds: float) -> str:
"""Format float seconds as ``HH:MM:SS.mmm`` (e.g. ``00:01:23.450``)."""
total_ms = max(0, round((seconds or 0.0) * 1000))
hours, rem_ms = divmod(total_ms, 3_600_000)
minutes, rem_ms = divmod(rem_ms, 60_000)
secs, ms = divmod(rem_ms, 1000)
return f"{hours:02d}:{minutes:02d}:{secs:02d}.{ms:03d}"
def group_words_by_segment(
words: Sequence[dict],
segments: Sequence[dict],
) -> List[List[dict]]:
"""Group flat *words* into sentences using *segments*' time windows.
``transcribe()`` returns ``segments`` and ``words`` as sibling flat lists —
the words are flattened out of the segments and the association is lost in
serialisation, leaving only the time ranges to rejoin them by. A word
belongs to the segment whose ``[start, end)`` contains its ``start``.
Words falling in no segment (rounding at a boundary, or a gap between
segments) attach to the group being built rather than being dropped —
losing a spoken word would silently drop it from the subtitles.
With no usable *segments*, every word comes back as a single group, which
the caller can still split on its own terms.
"""
kept = [
s for s in (segments or [])
if s.get('start') is not None and s.get('end') is not None
]
ordered = sorted(kept, key=lambda s: float(s['start']))
if not ordered:
return [list(words)] if words else []
groups: List[List[dict]] = []
current: List[dict] = []
current_index: Optional[int] = None
cursor = 0
for w in words:
start = float(w.get('start', 0.0))
# Segments and words are both chronological, so the search only ever
# moves forward.
while (
cursor + 1 < len(ordered)
and start >= float(ordered[cursor]['end'])
and start >= float(ordered[cursor + 1]['start'])
):
cursor += 1
seg = ordered[cursor]
inside = float(seg['start']) <= start < float(seg['end'])
index = cursor if inside else current_index
if current and index != current_index and inside:
groups.append(current)
current = []
if inside or current_index is None:
current_index = index if index is not None else cursor
current.append(w)
if current:
groups.append(current)
return groups
def segments_to_srt(segments: Sequence[dict]) -> str:
"""Render transcript segments as an SRT string (for captions import)."""
def stamp(seconds: float) -> str:
ms = int(round(seconds * 1000))
h, rem = divmod(ms, 3600000)
m, rem = divmod(rem, 60000)
s, ms = divmod(rem, 1000)
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
blocks = []
for i, seg in enumerate(segments, 1):
blocks.append(f"{i}\n{stamp(seg['start'])} --> {stamp(seg['end'])}\n{seg['text']}\n")
return "\n".join(blocks)