feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão

Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha
alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a
etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado.

- generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro,
  legenda dinâmica só nas frases de ênfase, e a comum é desativada
  (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali.
- validate_subtitle_layout ignora títulos com enabled="0" — corrige falso
  positivo de colisão contra o que está desativado no lugar dele.
- Corrige zoom/marcador sendo descartado quando a borda encosta exatamente
  no início de um corte.
- Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com
  fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia
  entre "ativa" na tela e o que já foi cortado no FCPXML.
- Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json)
  antes da cadeia de remoção de silêncio/legendas — antes, desativar uma
  frase na etapa 5 não tinha efeito nenhum no vídeo final.
- Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder
  aparece assim que termina, sem slide extra.
- Palavra clicável na etapa 5 agora funciona como toggle (clique de novo
  desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte).
- fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento
  fonético via whisperx e roteirização local via Ollama/Gemma.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-08-21 18:26:04 -04:00
co-authored by Claude Sonnet 5
parent 711c397dfe
commit 7b5aed79ee
36 changed files with 2922 additions and 624 deletions
+181
View File
@@ -0,0 +1,181 @@
"""Forced alignment — refine word timestamps against an acoustic model.
Why this exists
--------------
faster-whisper derives word times by cross-attention, which lands every word
*start* systematically ~0.3-0.5s early (the word-end is fine). That bias flows
straight into the voice timeline and makes zoom/cut land on the wrong frame —
measured on real footage in ``Engine/docs/05_EXPERIENCIAS.md`` (#14). Phonetic
forced alignment (wav2vec2, via whisperx) re-anchors each word against the
audio and brings that error down to ~30ms.
Design
------
* The dependency (``whisperx``) is **optional** and imported lazily, exactly
like the rest of this stack (librosa, faster-whisper). When it is missing, or
any step fails, :meth:`ForcedAligner.align` returns the words unchanged, so
transcription never breaks because alignment did.
* The aligner is a single responsibility class: it knows how to turn a
transcript into the shape whisperx wants, call it, and write the refined
times back. ``transcribe.py`` owns the decision of *whether* to align.
* Align models are cached per language on the instance so repeated calls
(e.g. many short clips) don't reload the wav2vec2 weights each time.
"""
import logging
from typing import List, Optional, Sequence
logger = logging.getLogger(__name__)
class ForcedAligner:
"""Refine word-level timestamps with whisperx phonetic forced alignment.
Usage::
aligner = ForcedAligner()
words = aligner.align(words, raw_segments, media_path, language, models_dir)
``words`` and ``raw_segments`` come straight from :func:`transcribe` —
``raw_segments`` carries the per-segment ``words`` lists (the same dict
objects as in ``words``) so the aligner knows which words belong to which
audio window. Returns a list of the *same* word dicts, with ``start``/``end``
overwritten in place where alignment produced a usable time.
"""
def __init__(self, device: Optional[str] = None):
self._device = device
self._models: dict = {}
# -- capability ------------------------------------------------------
@staticmethod
def available() -> bool:
"""Whether whisperx can be imported (the aligner can run at all)."""
try:
import whisperx # noqa: F401
except Exception:
return False
return True
def _resolve_device(self) -> str:
if self._device:
return self._device
try:
import torch
if torch.cuda.is_available():
return "cuda"
except Exception:
pass
return "cpu"
# -- public API ------------------------------------------------------
def align(
self,
words: Sequence[dict],
raw_segments: Sequence[dict],
audio_path: str,
language: str,
models_dir: Optional[str] = None,
) -> List[dict]:
"""Return ``words`` with forced-aligned timestamps where possible.
Falls back to the unchanged ``words`` on any failure (missing
dependency, model load error, audio read error, or a result that
doesn't line up with the input).
"""
if not words or not language:
return list(words)
try:
import whisperx
except Exception:
logger.info("whisperx not installed; skipping forced alignment")
return list(words)
try:
device = self._resolve_device()
align_input = self._build_align_input(words, raw_segments)
audio = whisperx.load_audio(audio_path)
if language not in self._models:
align_model, metadata = whisperx.load_align_model(
language_code=language,
device=device,
model_dir=str(models_dir) if models_dir else None,
)
self._models[language] = (align_model, metadata)
align_model, metadata = self._models[language]
result = whisperx.align(
align_input,
align_model,
metadata,
audio,
device,
return_char_alignments=False,
)
return self._merge_result(words, result.get("segments", []))
except Exception:
logger.warning(
"forced alignment failed for %s; using raw timestamps", audio_path
)
return list(words)
# -- internals -------------------------------------------------------
@staticmethod
def _build_align_input(
words: Sequence[dict], raw_segments: Sequence[dict]
) -> List[dict]:
"""Transcript in whisperx's expected shape: segments -> words.
whisperx.align requires each segment to carry ``text``/``start``/``end``
and a ``words`` list whose entries have ``word``/``start``/``end``/``score``.
We only read ``words`` from ``raw_segments`` (the flattened ``words``
list is the source of truth for counts), so the two stay consistent.
"""
align_segments: List[dict] = []
for seg in raw_segments:
seg_words = [
{
"word": w.get("word", ""),
"start": float(w.get("start", 0.0)),
"end": float(w.get("end", 0.0)),
"score": float(w.get("confidence", 0.0)),
}
for w in seg.get("words", [])
]
align_segments.append(
{
"text": (seg.get("text") or "").strip(),
"start": float(seg.get("start", 0.0)),
"end": float(seg.get("end", 0.0)),
"words": seg_words,
}
)
return align_segments
@staticmethod
def _merge_result(words: Sequence[dict], aligned_segments: Sequence[dict]) -> List[dict]:
"""Walk the aligned output in order and overwrite word times in place.
whisperx preserves word order within and across segments, so a single
running index over the output words lines up with ``words``. A word the
aligner failed to place gets ``None``/``0`` times — we skip those rather
than clobber a good timestamp, and if counts ever diverge we stop and
leave the rest untouched.
"""
out = list(words)
wi = 0
for seg in aligned_segments:
for aw in seg.get("words", []):
if wi >= len(out):
return out
start = aw.get("start")
end = aw.get("end")
if start is None or end is None or end < start:
wi += 1
continue
out[wi]["start"] = float(start)
out[wi]["end"] = float(end)
wi += 1
return out
+312
View File
@@ -0,0 +1,312 @@
"""Local LLM integration — the voice timeline meets a local model.
The voice timeline is *designed* to be handed to a language model: it is the
source of truth between speech analysis and editing, layered so a model can
reason about the narrative without parsing FCPXML. This module is the client
side of that contract. It formats the timeline into the editar-por-voz brief,
calls a local model server (Ollama, running Gemma 3 / Llama locally), and
parses the model's decisions back into a validated list of VoiceActions —
all inside the engine, so there is no wizard, no copy-paste, no manual step.
Transport: Ollama's HTTP chat API at ``http://localhost:11434/api/chat``.
Any model Ollama serves works; the default is Gemma 3 because that is what
runs locally here ("Lama com Gema 3"), but pass ``model=`` to switch.
The model is untrusted input: its JSON is validated row-by-row by
:func:`fcpxml.voice_actions.parse_actions`, so one malformed decision never
discards the edit. The brief is written so the model only ever emits the four
action kinds the applier understands.
"""
from __future__ import annotations
import json
import logging
import re
from typing import Any, Dict, Optional, Sequence, Tuple
import httpx
from .voice_actions import parse_actions
logger = logging.getLogger(__name__)
DEFAULT_BASE_URL = "http://localhost:11434"
# Gemma 3 12B reliably follows the editar-por-voz brief (keep the script, cut
# only backstage chatter; the 4B variant skips the "keep the main content"
# rule and deletes the script) but doesn't fit an 8GB machine. Qwen2.5 7B
# instruct (q4_K_M) is the fallback for constrained hardware — strong at
# strict JSON-schema following, the property this brief leans on hardest.
# Pass ``model=`` to switch to whatever Ollama serves.
DEFAULT_MODEL = "qwen2.5:7b-instruct-q4_K_M"
REQUEST_TIMEOUT = 600.0
# The brief. Ported from the editar-por-voz skill criteria (criterios/01..08),
# condensed into the instructions a model needs to emit valid actions. Kept in
# Portuguese because the decisions and their reasons are read by a human editor.
_SYSTEM_PROMPT = """Você é o editor de vídeo por voz deste sistema. Recebe um JSON de "linha do tempo de voz" — a medição de COMO foi falado (ênfase, energia, pausa, falante) de uma gravação — e devolve as DECISÕES de edição em JSON, nada mais. Você nunca escreve XML.
Regras (siga rigorosamente):
1. LEIA EM CAMADAS. "summary" dá o formato da peça; "segments" é onde você trabalha (cada fala com seu texto e agregados); "segments[].words" dá o instante exato de cada destaque. Não recalcule energia, tom ou ênfase — use os números do JSON.
2. SEPARAR ROTEIRO DE BASTIDOR.
- ROTEIRO = o conteúdo principal que a pessoa quer entregar: explicação, depoimento, roteiro decorado, a mensagem. É isso que VAI FICAR.
- BASTIDOR = papo casual de gravação, cumprimentos, conversa com a equipe ("cara, beleza?", "tá gravando?", "deixa eu ver o celular"), piadas fora do assunto, tomadas interrompidas ou repetidas. É isso que VIRA "cut".
Exemplo: num vídeo sobre mastopexia, a explicação da cirurgia É o roteiro (mantém); o "tá gravando? pois é" antes dela É bastidor (corta).
Use "gap_before" e "take_boundary" (silêncio > ~3s = a câmera parou/recomeçou) para agrupar tomadas — eles marcam ONDE a tomada recomeça, não o que cortar. Nunca corte o conteúdo principal só porque tem ênfase; corte o casual/off-topic.
REGRAS DE OURO:
- MANTENHA o conteúdo principal (explicação, depoimento, roteiro decorado). Ele É o vídeo.
- CORTE SÓ o casual/off-topic: cumprimentos, "tá gravando?", papo com a equipe, olhar o celular, repetições de tomada.
- Em dúvida, MANTENHA a fala. É melhor sobrar conteúdo do que cortar o que era pra ficar.
3. ESCOLHER A MELHOR TOMADA de cada frase quando há repetições: mantenha a mais limpa e corte as outras (cut cobrindo a frase inteira).
4. CORTE (kind "cut"): para REMOVER uma frase, cubra ela inteira (start..end = início..fim da frase). Para APARAR só uma hesitação no começo ou fim, corte só da borda até a palavra (corte de meia frase é ambíguo — passe de 60% e apaga a linha toda). Nunca corte o silêncio entre falas.
5. ZOOM (kind "zoom"): só em palavra de CONTEÚDO bem enfatizada (emphasis alto, não artigo). params.scale entre 1.0 e 3.0 (padrão 1.3 se omitido). Posicione em torno da palavra, segurando até o fim da frase.
6. TEXTO (kind "text"): params.content obrigatório (≤120 chars), fixa um termo central ou callout. MARKER (kind "marker"): opcional params.content vira o nome do marcador. Use para emendas/junções que o editor deve conferir.
7. TEMPOS em segundos da MÍDIA ORIGINAL (exatamente como no JSON). Nunca compense para "depois do corte" — o programa desloca sozinho. end sempre > start, ambos ≥ 0.
8. reason OBRIGATÓRIO em cada ação, em português, embasando a decisão (ex.: 'abertura: "Aquela mama" (ênfase 0.42)'). reason vazio é decisão sem critério.
Responda APENAS com um objeto JSON válido, sem markdown, sem comentário:
{"source": "<nome do arquivo>", "actions": [{"kind": "cut|zoom|text|marker", "start": <float>, "end": <float>, "params": {}, "reason": "<pt>", "speaker": "<id>"}]}
"""
_OUTPUT_REMINDER = """Gere as decisões de edição conforme o brief. Responda SOMENTE o JSON:
{"source": "<nome do arquivo>", "actions": [{"kind": "cut|zoom|text|marker", "start": <float>, "end": <float>, "params": {}, "reason": "<pt>", "speaker": "<id>"}]}
Não inclua explicações nem blocos markdown."""
def ollama_chat(
model: str = DEFAULT_MODEL,
messages: Optional[Sequence[Dict[str, str]]] = None,
base_url: str = DEFAULT_BASE_URL,
temperature: float = 0.2,
timeout: float = REQUEST_TIMEOUT,
num_ctx: int = 32768,
) -> str:
"""One chat completion from a local Ollama server.
Returns the assistant message content. Raises on transport/HTTP errors so
the caller can decide whether to retry or report — a model call is the
one I/O in this pipeline that can legitimately fail mid-run.
"""
payload = {
"model": model,
"messages": list(messages or []),
"stream": False,
"options": {"temperature": temperature, "num_ctx": num_ctx},
}
try:
response = httpx.post(
f"{base_url.rstrip('/')}/api/chat", json=payload, timeout=timeout
)
response.raise_for_status()
data = response.json()
except Exception as exc:
# Covers transport errors AND a dropped connection that yields an empty
# body (httpx/JSONDecodeError) — both must become a RuntimeError so the
# caller reports the failure instead of crashing the whole pipeline.
raise RuntimeError(f"Falha ao falar com o modelo local em {base_url}: {exc}") from exc
return (data.get("message") or {}).get("content", "") or ""
def list_ollama_models(base_url: str = DEFAULT_BASE_URL) -> list[str]:
"""Names of the models Ollama currently serves, for a model picker.
Returns an empty list when Ollama is unreachable so the UI can fall back to
a free-text field instead of erroring.
"""
try:
resp = httpx.get(f"{base_url.rstrip('/')}/api/tags", timeout=10.0)
resp.raise_for_status()
models = resp.json().get("models", [])
names = [m.get("name") for m in models if m.get("name")]
return sorted(names)
except Exception:
return []
def _extract_json(text: str) -> Any:
"""Pull a JSON value out of a model response, tolerating fences/wrappers."""
if not text:
return None
candidate = text.strip()
# Strip a ```json ... ``` (or bare ```) fence if the model added one.
fence = re.search(r"```(?:json)?\s*(.*?)\s*```", candidate, re.DOTALL)
if fence:
candidate = fence.group(1).strip()
# Otherwise take the outermost {...} / [...].
if not candidate.startswith(("{" if True else "", "[")):
start = min(
(i for i, c in enumerate(candidate) if c in "{["),
default=None,
)
end = max(
(i for i, c in enumerate(candidate) if c in "}"),
default=None,
)
if start is not None and end is not None and end > start:
candidate = candidate[start : end + 1]
try:
data = json.loads(candidate)
except json.JSONDecodeError:
return None
# Models sometimes wrap the expected `{"source", "actions"}` object inside a
# single-element list (`[{...}]`). Unwrap that so the actions aren't treated
# as one malformed row.
if (
isinstance(data, list)
and len(data) == 1
and isinstance(data[0], dict)
and "actions" in data[0] # the wrapper carries the actions key
):
data = data[0]
return data
# Only these fields reach the model — the raw timeline also carries heavy
# per-word audio features (energy, pitch, arousal...) and speaker `samples`
# that blow past the model's context window on any real recording. Dropping
# them is what keeps a 3-minute timeline inside `num_ctx`.
_SEGMENT_KEEP = (
"start", "end", "speaker", "text", "gap_before", "take_boundary",
"avg_energy", "peak_emphasis", "emotion", "emotion_confidence",
"arousal", "valence",
)
_WORD_KEEP = ("text", "start", "end", "speaker", "emphasis", "pause_before")
_SPEAKER_KEEP = ("id", "name")
_SKIP_ROOT = ("layers", "scales")
def _project_timeline(timeline: dict) -> dict:
"""Strip the timeline down to what the edit decision actually needs."""
out = {k: v for k, v in timeline.items() if k not in _SKIP_ROOT}
speakers = [
{k: sp[k] for k in _SPEAKER_KEEP if k in sp}
for sp in timeline.get("speakers", [])
]
if speakers:
out["speakers"] = speakers
segs = []
for seg in timeline.get("segments", []):
s = {k: seg[k] for k in _SEGMENT_KEEP if k in seg}
s["words"] = [
{k: w[k] for k in _WORD_KEEP if k in w}
for w in seg.get("words", [])
]
segs.append(s)
out["segments"] = segs
return out
def _shrink_to_fit(compact: dict, max_chars: int) -> dict:
"""Drop word detail from the lowest-emphasis segments until it fits."""
segs = [dict(s) for s in compact.get("segments", [])]
while True:
payload = json.dumps(
{**compact, "segments": segs}, ensure_ascii=False, indent=1
)
if len(payload) <= max_chars or not any(s.get("words") for s in segs):
break
idx = min(
(i for i, s in enumerate(segs) if s.get("words")),
key=lambda i: float(segs[i].get("peak_emphasis", 0.0)),
)
segs[idx] = {**segs[idx], "words": []}
compact = dict(compact)
compact["segments"] = segs
return compact
def build_edit_messages(
timeline: dict, max_words_per_segment: int = 200, max_chars: int = 110000
) -> Tuple[str, str]:
"""The (system, user) pair that sends a timeline to the model.
The user turn carries a *projected* timeline (see :func:`_project_timeline`)
— text, timing, speaker and emphasis only — so a real recording fits in the
model's context window. Very long segments still have their word detail
capped to ``max_words_per_segment`` (most emphatic + boundaries), and if the
whole payload would still exceed ``max_chars`` the lowest-emphasis segments
lose their words until it fits, so we never blow ``num_ctx``.
"""
compact = _project_timeline(timeline)
if max_words_per_segment:
segs = []
for seg in compact["segments"]:
words = seg.get("words", [])
if len(words) > max_words_per_segment:
ranked = sorted(
enumerate(words),
key=lambda kv: float(kv[1].get("emphasis", 0.0)),
reverse=True,
)[: max_words_per_segment - 2]
keep = sorted({0, len(words) - 1} | {i for i, _ in ranked})
seg = {**seg, "words": [words[i] for i in keep]}
segs.append(seg)
compact["segments"] = segs
payload = json.dumps(compact, ensure_ascii=False, indent=1)
if len(payload) > max_chars:
compact = _shrink_to_fit(compact, max_chars)
payload = json.dumps(compact, ensure_ascii=False, indent=1)
user = (
"Linha do tempo de voz (JSON):\n\n"
+ payload
+ "\n\n"
+ _OUTPUT_REMINDER
)
return _SYSTEM_PROMPT, user
def generate_voice_actions(
timeline: dict,
model: str = DEFAULT_MODEL,
base_url: str = DEFAULT_BASE_URL,
temperature: float = 0.2,
timeout: float = REQUEST_TIMEOUT,
num_ctx: int = 32768,
max_words_per_segment: int = 200,
) -> Dict[str, Any]:
"""Ask the local model to direct the edit, returning validated actions.
Returns ``{"actions": [VoiceAction], "raw": str, "errors": [str]}``.
``actions`` is empty when the model returned nothing usable; ``errors``
carries the per-row rejections from :func:`parse_actions` plus any
extraction failure, so the caller can report what went wrong instead of
only the wins.
"""
system, user = build_edit_messages(timeline, max_words_per_segment)
try:
raw = ollama_chat(
model=model,
messages=[
{"role": "system", "content": system},
{"role": "user", "content": user},
],
base_url=base_url,
temperature=temperature,
timeout=timeout,
num_ctx=num_ctx,
)
except RuntimeError as exc:
return {"actions": [], "raw": "", "errors": [str(exc)]}
data = _extract_json(raw)
if data is None:
return {
"actions": [],
"raw": raw,
"errors": ["O modelo não devolveu um JSON de decisões legível."],
}
actions, errors = parse_actions(data)
return {"actions": actions, "raw": raw, "errors": errors}
+15
View File
@@ -194,6 +194,7 @@ class FCPXMLParser:
media_path=media_path,
audio_role=elem.get('audioRole', ''),
video_role=elem.get('videoRole', ''),
rotation=self._parse_clip_rotation(elem),
)
clip.markers.extend(self._collect_markers(elem))
@@ -205,6 +206,19 @@ class FCPXMLParser:
return clip
def _parse_clip_rotation(self, elem: ET.Element) -> float:
"""Degrees from this clip's ``<adjust-transform rotation="...">`` —
an edit-time correction (e.g. straightening a tilted phone shot),
not the camera's own recorded orientation. FCP writes the rotation
as an attribute on that element, not as a filter param."""
transform = elem.find('adjust-transform')
if transform is None:
return 0.0
try:
return float(transform.get('rotation', '0'))
except ValueError:
return 0.0
def _parse_marker_element(self, elem: ET.Element) -> Optional[Marker]:
"""Parse any marker element (<marker> or <chapter-marker>).
@@ -337,6 +351,7 @@ class FCPXMLParser:
lane=lane, offset=offset, source_start=start,
media_path=media_path, clip_type=elem.tag, role=role,
ref_id=ref, parent_clip_name=parent_name,
rotation=self._parse_clip_rotation(elem),
)
connected.markers.extend(self._collect_markers(elem))
+1
View File
@@ -300,6 +300,7 @@ def build_phrase_review(
"version": PHRASE_REVIEW_VERSION,
"source": source,
"source_path": resolve_source(source, voice_timeline_path, extra_dirs),
"rotation": float(timeline.get("rotation", 0.0)),
"duration": round(phrases[-1]["end"], 3) if phrases else 0.0,
"speakers": timeline.get("speakers", []),
"emotion_available": bool(layers.get("emotion", False)),
+44 -12
View File
@@ -125,18 +125,27 @@ def transcribe(
model_size: str = "base",
language: Optional[str] = None,
progress_cb: Optional[Callable[[float], None]] = None,
align: bool = True,
) -> Optional[dict]:
"""Transcribe an audio/video file locally with word-level timestamps.
Requires the optional ``[transcribe]`` extra (faster-whisper). Returns
``None`` when the model is unavailable or the file is missing/unreadable.
When ``align`` is true (default) and the optional ``whisperx`` dependency is
present, word timestamps are refined by phonetic forced alignment, which
corrects faster-whisper's systematic ~0.3-0.5s early bias on word *starts*
(see ``Engine/docs/05_EXPERIENCIAS.md`` #14). The transcript reports
whether this ran via the ``alignment`` flag, so downstream consumers can
rely on the times without re-measuring.
The model weights are resolved from the configured models directory (see
``model_manager.get_models_dir``), so a model selected/downloaded through
the app is found without an implicit download to the default HF cache.
Returns:
``{"language": str, "duration": float, "text": str,
"alignment": bool,
"segments": [{"text", "start", "end", "start_fmt", "end_fmt"}, ...],
"words": [{"word", "start", "end", "confidence"}, ...]}``
"""
@@ -175,6 +184,7 @@ def transcribe(
vad_filter=True,
)
segments: List[dict] = []
raw_segments: List[dict] = []
words: List[dict] = []
# `info.duration` is known upfront (from the container), so each
# segment's end time — yielded lazily as faster-whisper decodes —
@@ -183,6 +193,20 @@ def transcribe(
for seg in segments_iter:
start = float(seg.start)
end = float(seg.end)
seg_words: List[dict] = []
if progress_cb is not None and total_duration > 0:
progress_cb(min(end / total_duration, 1.0))
for w in seg.words or []:
ws = float(w.start)
we = float(w.end)
word = {
"word": w.word.strip(),
"start": ws,
"end": we,
"confidence": float(w.probability),
}
words.append(word)
seg_words.append(word)
segments.append(
{
"text": seg.text.strip(),
@@ -192,19 +216,26 @@ def transcribe(
"end_fmt": format_timestamp(end),
}
)
if progress_cb is not None and total_duration > 0:
progress_cb(min(end / total_duration, 1.0))
for w in seg.words or []:
ws = float(w.start)
we = float(w.end)
words.append(
{
"word": w.word.strip(),
"start": ws,
"end": we,
"confidence": float(w.probability),
}
raw_segments.append(
{
"text": seg.text.strip(),
"start": start,
"end": end,
"words": seg_words,
}
)
alignment_ran = False
if align and raw_segments:
from .forced_align import ForcedAligner
try:
words = ForcedAligner().align(
words, raw_segments, str(file_path), info.language, str(models_dir)
)
alignment_ran = True
except Exception:
logger.warning("forced alignment step failed; keeping raw timestamps")
except Exception:
logger.warning("whisper transcription failed for %s", file_path)
return None
@@ -212,6 +243,7 @@ def transcribe(
"language": info.language,
"duration": float(info.duration),
"text": " ".join(s["text"] for s in segments),
"alignment": alignment_ran,
"segments": segments,
"words": words,
}
+6
View File
@@ -512,6 +512,7 @@ def build_voice_timeline(
emphasis_floor: float = 0.25,
emotion_enabled: bool = False,
emotion_sensitivity: float = 0.5,
rotation: float = 0.0,
progress_cb: Optional[Callable[[float, str], None]] = None,
) -> dict:
"""Build the consolidated voice timeline for one media file.
@@ -545,6 +546,10 @@ def build_voice_timeline(
return {
"version": VOICE_TIMELINE_VERSION,
"source": Path(media_path).name,
# Edit-time correction from the clip's Transform filter in the FCPXML
# (e.g. straightening a tilted phone shot) — 0.0 when the clip has none
# or the caller didn't resolve one.
"rotation": rotation,
"language": transcript.get("language", ""),
# What actually ran, not what was installed — a consumer must be able
# to tell "this speech is flat" from "the acoustics never loaded",
@@ -554,6 +559,7 @@ def build_voice_timeline(
"acoustics": pitch_track is not None or energy_track is not None,
"speakers": tracks is not None,
"emotion": bool(emotion_enabled),
"alignment": bool(transcript.get("alignment")),
},
"scales": VALUE_SCALES,
"summary": _summary(
+7
View File
@@ -550,6 +550,13 @@ class TitlesMixin:
"""
titles = []
for elem in self.root.iter('title'):
# enabled="0" never renders in Final Cut (see
# generate_subtitles_by_emphasis, which disables plain titles
# under an emphasis phrase instead of never creating them) — a
# title that is off by design must not count as a collision
# against the one drawn in its place.
if elem.get('enabled', '1') == '0':
continue
text_el = elem.find('text/text-style')
text = (text_el.text or '').strip() if text_el is not None else ''
style = elem.find('text-style-def/text-style')