feat(voz): legenda por ênfase, forced align, IA local e correções de zoom/revisão
Trabalho da branch feat/revisao-enfases: pipeline de edição por voz ganha alinhamento forçado (whisperx), roteirização por LLM local (Ollama), e a etapa 5 (revisão de frases) passa a refletir de verdade o que é aplicado. - generate_subtitles_by_emphasis: legenda comum cobre o clipe inteiro, legenda dinâmica só nas frases de ênfase, e a comum é desativada (enabled="0") onde a dinâmica cobre, em vez de nunca ser gerada ali. - validate_subtitle_layout ignora títulos com enabled="0" — corrige falso positivo de colisão contra o que está desativado no lugar dele. - Corrige zoom/marcador sendo descartado quando a borda encosta exatamente no início de um corte. - Etapa 5 do Assistente: recarrega quando as decisões da IA mudam (com fresh=true, ignorando a revisão salva antiga) — resolve a dessincronia entre "ativa" na tela e o que já foi cortado no FCPXML. - Etapa "Processar" reaplica as decisões da revisão (_phrase_actions.json) antes da cadeia de remoção de silêncio/legendas — antes, desativar uma frase na etapa 5 não tinha efeito nenhum no vídeo final. - Etapa "Concluído" fundida em "Processar" — abrir no Final Cut/Finder aparece assim que termina, sem slide extra. - Palavra clicável na etapa 5 agora funciona como toggle (clique de novo desfaz) e mostra a própria ênfase (sublinhado colorido + peso da fonte). - fcpxml/forced_align.py, fcpxml/llm_local.py, ai_edit.py: alinhamento fonético via whisperx e roteirização local via Ollama/Gemma. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
711c397dfe
commit
7b5aed79ee
@@ -0,0 +1,181 @@
|
||||
"""Forced alignment — refine word timestamps against an acoustic model.
|
||||
|
||||
Why this exists
|
||||
--------------
|
||||
faster-whisper derives word times by cross-attention, which lands every word
|
||||
*start* systematically ~0.3-0.5s early (the word-end is fine). That bias flows
|
||||
straight into the voice timeline and makes zoom/cut land on the wrong frame —
|
||||
measured on real footage in ``Engine/docs/05_EXPERIENCIAS.md`` (#14). Phonetic
|
||||
forced alignment (wav2vec2, via whisperx) re-anchors each word against the
|
||||
audio and brings that error down to ~30ms.
|
||||
|
||||
Design
|
||||
------
|
||||
* The dependency (``whisperx``) is **optional** and imported lazily, exactly
|
||||
like the rest of this stack (librosa, faster-whisper). When it is missing, or
|
||||
any step fails, :meth:`ForcedAligner.align` returns the words unchanged, so
|
||||
transcription never breaks because alignment did.
|
||||
* The aligner is a single responsibility class: it knows how to turn a
|
||||
transcript into the shape whisperx wants, call it, and write the refined
|
||||
times back. ``transcribe.py`` owns the decision of *whether* to align.
|
||||
* Align models are cached per language on the instance so repeated calls
|
||||
(e.g. many short clips) don't reload the wav2vec2 weights each time.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from typing import List, Optional, Sequence
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ForcedAligner:
|
||||
"""Refine word-level timestamps with whisperx phonetic forced alignment.
|
||||
|
||||
Usage::
|
||||
|
||||
aligner = ForcedAligner()
|
||||
words = aligner.align(words, raw_segments, media_path, language, models_dir)
|
||||
|
||||
``words`` and ``raw_segments`` come straight from :func:`transcribe` —
|
||||
``raw_segments`` carries the per-segment ``words`` lists (the same dict
|
||||
objects as in ``words``) so the aligner knows which words belong to which
|
||||
audio window. Returns a list of the *same* word dicts, with ``start``/``end``
|
||||
overwritten in place where alignment produced a usable time.
|
||||
"""
|
||||
|
||||
def __init__(self, device: Optional[str] = None):
|
||||
self._device = device
|
||||
self._models: dict = {}
|
||||
|
||||
# -- capability ------------------------------------------------------
|
||||
@staticmethod
|
||||
def available() -> bool:
|
||||
"""Whether whisperx can be imported (the aligner can run at all)."""
|
||||
try:
|
||||
import whisperx # noqa: F401
|
||||
except Exception:
|
||||
return False
|
||||
return True
|
||||
|
||||
def _resolve_device(self) -> str:
|
||||
if self._device:
|
||||
return self._device
|
||||
try:
|
||||
import torch
|
||||
|
||||
if torch.cuda.is_available():
|
||||
return "cuda"
|
||||
except Exception:
|
||||
pass
|
||||
return "cpu"
|
||||
|
||||
# -- public API ------------------------------------------------------
|
||||
def align(
|
||||
self,
|
||||
words: Sequence[dict],
|
||||
raw_segments: Sequence[dict],
|
||||
audio_path: str,
|
||||
language: str,
|
||||
models_dir: Optional[str] = None,
|
||||
) -> List[dict]:
|
||||
"""Return ``words`` with forced-aligned timestamps where possible.
|
||||
|
||||
Falls back to the unchanged ``words`` on any failure (missing
|
||||
dependency, model load error, audio read error, or a result that
|
||||
doesn't line up with the input).
|
||||
"""
|
||||
if not words or not language:
|
||||
return list(words)
|
||||
try:
|
||||
import whisperx
|
||||
except Exception:
|
||||
logger.info("whisperx not installed; skipping forced alignment")
|
||||
return list(words)
|
||||
|
||||
try:
|
||||
device = self._resolve_device()
|
||||
align_input = self._build_align_input(words, raw_segments)
|
||||
audio = whisperx.load_audio(audio_path)
|
||||
|
||||
if language not in self._models:
|
||||
align_model, metadata = whisperx.load_align_model(
|
||||
language_code=language,
|
||||
device=device,
|
||||
model_dir=str(models_dir) if models_dir else None,
|
||||
)
|
||||
self._models[language] = (align_model, metadata)
|
||||
align_model, metadata = self._models[language]
|
||||
|
||||
result = whisperx.align(
|
||||
align_input,
|
||||
align_model,
|
||||
metadata,
|
||||
audio,
|
||||
device,
|
||||
return_char_alignments=False,
|
||||
)
|
||||
return self._merge_result(words, result.get("segments", []))
|
||||
except Exception:
|
||||
logger.warning(
|
||||
"forced alignment failed for %s; using raw timestamps", audio_path
|
||||
)
|
||||
return list(words)
|
||||
|
||||
# -- internals -------------------------------------------------------
|
||||
@staticmethod
|
||||
def _build_align_input(
|
||||
words: Sequence[dict], raw_segments: Sequence[dict]
|
||||
) -> List[dict]:
|
||||
"""Transcript in whisperx's expected shape: segments -> words.
|
||||
|
||||
whisperx.align requires each segment to carry ``text``/``start``/``end``
|
||||
and a ``words`` list whose entries have ``word``/``start``/``end``/``score``.
|
||||
We only read ``words`` from ``raw_segments`` (the flattened ``words``
|
||||
list is the source of truth for counts), so the two stay consistent.
|
||||
"""
|
||||
align_segments: List[dict] = []
|
||||
for seg in raw_segments:
|
||||
seg_words = [
|
||||
{
|
||||
"word": w.get("word", ""),
|
||||
"start": float(w.get("start", 0.0)),
|
||||
"end": float(w.get("end", 0.0)),
|
||||
"score": float(w.get("confidence", 0.0)),
|
||||
}
|
||||
for w in seg.get("words", [])
|
||||
]
|
||||
align_segments.append(
|
||||
{
|
||||
"text": (seg.get("text") or "").strip(),
|
||||
"start": float(seg.get("start", 0.0)),
|
||||
"end": float(seg.get("end", 0.0)),
|
||||
"words": seg_words,
|
||||
}
|
||||
)
|
||||
return align_segments
|
||||
|
||||
@staticmethod
|
||||
def _merge_result(words: Sequence[dict], aligned_segments: Sequence[dict]) -> List[dict]:
|
||||
"""Walk the aligned output in order and overwrite word times in place.
|
||||
|
||||
whisperx preserves word order within and across segments, so a single
|
||||
running index over the output words lines up with ``words``. A word the
|
||||
aligner failed to place gets ``None``/``0`` times — we skip those rather
|
||||
than clobber a good timestamp, and if counts ever diverge we stop and
|
||||
leave the rest untouched.
|
||||
"""
|
||||
out = list(words)
|
||||
wi = 0
|
||||
for seg in aligned_segments:
|
||||
for aw in seg.get("words", []):
|
||||
if wi >= len(out):
|
||||
return out
|
||||
start = aw.get("start")
|
||||
end = aw.get("end")
|
||||
if start is None or end is None or end < start:
|
||||
wi += 1
|
||||
continue
|
||||
out[wi]["start"] = float(start)
|
||||
out[wi]["end"] = float(end)
|
||||
wi += 1
|
||||
return out
|
||||
@@ -0,0 +1,312 @@
|
||||
"""Local LLM integration — the voice timeline meets a local model.
|
||||
|
||||
The voice timeline is *designed* to be handed to a language model: it is the
|
||||
source of truth between speech analysis and editing, layered so a model can
|
||||
reason about the narrative without parsing FCPXML. This module is the client
|
||||
side of that contract. It formats the timeline into the editar-por-voz brief,
|
||||
calls a local model server (Ollama, running Gemma 3 / Llama locally), and
|
||||
parses the model's decisions back into a validated list of VoiceActions —
|
||||
all inside the engine, so there is no wizard, no copy-paste, no manual step.
|
||||
|
||||
Transport: Ollama's HTTP chat API at ``http://localhost:11434/api/chat``.
|
||||
Any model Ollama serves works; the default is Gemma 3 because that is what
|
||||
runs locally here ("Lama com Gema 3"), but pass ``model=`` to switch.
|
||||
|
||||
The model is untrusted input: its JSON is validated row-by-row by
|
||||
:func:`fcpxml.voice_actions.parse_actions`, so one malformed decision never
|
||||
discards the edit. The brief is written so the model only ever emits the four
|
||||
action kinds the applier understands.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from typing import Any, Dict, Optional, Sequence, Tuple
|
||||
|
||||
import httpx
|
||||
|
||||
from .voice_actions import parse_actions
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_BASE_URL = "http://localhost:11434"
|
||||
# Gemma 3 12B reliably follows the editar-por-voz brief (keep the script, cut
|
||||
# only backstage chatter; the 4B variant skips the "keep the main content"
|
||||
# rule and deletes the script) but doesn't fit an 8GB machine. Qwen2.5 7B
|
||||
# instruct (q4_K_M) is the fallback for constrained hardware — strong at
|
||||
# strict JSON-schema following, the property this brief leans on hardest.
|
||||
# Pass ``model=`` to switch to whatever Ollama serves.
|
||||
DEFAULT_MODEL = "qwen2.5:7b-instruct-q4_K_M"
|
||||
REQUEST_TIMEOUT = 600.0
|
||||
|
||||
# The brief. Ported from the editar-por-voz skill criteria (criterios/01..08),
|
||||
# condensed into the instructions a model needs to emit valid actions. Kept in
|
||||
# Portuguese because the decisions and their reasons are read by a human editor.
|
||||
_SYSTEM_PROMPT = """Você é o editor de vídeo por voz deste sistema. Recebe um JSON de "linha do tempo de voz" — a medição de COMO foi falado (ênfase, energia, pausa, falante) de uma gravação — e devolve as DECISÕES de edição em JSON, nada mais. Você nunca escreve XML.
|
||||
|
||||
Regras (siga rigorosamente):
|
||||
|
||||
1. LEIA EM CAMADAS. "summary" dá o formato da peça; "segments" é onde você trabalha (cada fala com seu texto e agregados); "segments[].words" dá o instante exato de cada destaque. Não recalcule energia, tom ou ênfase — use os números do JSON.
|
||||
|
||||
2. SEPARAR ROTEIRO DE BASTIDOR.
|
||||
- ROTEIRO = o conteúdo principal que a pessoa quer entregar: explicação, depoimento, roteiro decorado, a mensagem. É isso que VAI FICAR.
|
||||
- BASTIDOR = papo casual de gravação, cumprimentos, conversa com a equipe ("cara, beleza?", "tá gravando?", "deixa eu ver o celular"), piadas fora do assunto, tomadas interrompidas ou repetidas. É isso que VIRA "cut".
|
||||
Exemplo: num vídeo sobre mastopexia, a explicação da cirurgia É o roteiro (mantém); o "tá gravando? pois é" antes dela É bastidor (corta).
|
||||
Use "gap_before" e "take_boundary" (silêncio > ~3s = a câmera parou/recomeçou) para agrupar tomadas — eles marcam ONDE a tomada recomeça, não o que cortar. Nunca corte o conteúdo principal só porque tem ênfase; corte o casual/off-topic.
|
||||
|
||||
REGRAS DE OURO:
|
||||
- MANTENHA o conteúdo principal (explicação, depoimento, roteiro decorado). Ele É o vídeo.
|
||||
- CORTE SÓ o casual/off-topic: cumprimentos, "tá gravando?", papo com a equipe, olhar o celular, repetições de tomada.
|
||||
- Em dúvida, MANTENHA a fala. É melhor sobrar conteúdo do que cortar o que era pra ficar.
|
||||
|
||||
3. ESCOLHER A MELHOR TOMADA de cada frase quando há repetições: mantenha a mais limpa e corte as outras (cut cobrindo a frase inteira).
|
||||
|
||||
4. CORTE (kind "cut"): para REMOVER uma frase, cubra ela inteira (start..end = início..fim da frase). Para APARAR só uma hesitação no começo ou fim, corte só da borda até a palavra (corte de meia frase é ambíguo — passe de 60% e apaga a linha toda). Nunca corte o silêncio entre falas.
|
||||
|
||||
5. ZOOM (kind "zoom"): só em palavra de CONTEÚDO bem enfatizada (emphasis alto, não artigo). params.scale entre 1.0 e 3.0 (padrão 1.3 se omitido). Posicione em torno da palavra, segurando até o fim da frase.
|
||||
|
||||
6. TEXTO (kind "text"): params.content obrigatório (≤120 chars), fixa um termo central ou callout. MARKER (kind "marker"): opcional params.content vira o nome do marcador. Use para emendas/junções que o editor deve conferir.
|
||||
|
||||
7. TEMPOS em segundos da MÍDIA ORIGINAL (exatamente como no JSON). Nunca compense para "depois do corte" — o programa desloca sozinho. end sempre > start, ambos ≥ 0.
|
||||
|
||||
8. reason OBRIGATÓRIO em cada ação, em português, embasando a decisão (ex.: 'abertura: "Aquela mama" (ênfase 0.42)'). reason vazio é decisão sem critério.
|
||||
|
||||
Responda APENAS com um objeto JSON válido, sem markdown, sem comentário:
|
||||
{"source": "<nome do arquivo>", "actions": [{"kind": "cut|zoom|text|marker", "start": <float>, "end": <float>, "params": {}, "reason": "<pt>", "speaker": "<id>"}]}
|
||||
"""
|
||||
|
||||
_OUTPUT_REMINDER = """Gere as decisões de edição conforme o brief. Responda SOMENTE o JSON:
|
||||
{"source": "<nome do arquivo>", "actions": [{"kind": "cut|zoom|text|marker", "start": <float>, "end": <float>, "params": {}, "reason": "<pt>", "speaker": "<id>"}]}
|
||||
Não inclua explicações nem blocos markdown."""
|
||||
|
||||
|
||||
def ollama_chat(
|
||||
model: str = DEFAULT_MODEL,
|
||||
messages: Optional[Sequence[Dict[str, str]]] = None,
|
||||
base_url: str = DEFAULT_BASE_URL,
|
||||
temperature: float = 0.2,
|
||||
timeout: float = REQUEST_TIMEOUT,
|
||||
num_ctx: int = 32768,
|
||||
) -> str:
|
||||
"""One chat completion from a local Ollama server.
|
||||
|
||||
Returns the assistant message content. Raises on transport/HTTP errors so
|
||||
the caller can decide whether to retry or report — a model call is the
|
||||
one I/O in this pipeline that can legitimately fail mid-run.
|
||||
"""
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": list(messages or []),
|
||||
"stream": False,
|
||||
"options": {"temperature": temperature, "num_ctx": num_ctx},
|
||||
}
|
||||
try:
|
||||
response = httpx.post(
|
||||
f"{base_url.rstrip('/')}/api/chat", json=payload, timeout=timeout
|
||||
)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
except Exception as exc:
|
||||
# Covers transport errors AND a dropped connection that yields an empty
|
||||
# body (httpx/JSONDecodeError) — both must become a RuntimeError so the
|
||||
# caller reports the failure instead of crashing the whole pipeline.
|
||||
raise RuntimeError(f"Falha ao falar com o modelo local em {base_url}: {exc}") from exc
|
||||
|
||||
return (data.get("message") or {}).get("content", "") or ""
|
||||
|
||||
|
||||
def list_ollama_models(base_url: str = DEFAULT_BASE_URL) -> list[str]:
|
||||
"""Names of the models Ollama currently serves, for a model picker.
|
||||
|
||||
Returns an empty list when Ollama is unreachable so the UI can fall back to
|
||||
a free-text field instead of erroring.
|
||||
"""
|
||||
try:
|
||||
resp = httpx.get(f"{base_url.rstrip('/')}/api/tags", timeout=10.0)
|
||||
resp.raise_for_status()
|
||||
models = resp.json().get("models", [])
|
||||
names = [m.get("name") for m in models if m.get("name")]
|
||||
return sorted(names)
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
def _extract_json(text: str) -> Any:
|
||||
"""Pull a JSON value out of a model response, tolerating fences/wrappers."""
|
||||
if not text:
|
||||
return None
|
||||
candidate = text.strip()
|
||||
# Strip a ```json ... ``` (or bare ```) fence if the model added one.
|
||||
fence = re.search(r"```(?:json)?\s*(.*?)\s*```", candidate, re.DOTALL)
|
||||
if fence:
|
||||
candidate = fence.group(1).strip()
|
||||
# Otherwise take the outermost {...} / [...].
|
||||
if not candidate.startswith(("{" if True else "", "[")):
|
||||
start = min(
|
||||
(i for i, c in enumerate(candidate) if c in "{["),
|
||||
default=None,
|
||||
)
|
||||
end = max(
|
||||
(i for i, c in enumerate(candidate) if c in "}"),
|
||||
default=None,
|
||||
)
|
||||
if start is not None and end is not None and end > start:
|
||||
candidate = candidate[start : end + 1]
|
||||
try:
|
||||
data = json.loads(candidate)
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
|
||||
# Models sometimes wrap the expected `{"source", "actions"}` object inside a
|
||||
# single-element list (`[{...}]`). Unwrap that so the actions aren't treated
|
||||
# as one malformed row.
|
||||
if (
|
||||
isinstance(data, list)
|
||||
and len(data) == 1
|
||||
and isinstance(data[0], dict)
|
||||
and "actions" in data[0] # the wrapper carries the actions key
|
||||
):
|
||||
data = data[0]
|
||||
return data
|
||||
|
||||
|
||||
# Only these fields reach the model — the raw timeline also carries heavy
|
||||
# per-word audio features (energy, pitch, arousal...) and speaker `samples`
|
||||
# that blow past the model's context window on any real recording. Dropping
|
||||
# them is what keeps a 3-minute timeline inside `num_ctx`.
|
||||
_SEGMENT_KEEP = (
|
||||
"start", "end", "speaker", "text", "gap_before", "take_boundary",
|
||||
"avg_energy", "peak_emphasis", "emotion", "emotion_confidence",
|
||||
"arousal", "valence",
|
||||
)
|
||||
_WORD_KEEP = ("text", "start", "end", "speaker", "emphasis", "pause_before")
|
||||
_SPEAKER_KEEP = ("id", "name")
|
||||
_SKIP_ROOT = ("layers", "scales")
|
||||
|
||||
|
||||
def _project_timeline(timeline: dict) -> dict:
|
||||
"""Strip the timeline down to what the edit decision actually needs."""
|
||||
out = {k: v for k, v in timeline.items() if k not in _SKIP_ROOT}
|
||||
speakers = [
|
||||
{k: sp[k] for k in _SPEAKER_KEEP if k in sp}
|
||||
for sp in timeline.get("speakers", [])
|
||||
]
|
||||
if speakers:
|
||||
out["speakers"] = speakers
|
||||
segs = []
|
||||
for seg in timeline.get("segments", []):
|
||||
s = {k: seg[k] for k in _SEGMENT_KEEP if k in seg}
|
||||
s["words"] = [
|
||||
{k: w[k] for k in _WORD_KEEP if k in w}
|
||||
for w in seg.get("words", [])
|
||||
]
|
||||
segs.append(s)
|
||||
out["segments"] = segs
|
||||
return out
|
||||
|
||||
|
||||
def _shrink_to_fit(compact: dict, max_chars: int) -> dict:
|
||||
"""Drop word detail from the lowest-emphasis segments until it fits."""
|
||||
segs = [dict(s) for s in compact.get("segments", [])]
|
||||
while True:
|
||||
payload = json.dumps(
|
||||
{**compact, "segments": segs}, ensure_ascii=False, indent=1
|
||||
)
|
||||
if len(payload) <= max_chars or not any(s.get("words") for s in segs):
|
||||
break
|
||||
idx = min(
|
||||
(i for i, s in enumerate(segs) if s.get("words")),
|
||||
key=lambda i: float(segs[i].get("peak_emphasis", 0.0)),
|
||||
)
|
||||
segs[idx] = {**segs[idx], "words": []}
|
||||
compact = dict(compact)
|
||||
compact["segments"] = segs
|
||||
return compact
|
||||
|
||||
|
||||
def build_edit_messages(
|
||||
timeline: dict, max_words_per_segment: int = 200, max_chars: int = 110000
|
||||
) -> Tuple[str, str]:
|
||||
"""The (system, user) pair that sends a timeline to the model.
|
||||
|
||||
The user turn carries a *projected* timeline (see :func:`_project_timeline`)
|
||||
— text, timing, speaker and emphasis only — so a real recording fits in the
|
||||
model's context window. Very long segments still have their word detail
|
||||
capped to ``max_words_per_segment`` (most emphatic + boundaries), and if the
|
||||
whole payload would still exceed ``max_chars`` the lowest-emphasis segments
|
||||
lose their words until it fits, so we never blow ``num_ctx``.
|
||||
"""
|
||||
compact = _project_timeline(timeline)
|
||||
if max_words_per_segment:
|
||||
segs = []
|
||||
for seg in compact["segments"]:
|
||||
words = seg.get("words", [])
|
||||
if len(words) > max_words_per_segment:
|
||||
ranked = sorted(
|
||||
enumerate(words),
|
||||
key=lambda kv: float(kv[1].get("emphasis", 0.0)),
|
||||
reverse=True,
|
||||
)[: max_words_per_segment - 2]
|
||||
keep = sorted({0, len(words) - 1} | {i for i, _ in ranked})
|
||||
seg = {**seg, "words": [words[i] for i in keep]}
|
||||
segs.append(seg)
|
||||
compact["segments"] = segs
|
||||
|
||||
payload = json.dumps(compact, ensure_ascii=False, indent=1)
|
||||
if len(payload) > max_chars:
|
||||
compact = _shrink_to_fit(compact, max_chars)
|
||||
payload = json.dumps(compact, ensure_ascii=False, indent=1)
|
||||
|
||||
user = (
|
||||
"Linha do tempo de voz (JSON):\n\n"
|
||||
+ payload
|
||||
+ "\n\n"
|
||||
+ _OUTPUT_REMINDER
|
||||
)
|
||||
return _SYSTEM_PROMPT, user
|
||||
|
||||
|
||||
def generate_voice_actions(
|
||||
timeline: dict,
|
||||
model: str = DEFAULT_MODEL,
|
||||
base_url: str = DEFAULT_BASE_URL,
|
||||
temperature: float = 0.2,
|
||||
timeout: float = REQUEST_TIMEOUT,
|
||||
num_ctx: int = 32768,
|
||||
max_words_per_segment: int = 200,
|
||||
) -> Dict[str, Any]:
|
||||
"""Ask the local model to direct the edit, returning validated actions.
|
||||
|
||||
Returns ``{"actions": [VoiceAction], "raw": str, "errors": [str]}``.
|
||||
``actions`` is empty when the model returned nothing usable; ``errors``
|
||||
carries the per-row rejections from :func:`parse_actions` plus any
|
||||
extraction failure, so the caller can report what went wrong instead of
|
||||
only the wins.
|
||||
"""
|
||||
system, user = build_edit_messages(timeline, max_words_per_segment)
|
||||
try:
|
||||
raw = ollama_chat(
|
||||
model=model,
|
||||
messages=[
|
||||
{"role": "system", "content": system},
|
||||
{"role": "user", "content": user},
|
||||
],
|
||||
base_url=base_url,
|
||||
temperature=temperature,
|
||||
timeout=timeout,
|
||||
num_ctx=num_ctx,
|
||||
)
|
||||
except RuntimeError as exc:
|
||||
return {"actions": [], "raw": "", "errors": [str(exc)]}
|
||||
|
||||
data = _extract_json(raw)
|
||||
if data is None:
|
||||
return {
|
||||
"actions": [],
|
||||
"raw": raw,
|
||||
"errors": ["O modelo não devolveu um JSON de decisões legível."],
|
||||
}
|
||||
actions, errors = parse_actions(data)
|
||||
return {"actions": actions, "raw": raw, "errors": errors}
|
||||
@@ -194,6 +194,7 @@ class FCPXMLParser:
|
||||
media_path=media_path,
|
||||
audio_role=elem.get('audioRole', ''),
|
||||
video_role=elem.get('videoRole', ''),
|
||||
rotation=self._parse_clip_rotation(elem),
|
||||
)
|
||||
|
||||
clip.markers.extend(self._collect_markers(elem))
|
||||
@@ -205,6 +206,19 @@ class FCPXMLParser:
|
||||
|
||||
return clip
|
||||
|
||||
def _parse_clip_rotation(self, elem: ET.Element) -> float:
|
||||
"""Degrees from this clip's ``<adjust-transform rotation="...">`` —
|
||||
an edit-time correction (e.g. straightening a tilted phone shot),
|
||||
not the camera's own recorded orientation. FCP writes the rotation
|
||||
as an attribute on that element, not as a filter param."""
|
||||
transform = elem.find('adjust-transform')
|
||||
if transform is None:
|
||||
return 0.0
|
||||
try:
|
||||
return float(transform.get('rotation', '0'))
|
||||
except ValueError:
|
||||
return 0.0
|
||||
|
||||
def _parse_marker_element(self, elem: ET.Element) -> Optional[Marker]:
|
||||
"""Parse any marker element (<marker> or <chapter-marker>).
|
||||
|
||||
@@ -337,6 +351,7 @@ class FCPXMLParser:
|
||||
lane=lane, offset=offset, source_start=start,
|
||||
media_path=media_path, clip_type=elem.tag, role=role,
|
||||
ref_id=ref, parent_clip_name=parent_name,
|
||||
rotation=self._parse_clip_rotation(elem),
|
||||
)
|
||||
|
||||
connected.markers.extend(self._collect_markers(elem))
|
||||
|
||||
@@ -300,6 +300,7 @@ def build_phrase_review(
|
||||
"version": PHRASE_REVIEW_VERSION,
|
||||
"source": source,
|
||||
"source_path": resolve_source(source, voice_timeline_path, extra_dirs),
|
||||
"rotation": float(timeline.get("rotation", 0.0)),
|
||||
"duration": round(phrases[-1]["end"], 3) if phrases else 0.0,
|
||||
"speakers": timeline.get("speakers", []),
|
||||
"emotion_available": bool(layers.get("emotion", False)),
|
||||
|
||||
+44
-12
@@ -125,18 +125,27 @@ def transcribe(
|
||||
model_size: str = "base",
|
||||
language: Optional[str] = None,
|
||||
progress_cb: Optional[Callable[[float], None]] = None,
|
||||
align: bool = True,
|
||||
) -> Optional[dict]:
|
||||
"""Transcribe an audio/video file locally with word-level timestamps.
|
||||
|
||||
Requires the optional ``[transcribe]`` extra (faster-whisper). Returns
|
||||
``None`` when the model is unavailable or the file is missing/unreadable.
|
||||
|
||||
When ``align`` is true (default) and the optional ``whisperx`` dependency is
|
||||
present, word timestamps are refined by phonetic forced alignment, which
|
||||
corrects faster-whisper's systematic ~0.3-0.5s early bias on word *starts*
|
||||
(see ``Engine/docs/05_EXPERIENCIAS.md`` #14). The transcript reports
|
||||
whether this ran via the ``alignment`` flag, so downstream consumers can
|
||||
rely on the times without re-measuring.
|
||||
|
||||
The model weights are resolved from the configured models directory (see
|
||||
``model_manager.get_models_dir``), so a model selected/downloaded through
|
||||
the app is found without an implicit download to the default HF cache.
|
||||
|
||||
Returns:
|
||||
``{"language": str, "duration": float, "text": str,
|
||||
"alignment": bool,
|
||||
"segments": [{"text", "start", "end", "start_fmt", "end_fmt"}, ...],
|
||||
"words": [{"word", "start", "end", "confidence"}, ...]}``
|
||||
"""
|
||||
@@ -175,6 +184,7 @@ def transcribe(
|
||||
vad_filter=True,
|
||||
)
|
||||
segments: List[dict] = []
|
||||
raw_segments: List[dict] = []
|
||||
words: List[dict] = []
|
||||
# `info.duration` is known upfront (from the container), so each
|
||||
# segment's end time — yielded lazily as faster-whisper decodes —
|
||||
@@ -183,6 +193,20 @@ def transcribe(
|
||||
for seg in segments_iter:
|
||||
start = float(seg.start)
|
||||
end = float(seg.end)
|
||||
seg_words: List[dict] = []
|
||||
if progress_cb is not None and total_duration > 0:
|
||||
progress_cb(min(end / total_duration, 1.0))
|
||||
for w in seg.words or []:
|
||||
ws = float(w.start)
|
||||
we = float(w.end)
|
||||
word = {
|
||||
"word": w.word.strip(),
|
||||
"start": ws,
|
||||
"end": we,
|
||||
"confidence": float(w.probability),
|
||||
}
|
||||
words.append(word)
|
||||
seg_words.append(word)
|
||||
segments.append(
|
||||
{
|
||||
"text": seg.text.strip(),
|
||||
@@ -192,19 +216,26 @@ def transcribe(
|
||||
"end_fmt": format_timestamp(end),
|
||||
}
|
||||
)
|
||||
if progress_cb is not None and total_duration > 0:
|
||||
progress_cb(min(end / total_duration, 1.0))
|
||||
for w in seg.words or []:
|
||||
ws = float(w.start)
|
||||
we = float(w.end)
|
||||
words.append(
|
||||
{
|
||||
"word": w.word.strip(),
|
||||
"start": ws,
|
||||
"end": we,
|
||||
"confidence": float(w.probability),
|
||||
}
|
||||
raw_segments.append(
|
||||
{
|
||||
"text": seg.text.strip(),
|
||||
"start": start,
|
||||
"end": end,
|
||||
"words": seg_words,
|
||||
}
|
||||
)
|
||||
|
||||
alignment_ran = False
|
||||
if align and raw_segments:
|
||||
from .forced_align import ForcedAligner
|
||||
|
||||
try:
|
||||
words = ForcedAligner().align(
|
||||
words, raw_segments, str(file_path), info.language, str(models_dir)
|
||||
)
|
||||
alignment_ran = True
|
||||
except Exception:
|
||||
logger.warning("forced alignment step failed; keeping raw timestamps")
|
||||
except Exception:
|
||||
logger.warning("whisper transcription failed for %s", file_path)
|
||||
return None
|
||||
@@ -212,6 +243,7 @@ def transcribe(
|
||||
"language": info.language,
|
||||
"duration": float(info.duration),
|
||||
"text": " ".join(s["text"] for s in segments),
|
||||
"alignment": alignment_ran,
|
||||
"segments": segments,
|
||||
"words": words,
|
||||
}
|
||||
|
||||
@@ -512,6 +512,7 @@ def build_voice_timeline(
|
||||
emphasis_floor: float = 0.25,
|
||||
emotion_enabled: bool = False,
|
||||
emotion_sensitivity: float = 0.5,
|
||||
rotation: float = 0.0,
|
||||
progress_cb: Optional[Callable[[float, str], None]] = None,
|
||||
) -> dict:
|
||||
"""Build the consolidated voice timeline for one media file.
|
||||
@@ -545,6 +546,10 @@ def build_voice_timeline(
|
||||
return {
|
||||
"version": VOICE_TIMELINE_VERSION,
|
||||
"source": Path(media_path).name,
|
||||
# Edit-time correction from the clip's Transform filter in the FCPXML
|
||||
# (e.g. straightening a tilted phone shot) — 0.0 when the clip has none
|
||||
# or the caller didn't resolve one.
|
||||
"rotation": rotation,
|
||||
"language": transcript.get("language", ""),
|
||||
# What actually ran, not what was installed — a consumer must be able
|
||||
# to tell "this speech is flat" from "the acoustics never loaded",
|
||||
@@ -554,6 +559,7 @@ def build_voice_timeline(
|
||||
"acoustics": pitch_track is not None or energy_track is not None,
|
||||
"speakers": tracks is not None,
|
||||
"emotion": bool(emotion_enabled),
|
||||
"alignment": bool(transcript.get("alignment")),
|
||||
},
|
||||
"scales": VALUE_SCALES,
|
||||
"summary": _summary(
|
||||
|
||||
@@ -550,6 +550,13 @@ class TitlesMixin:
|
||||
"""
|
||||
titles = []
|
||||
for elem in self.root.iter('title'):
|
||||
# enabled="0" never renders in Final Cut (see
|
||||
# generate_subtitles_by_emphasis, which disables plain titles
|
||||
# under an emphasis phrase instead of never creating them) — a
|
||||
# title that is off by design must not count as a collision
|
||||
# against the one drawn in its place.
|
||||
if elem.get('enabled', '1') == '0':
|
||||
continue
|
||||
text_el = elem.find('text/text-style')
|
||||
text = (text_el.text or '').strip() if text_el is not None else ''
|
||||
style = elem.find('text-style-def/text-style')
|
||||
|
||||
Reference in New Issue
Block a user