feat: etapa 5 do assistente — revisão de ênfases com timeline
Transforma a etapa "colar decisões" numa tela de lapidação: a sugestão da IA chega carregada e o editor afina frase a frase o que é ênfase e o que fica fora. Essa marcação é o norte da etapa 6 — só as frases com ênfase recebem zoom e legenda dinâmica; as demais ficam com legenda comum. O campo de colar o JSON sobe para a etapa 4, então a numeração das etapas não muda e a etapa 6 segue intacta. Backend (fcpxml/phrase_review.py): - build_phrase_review funde o _voice_timeline.json com as actions da IA - trim por frase que anda em fronteira de palavra; corte parcial da IA chega como trim em vez de ser arredondado fora - phrase_review_to_actions volta a cuts/zooms + emphasis_spans - merge_saved_decisions reaplica só as decisões salvas sobre uma revisão remontada da análise atual, para reprocessar a voz não ficar mascarado - resolve_source acha a mídia: o voice timeline guarda só o nome do arquivo App (SwiftUI): - layout de sala de edição: preview em cima, inspector à direita, timeline atravessando embaixo com seis trilhas rotuladas - preview enquadra no formato de entrega lido do .fcpxml (fonte horizontal, projeto vertical), com alternância para a mídia original - reprodução pula os trechos removidos e para no fim do trecho - zoom manual por trecho marcado, sem guardar escala: a forma vem das configurações de Análise de Voz no render - emoção da fala exposta por frase Correções encontradas no caminho: - VideoPlayer (AVKit) aborta em runtime no app compilado por swiftc; trocado por AVPlayerLayer (ver Engine/docs/05_EXPERIENCIAS.md #22) - teste que ainda afirmava o default zoom scale=1.3 removido do parser (#21) Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
e7748c2c58
commit
1bebee4359
@@ -382,6 +382,10 @@ DEFAULT_VOICE_ANALYSIS_CONFIG: dict = {
|
||||
"emphasis_floor": 0.25,
|
||||
"emotion_enabled": False,
|
||||
"emotion_sensitivity": 0.5,
|
||||
"zoom_scale": 1.30,
|
||||
"zoom_mode": "in_out",
|
||||
"zoom_ease_in": 0.25,
|
||||
"zoom_ease_out": 0.04,
|
||||
}
|
||||
|
||||
|
||||
@@ -402,12 +406,28 @@ def load_voice_analysis_config() -> dict:
|
||||
stored = _load_config().get("voice_analysis")
|
||||
if not isinstance(stored, dict):
|
||||
return cfg
|
||||
for key in ("energy_threshold", "peak_percentile", "emphasis_floor", "emotion_sensitivity"):
|
||||
for key in (
|
||||
"energy_threshold", "peak_percentile", "emphasis_floor",
|
||||
"emotion_sensitivity", "zoom_scale", "zoom_ease_in", "zoom_ease_out",
|
||||
):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = max(0.0, min(1.0, float(stored[key])))
|
||||
value = float(stored[key])
|
||||
if key == "zoom_scale":
|
||||
cfg[key] = max(1.0, min(3.0, value))
|
||||
elif key.startswith("zoom_ease"):
|
||||
cfg[key] = max(0.01, min(5.0, value))
|
||||
else:
|
||||
cfg[key] = max(0.0, min(1.0, value))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if "emphasis_threshold" in stored and "emphasis_floor" not in stored:
|
||||
try:
|
||||
cfg["emphasis_floor"] = max(0.0, min(1.0, float(stored["emphasis_threshold"])))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if stored.get("zoom_mode") in ("in_out", "in", "out"):
|
||||
cfg["zoom_mode"] = stored["zoom_mode"]
|
||||
if "emotion_enabled" in stored:
|
||||
cfg["emotion_enabled"] = bool(stored["emotion_enabled"])
|
||||
weights = stored.get("emphasis_weights")
|
||||
@@ -428,6 +448,10 @@ def save_voice_analysis_config(
|
||||
emphasis_floor: float | None = None,
|
||||
emotion_enabled: bool | None = None,
|
||||
emotion_sensitivity: float | None = None,
|
||||
zoom_scale: float | None = None,
|
||||
zoom_mode: str | None = None,
|
||||
zoom_ease_in: float | None = None,
|
||||
zoom_ease_out: float | None = None,
|
||||
) -> dict:
|
||||
"""Persist voice-analysis thresholds/weights. Only given fields change.
|
||||
|
||||
@@ -446,6 +470,14 @@ def save_voice_analysis_config(
|
||||
cfg["emotion_enabled"] = bool(emotion_enabled)
|
||||
if emotion_sensitivity is not None:
|
||||
cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity)))
|
||||
if zoom_scale is not None:
|
||||
cfg["zoom_scale"] = max(1.0, min(3.0, float(zoom_scale)))
|
||||
if zoom_mode in ("in_out", "in", "out"):
|
||||
cfg["zoom_mode"] = zoom_mode
|
||||
if zoom_ease_in is not None:
|
||||
cfg["zoom_ease_in"] = max(0.01, min(5.0, float(zoom_ease_in)))
|
||||
if zoom_ease_out is not None:
|
||||
cfg["zoom_ease_out"] = max(0.01, min(5.0, float(zoom_ease_out)))
|
||||
if emphasis_weights is not None:
|
||||
for key, value in emphasis_weights.items():
|
||||
if key in cfg["emphasis_weights"] and value is not None:
|
||||
@@ -536,6 +568,73 @@ def save_dynamic_subtitle_config(**fields) -> dict:
|
||||
return cfg
|
||||
|
||||
|
||||
DEFAULT_PLAIN_SUBTITLE_CONFIG: dict = {
|
||||
"font": "Helvetica Neue",
|
||||
"font_size": 82,
|
||||
"font_color": "1 1 1 1",
|
||||
"max_words": 7,
|
||||
"position_y": -820.0,
|
||||
"uppercase": False,
|
||||
"keep_punctuation": True,
|
||||
"text_scale": 2.0,
|
||||
}
|
||||
|
||||
|
||||
def load_plain_subtitle_config() -> dict:
|
||||
"""Persisted style for simple editable FCPXML title subtitles."""
|
||||
cfg = dict(DEFAULT_PLAIN_SUBTITLE_CONFIG)
|
||||
stored = _load_config().get("plain_subtitles")
|
||||
if not isinstance(stored, dict):
|
||||
return cfg
|
||||
for key in ("position_y", "text_scale"):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = float(stored[key])
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
for key in ("font_size", "max_words"):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = int(stored[key])
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
for key in ("font", "font_color"):
|
||||
if key in stored and isinstance(stored[key], str) and stored[key]:
|
||||
cfg[key] = stored[key]
|
||||
for key in ("uppercase", "keep_punctuation"):
|
||||
if key in stored:
|
||||
cfg[key] = bool(stored[key])
|
||||
cfg["max_words"] = max(1, int(cfg["max_words"]))
|
||||
return cfg
|
||||
|
||||
|
||||
def save_plain_subtitle_config(**fields) -> dict:
|
||||
"""Persist simple subtitle style fields. Only given fields change."""
|
||||
cfg = load_plain_subtitle_config()
|
||||
for key, value in fields.items():
|
||||
if key not in DEFAULT_PLAIN_SUBTITLE_CONFIG or value is None:
|
||||
continue
|
||||
if isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], bool):
|
||||
cfg[key] = bool(value)
|
||||
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], float):
|
||||
try:
|
||||
cfg[key] = float(value)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
elif isinstance(DEFAULT_PLAIN_SUBTITLE_CONFIG[key], int):
|
||||
try:
|
||||
cfg[key] = int(value)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
else:
|
||||
cfg[key] = str(value)
|
||||
cfg["max_words"] = max(1, int(cfg["max_words"]))
|
||||
data = _load_config()
|
||||
data["plain_subtitles"] = cfg
|
||||
_write_config(data)
|
||||
return cfg
|
||||
|
||||
|
||||
# Mirrors the silence thresholds the detection/removal handlers use when no
|
||||
# argument is passed (server_tools/qc.py). Persisted so the app's slider and
|
||||
# any later run agree without threading three fields through every call.
|
||||
|
||||
@@ -0,0 +1,547 @@
|
||||
"""Phrase review — the human pass between the AI's decisions and the render.
|
||||
|
||||
A voice timeline says *how* every line was spoken; a list of voice actions says
|
||||
what the model decided to do about it. Neither is reviewable on its own: the
|
||||
timeline has no editorial intent, and the action list is a set of timecodes with
|
||||
no text attached. This module joins them into the one view an editor can
|
||||
actually judge — the script, phrase by phrase, each carrying the decision that
|
||||
was made about it.
|
||||
|
||||
The phrase is the unit on purpose. Emphasis, in this pipeline, is not a property
|
||||
of a word but of a line: an emphasized phrase gets a punch-in and a dynamic
|
||||
caption, everything else gets a plain caption. Keeping the same granularity in
|
||||
the review, the JSON, and the render means a toggle in the UI maps to exactly
|
||||
one editorial outcome, with nothing to reconcile in between.
|
||||
|
||||
Trimming stays inside the phrase for the same reason. A line is rarely wrong as
|
||||
a whole — it has a false start, or a trailing "né" — so each phrase carries a
|
||||
``trim_start``/``trim_end`` pair that rides on word boundaries. Editing a cut
|
||||
therefore means picking a word, never hunting for a frame, and a partial cut
|
||||
from the model arrives as a trim instead of being rounded away.
|
||||
|
||||
Round-tripping is the other half of the contract. :func:`build_phrase_review`
|
||||
derives the review from actions, :func:`phrase_review_to_actions` derives
|
||||
actions back from the edited review, and everything the editor touched wins over
|
||||
what was inferred — so re-opening the screen shows what was left there, not a
|
||||
re-derivation that quietly discards the edits.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Sequence, Tuple
|
||||
|
||||
from .voice_actions import VoiceAction, merge_cut_ranges, parse_actions
|
||||
|
||||
PHRASE_REVIEW_VERSION = "1.0"
|
||||
|
||||
# Emphasis is stored 0-3 rather than as a float so the UI, the JSON and the
|
||||
# render agree on the same discrete decision. The thresholds map the continuous
|
||||
# `peak_emphasis` of the voice timeline onto those levels when the model gave no
|
||||
# explicit direction for a phrase.
|
||||
EMPHASIS_LEVELS = (0, 1, 2, 3)
|
||||
EMPHASIS_THRESHOLDS = (0.25, 0.45, 0.65)
|
||||
|
||||
# Zoom scale applied per emphasis level when the review is turned back into
|
||||
# actions. Level 0 never produces a zoom. The values stay inside
|
||||
# voice_actions.MIN_ZOOM_SCALE..MAX_ZOOM_SCALE.
|
||||
ZOOM_SCALE_BY_LEVEL = {1: 1.15, 2: 1.3, 3: 1.5}
|
||||
|
||||
# A phrase only survives if most of it does. Speech boundaries from a transcript
|
||||
# are approximate, so a cut clipping a fraction of a second off the tail is a
|
||||
# trim, not a removal — treating that as "phrase deleted" would grey out lines
|
||||
# that are still fully audible.
|
||||
CUT_COVERAGE_TO_DEACTIVATE = 0.6
|
||||
|
||||
# A punch-in shorter than this has no time to ramp in and back out — the writer
|
||||
# rejects the window anyway (see the zoom ease-in/ease-out shape), so refusing
|
||||
# it here turns a silent drop at render time into nothing being placed at all.
|
||||
MIN_ZOOM_DURATION = 0.4
|
||||
|
||||
TRACK_SCRIPT = "roteiro"
|
||||
TRACK_BACKSTAGE = "bastidor"
|
||||
TRACKS = (TRACK_SCRIPT, TRACK_BACKSTAGE)
|
||||
|
||||
|
||||
def resolve_source(
|
||||
source: str, voice_timeline_path: str, extra_dirs: Sequence[str] = ()
|
||||
) -> str:
|
||||
"""The playable path for a timeline's ``source``, or "" when it's gone.
|
||||
|
||||
The voice timeline stores only the media's *file name* — it is written to be
|
||||
read by a model, where a machine-specific absolute path is noise. That makes
|
||||
it useless for opening a preview, so the file is looked up where it can
|
||||
actually be: beside its own timeline JSON first (that is where
|
||||
``analyze_voice`` writes it), then in whatever project folders the caller
|
||||
knows about.
|
||||
"""
|
||||
if not source:
|
||||
return ""
|
||||
candidate = Path(source)
|
||||
if candidate.is_absolute() and candidate.is_file():
|
||||
return str(candidate)
|
||||
|
||||
directories = [Path(voice_timeline_path).parent] if voice_timeline_path else []
|
||||
directories += [Path(d) for d in extra_dirs if d]
|
||||
for directory in directories:
|
||||
found = directory / candidate.name
|
||||
if found.is_file():
|
||||
return str(found)
|
||||
return ""
|
||||
|
||||
|
||||
def _overlap(a_start: float, a_end: float, b_start: float, b_end: float) -> float:
|
||||
"""Seconds shared by two spans (0.0 when they don't touch)."""
|
||||
return max(0.0, min(a_end, b_end) - max(a_start, b_start))
|
||||
|
||||
|
||||
def _cut_coverage(
|
||||
start: float, end: float, cuts: Sequence[Tuple[float, float]]
|
||||
) -> float:
|
||||
"""Fraction of ``start``-``end`` that falls inside ``cuts`` (0-1)."""
|
||||
span = end - start
|
||||
if span <= 0:
|
||||
return 0.0
|
||||
removed = sum(_overlap(start, end, c_start, c_end) for c_start, c_end in cuts)
|
||||
return min(1.0, removed / span)
|
||||
|
||||
|
||||
def snap_to_words(
|
||||
time: float, words: Sequence[dict], fallback: float, edge: str
|
||||
) -> float:
|
||||
"""Move ``time`` onto the nearest word boundary of this phrase.
|
||||
|
||||
Trims are expressed by pointing at a word, so a trim handle that landed
|
||||
mid-word would cut a syllable in half. ``edge`` is ``"in"`` (snap to word
|
||||
starts) or ``"out"`` (snap to word ends); with no word timings available the
|
||||
time is left as-is.
|
||||
"""
|
||||
boundaries = [
|
||||
float(word.get("start" if edge == "in" else "end", 0.0)) for word in words
|
||||
]
|
||||
boundaries = [b for b in boundaries if b > 0]
|
||||
if not boundaries:
|
||||
return fallback
|
||||
return min(boundaries, key=lambda b: abs(b - time))
|
||||
|
||||
|
||||
def _trim_from_cuts(
|
||||
start: float,
|
||||
end: float,
|
||||
words: Sequence[dict],
|
||||
cuts: Sequence[Tuple[float, float]],
|
||||
) -> Tuple[float, float]:
|
||||
"""Read a partial cut over this phrase as a head/tail trim.
|
||||
|
||||
Only cuts that touch an edge become trims: a cut carved out of the middle of
|
||||
a line has no representation here (the phrase is the unit), so it is left
|
||||
for the whole-phrase coverage rule to decide.
|
||||
"""
|
||||
trim_start, trim_end = start, end
|
||||
for cut_start, cut_end in cuts:
|
||||
if _overlap(start, end, cut_start, cut_end) <= 0:
|
||||
continue
|
||||
if cut_start <= trim_start < cut_end < end:
|
||||
trim_start = snap_to_words(cut_end, words, cut_end, "in")
|
||||
if start < cut_start < trim_end <= cut_end:
|
||||
trim_end = snap_to_words(cut_start, words, cut_start, "out")
|
||||
if trim_end <= trim_start:
|
||||
return start, end
|
||||
return trim_start, trim_end
|
||||
|
||||
|
||||
def _level_from_peak(peak: float) -> int:
|
||||
"""Map a 0-1 ``peak_emphasis`` onto a 0-3 level."""
|
||||
for level, threshold in enumerate(EMPHASIS_THRESHOLDS):
|
||||
if peak < threshold:
|
||||
return level
|
||||
return 3
|
||||
|
||||
|
||||
def _level_from_scale(scale: Optional[float]) -> int:
|
||||
"""Map a zoom's scale factor back onto a 0-3 level.
|
||||
|
||||
The model is free to send any scale inside the allowed range, so this picks
|
||||
the nearest level rather than requiring one of our own three values.
|
||||
"""
|
||||
if scale is None:
|
||||
return 2
|
||||
best = 1
|
||||
smallest = None
|
||||
for level, level_scale in ZOOM_SCALE_BY_LEVEL.items():
|
||||
distance = abs(level_scale - float(scale))
|
||||
if smallest is None or distance < smallest:
|
||||
smallest, best = distance, level
|
||||
return best
|
||||
|
||||
|
||||
def _emphasis_from_actions(
|
||||
start: float,
|
||||
end: float,
|
||||
actions: Sequence[VoiceAction],
|
||||
) -> Tuple[Optional[int], str]:
|
||||
"""The level the model asked for on this phrase, and why.
|
||||
|
||||
A ``zoom`` or ``text`` action anywhere inside the phrase is read as "this
|
||||
line is the emphasis" — the model places them on the word that carries the
|
||||
point, not on the whole line, so requiring a full-span match would find
|
||||
nothing. Returns ``(None, "")`` when no action touches the phrase.
|
||||
"""
|
||||
level: Optional[int] = None
|
||||
reason = ""
|
||||
for action in actions:
|
||||
if action.kind not in ("zoom", "text"):
|
||||
continue
|
||||
if _overlap(start, end, action.start, action.end) <= 0:
|
||||
continue
|
||||
if action.kind == "zoom":
|
||||
candidate = _level_from_scale(action.params.get("scale"))
|
||||
else:
|
||||
candidate = 2
|
||||
if level is None or candidate > level:
|
||||
level = candidate
|
||||
reason = action.reason
|
||||
return level, reason
|
||||
|
||||
|
||||
def _cut_reason(
|
||||
start: float, end: float, actions: Sequence[VoiceAction]
|
||||
) -> str:
|
||||
"""The reason given for the cut that removes this phrase."""
|
||||
for action in actions:
|
||||
if action.kind != "cut":
|
||||
continue
|
||||
if _overlap(start, end, action.start, action.end) > 0 and action.reason:
|
||||
return action.reason
|
||||
return ""
|
||||
|
||||
|
||||
def build_phrase_review(
|
||||
timeline: dict,
|
||||
actions: Any = None,
|
||||
voice_timeline_path: str = "",
|
||||
extra_dirs: Sequence[str] = (),
|
||||
) -> dict:
|
||||
"""Join a voice timeline with the AI's actions into a reviewable script.
|
||||
|
||||
``actions`` accepts whatever :func:`~.voice_actions.parse_actions` accepts —
|
||||
a bare list, ``{"actions": [...]}``, or ``None`` when there is no AI pass and
|
||||
the review starts from the acoustics alone. Malformed rows are skipped and
|
||||
reported in ``errors`` rather than raising, matching the rest of the
|
||||
decision pipeline.
|
||||
"""
|
||||
parsed, errors = parse_actions(actions) if actions else ([], [])
|
||||
cuts = merge_cut_ranges(parsed)
|
||||
|
||||
phrases: List[dict] = []
|
||||
for index, segment in enumerate(timeline.get("segments", [])):
|
||||
start = float(segment.get("start", 0.0))
|
||||
end = float(segment.get("end", 0.0))
|
||||
peak = float(segment.get("peak_emphasis", 0.0))
|
||||
take_boundary = bool(segment.get("take_boundary", False))
|
||||
|
||||
words = list(segment.get("words", []))
|
||||
coverage = _cut_coverage(start, end, cuts)
|
||||
active = coverage < CUT_COVERAGE_TO_DEACTIVATE
|
||||
trim_start, trim_end = (
|
||||
_trim_from_cuts(start, end, words, cuts) if active else (start, end)
|
||||
)
|
||||
|
||||
asked_level, asked_reason = _emphasis_from_actions(start, end, parsed)
|
||||
if asked_level is not None:
|
||||
emphasis, reason = asked_level, asked_reason
|
||||
else:
|
||||
emphasis = _level_from_peak(peak)
|
||||
reason = f"ênfase {peak:.2f}" if emphasis else ""
|
||||
if not active:
|
||||
# A removed line carries the reason it was removed; the emphasis it
|
||||
# would have had is kept so re-activating it restores the decision.
|
||||
reason = _cut_reason(start, end, parsed) or reason
|
||||
|
||||
phrases.append(
|
||||
{
|
||||
"index": index,
|
||||
"start": round(start, 3),
|
||||
"end": round(end, 3),
|
||||
"trim_start": round(trim_start, 3),
|
||||
"trim_end": round(trim_end, 3),
|
||||
"text": str(segment.get("text", "")).strip(),
|
||||
"speaker": str(segment.get("speaker", "")),
|
||||
"active": active,
|
||||
"emphasis": emphasis,
|
||||
"track": TRACK_BACKSTAGE if (not active and take_boundary) else TRACK_SCRIPT,
|
||||
"peak_emphasis": round(peak, 3),
|
||||
# Delivery emotion is a heuristic over the acoustics (see
|
||||
# voice_timeline._emotion_for_word) and only means anything when
|
||||
# the analysis actually ran — `emotion_available` below is what
|
||||
# separates "spoken flat" from "never measured".
|
||||
"emotion": str(segment.get("emotion", "neutral")),
|
||||
"emotion_confidence": round(
|
||||
float(segment.get("emotion_confidence", 0.0)), 3
|
||||
),
|
||||
"take_boundary": take_boundary,
|
||||
"gap_before": round(float(segment.get("gap_before", 0.0)), 3),
|
||||
"reason": reason,
|
||||
"words": [
|
||||
{
|
||||
"text": str(word.get("text", "")),
|
||||
"start": round(float(word.get("start", 0.0)), 3),
|
||||
"end": round(float(word.get("end", 0.0)), 3),
|
||||
"energy": round(float(word.get("energy", 0.0)), 3),
|
||||
"emphasis": round(float(word.get("emphasis", 0.0)), 3),
|
||||
}
|
||||
for word in words
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
source = timeline.get("source", "")
|
||||
layers = timeline.get("layers", {}) if isinstance(timeline.get("layers"), dict) else {}
|
||||
return {
|
||||
"version": PHRASE_REVIEW_VERSION,
|
||||
"source": source,
|
||||
"source_path": resolve_source(source, voice_timeline_path, extra_dirs),
|
||||
"duration": round(phrases[-1]["end"], 3) if phrases else 0.0,
|
||||
"speakers": timeline.get("speakers", []),
|
||||
"emotion_available": bool(layers.get("emotion", False)),
|
||||
"phrases": phrases,
|
||||
# Punch-ins the editor places by hand on an arbitrary range, alongside
|
||||
# the whole-phrase zoom that an emphasis level produces. Both end up as
|
||||
# zoom actions; this one exists because the moment worth punching into
|
||||
# is not always a whole sentence.
|
||||
"zooms": [],
|
||||
"errors": errors,
|
||||
}
|
||||
|
||||
|
||||
def _coerce_zoom(raw: Any) -> Optional[Dict[str, float]]:
|
||||
"""Normalize one manually placed zoom range."""
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
try:
|
||||
start = float(raw.get("start"))
|
||||
end = float(raw.get("end"))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if end - start < MIN_ZOOM_DURATION:
|
||||
return None
|
||||
return {"start": start, "end": end}
|
||||
|
||||
|
||||
def _coerce_phrase(raw: Any, index: int) -> Optional[Dict[str, Any]]:
|
||||
"""Normalize one edited phrase row coming back from the UI."""
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
try:
|
||||
start = float(raw.get("start"))
|
||||
end = float(raw.get("end"))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if end <= start:
|
||||
return None
|
||||
try:
|
||||
emphasis = int(raw.get("emphasis", 0))
|
||||
except (TypeError, ValueError):
|
||||
emphasis = 0
|
||||
try:
|
||||
trim_start = float(raw.get("trim_start", start))
|
||||
trim_end = float(raw.get("trim_end", end))
|
||||
except (TypeError, ValueError):
|
||||
trim_start, trim_end = start, end
|
||||
# A trim that escaped the phrase, or inverted, is treated as no trim at all:
|
||||
# the UI is the only thing that writes these, and silently discarding a bad
|
||||
# pair keeps a rounding slip from deleting material the editor kept.
|
||||
if not (start <= trim_start < trim_end <= end):
|
||||
trim_start, trim_end = start, end
|
||||
track = str(raw.get("track", TRACK_SCRIPT))
|
||||
return {
|
||||
"index": int(raw.get("index", index)),
|
||||
"start": start,
|
||||
"end": end,
|
||||
"trim_start": trim_start,
|
||||
"trim_end": trim_end,
|
||||
"text": str(raw.get("text", "")).strip(),
|
||||
"speaker": str(raw.get("speaker", "")),
|
||||
"active": bool(raw.get("active", True)),
|
||||
"emphasis": min(3, max(0, emphasis)),
|
||||
"track": track if track in TRACKS else TRACK_SCRIPT,
|
||||
"reason": str(raw.get("reason", "")),
|
||||
}
|
||||
|
||||
|
||||
def phrase_review_to_actions(review: dict) -> dict:
|
||||
"""Turn an edited review back into the action list the applier consumes.
|
||||
|
||||
Every deactivated phrase becomes a ``cut``, a trimmed one becomes a cut over
|
||||
the head and/or tail it lost, and every emphasized one becomes a ``zoom``
|
||||
scaled by its level. The emphasis flags ride along in ``emphasis_spans`` so
|
||||
the caption step can give those lines the dynamic treatment and everything
|
||||
else the plain one, without re-deriving the decision from the acoustics.
|
||||
"""
|
||||
phrases = [
|
||||
coerced
|
||||
for index, raw in enumerate(review.get("phrases", []))
|
||||
if (coerced := _coerce_phrase(raw, index)) is not None
|
||||
]
|
||||
|
||||
actions: List[dict] = []
|
||||
emphasis_spans: List[dict] = []
|
||||
for phrase in phrases:
|
||||
if not phrase["active"]:
|
||||
actions.append(
|
||||
VoiceAction(
|
||||
kind="cut",
|
||||
start=phrase["start"],
|
||||
end=phrase["end"],
|
||||
reason=phrase["reason"] or "desativada na revisão",
|
||||
speaker=phrase["speaker"],
|
||||
).as_dict()
|
||||
)
|
||||
continue
|
||||
|
||||
# Head and tail the editor trimmed off — each becomes its own cut, so a
|
||||
# false start disappears without taking the line with it.
|
||||
for trim_start, trim_end, where in (
|
||||
(phrase["start"], phrase["trim_start"], "início"),
|
||||
(phrase["trim_end"], phrase["end"], "fim"),
|
||||
):
|
||||
if trim_end - trim_start <= 0:
|
||||
continue
|
||||
actions.append(
|
||||
VoiceAction(
|
||||
kind="cut",
|
||||
start=trim_start,
|
||||
end=trim_end,
|
||||
reason=f"trecho do {where} da frase removido na revisão",
|
||||
speaker=phrase["speaker"],
|
||||
).as_dict()
|
||||
)
|
||||
|
||||
if phrase["emphasis"] >= 1:
|
||||
actions.append(
|
||||
VoiceAction(
|
||||
kind="zoom",
|
||||
start=phrase["trim_start"],
|
||||
end=phrase["trim_end"],
|
||||
params={"scale": ZOOM_SCALE_BY_LEVEL[phrase["emphasis"]]},
|
||||
reason=phrase["reason"] or f"ênfase nível {phrase['emphasis']}",
|
||||
speaker=phrase["speaker"],
|
||||
).as_dict()
|
||||
)
|
||||
emphasis_spans.append(
|
||||
{
|
||||
"start": phrase["trim_start"],
|
||||
"end": phrase["trim_end"],
|
||||
"level": phrase["emphasis"],
|
||||
"text": phrase["text"],
|
||||
}
|
||||
)
|
||||
|
||||
# Hand-placed punch-ins carry no scale on purpose: an omitted scale lets the
|
||||
# applier use the shape configured in "Análise de Voz" (zoom_scale, ease in
|
||||
# and out), so changing that setting restyles every manual zoom instead of
|
||||
# leaving a scale frozen into each one at the moment it was drawn.
|
||||
for raw in review.get("zooms", []):
|
||||
zoom = _coerce_zoom(raw)
|
||||
if zoom is None:
|
||||
continue
|
||||
actions.append(
|
||||
VoiceAction(
|
||||
kind="zoom",
|
||||
start=zoom["start"],
|
||||
end=zoom["end"],
|
||||
reason="zoom marcado na revisão",
|
||||
).as_dict()
|
||||
)
|
||||
|
||||
return {
|
||||
"source": review.get("source", ""),
|
||||
"actions": actions,
|
||||
"emphasis_spans": emphasis_spans,
|
||||
}
|
||||
|
||||
|
||||
def merge_saved_decisions(review: dict, saved: Optional[dict]) -> dict:
|
||||
"""Lay a previously saved review's decisions over a freshly built one.
|
||||
|
||||
Only the editorial fields travel — active, emphasis, track, text, trims.
|
||||
Everything else (words, emotion, energy) is re-derived from the current
|
||||
analysis, so re-running the voice pass with better settings improves the
|
||||
screen instead of being masked by a stale copy of itself, and the saved file
|
||||
never has to carry a duplicate of data it does not own.
|
||||
|
||||
Phrases are matched by index *and* start time: if the analysis changed
|
||||
enough to move a line, the old decision for that slot is dropped rather than
|
||||
applied to a different sentence.
|
||||
"""
|
||||
if not saved:
|
||||
return review
|
||||
|
||||
review["zooms"] = [
|
||||
zoom for raw in saved.get("zooms", []) if (zoom := _coerce_zoom(raw)) is not None
|
||||
]
|
||||
|
||||
by_index = {}
|
||||
for raw in saved.get("phrases", []):
|
||||
if isinstance(raw, dict) and "index" in raw:
|
||||
by_index[raw["index"]] = raw
|
||||
|
||||
for phrase in review["phrases"]:
|
||||
previous = by_index.get(phrase["index"])
|
||||
if previous is None:
|
||||
continue
|
||||
if abs(float(previous.get("start", -1)) - phrase["start"]) > 0.25:
|
||||
continue
|
||||
phrase["active"] = bool(previous.get("active", phrase["active"]))
|
||||
phrase["emphasis"] = min(3, max(0, int(previous.get("emphasis", phrase["emphasis"]))))
|
||||
track = str(previous.get("track", phrase["track"]))
|
||||
phrase["track"] = track if track in TRACKS else phrase["track"]
|
||||
if previous.get("text"):
|
||||
phrase["text"] = str(previous["text"])
|
||||
trim_start = float(previous.get("trim_start", phrase["trim_start"]))
|
||||
trim_end = float(previous.get("trim_end", phrase["trim_end"]))
|
||||
if phrase["start"] <= trim_start < trim_end <= phrase["end"]:
|
||||
phrase["trim_start"], phrase["trim_end"] = trim_start, trim_end
|
||||
|
||||
return review
|
||||
|
||||
|
||||
def review_paths(voice_timeline_path: str) -> Tuple[Path, Path]:
|
||||
"""Where the review and its derived actions live, next to the timeline.
|
||||
|
||||
Both files sit beside the ``_voice_timeline.json`` they came from and are
|
||||
named after it, so a project folder stays readable and re-running the wizard
|
||||
on the same take overwrites its own files instead of accumulating copies.
|
||||
"""
|
||||
base = Path(voice_timeline_path)
|
||||
stem = base.stem
|
||||
if stem.endswith("_voice_timeline"):
|
||||
stem = stem[: -len("_voice_timeline")]
|
||||
return (
|
||||
base.with_name(f"{stem}_phrase_review.json"),
|
||||
base.with_name(f"{stem}_phrase_actions.json"),
|
||||
)
|
||||
|
||||
|
||||
def save_phrase_review(voice_timeline_path: str, review: dict) -> Tuple[Path, Path]:
|
||||
"""Write the edited review and the actions derived from it. Returns both paths."""
|
||||
review_path, actions_path = review_paths(voice_timeline_path)
|
||||
review_path.write_text(
|
||||
json.dumps(review, ensure_ascii=False, indent=2), encoding="utf-8"
|
||||
)
|
||||
actions_path.write_text(
|
||||
json.dumps(phrase_review_to_actions(review), ensure_ascii=False, indent=2),
|
||||
encoding="utf-8",
|
||||
)
|
||||
return review_path, actions_path
|
||||
|
||||
|
||||
def load_phrase_review(voice_timeline_path: str) -> Optional[dict]:
|
||||
"""The review saved earlier for this timeline, or ``None`` if there is none."""
|
||||
review_path, _ = review_paths(voice_timeline_path)
|
||||
if not review_path.is_file():
|
||||
return None
|
||||
try:
|
||||
data = json.loads(review_path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return None
|
||||
return data if isinstance(data, dict) else None
|
||||
@@ -28,8 +28,11 @@ ALLOWED_MODELS = (
|
||||
)
|
||||
|
||||
# Conservative by default: interjections that are near-universally filler.
|
||||
# Portuguese "um"/"uma" are usually articles/numerals inside real phrases
|
||||
# ("de um jeito") rather than discardable hesitations, so only cut them when
|
||||
# the caller explicitly opts in through the fillers argument.
|
||||
# "like" / "so" / "actually" are speech, not noise, unless the user opts in.
|
||||
DEFAULT_FILLERS = ("um", "uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm")
|
||||
DEFAULT_FILLERS = ("uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm")
|
||||
|
||||
_NORM_RE = re.compile(r"[^\w']+")
|
||||
|
||||
|
||||
@@ -82,21 +82,36 @@ def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]:
|
||||
params = dict(params) if isinstance(params, dict) else {}
|
||||
|
||||
if kind == "zoom":
|
||||
try:
|
||||
scale = float(params.get("scale", 1.3))
|
||||
except (TypeError, ValueError):
|
||||
return None, f"{where}: zoom scale must be a number"
|
||||
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
|
||||
return None, (
|
||||
f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}"
|
||||
)
|
||||
params["scale"] = scale
|
||||
if "scale" in params and params.get("scale") is not None:
|
||||
try:
|
||||
scale = float(params["scale"])
|
||||
except (TypeError, ValueError):
|
||||
return None, f"{where}: zoom scale must be a number"
|
||||
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
|
||||
return None, (
|
||||
f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}"
|
||||
)
|
||||
params["scale"] = scale
|
||||
|
||||
if kind == "text":
|
||||
content = str(params.get("content", "")).strip()
|
||||
if not content:
|
||||
return None, f"{where}: text action needs params.content"
|
||||
params["content"] = content[:MAX_TEXT_LENGTH]
|
||||
# Style is optional — omitted fields fall back to the "Legendas
|
||||
# Dinâmicas" emphasis style at apply time (see _apply_placed_action),
|
||||
# so a callout matches the captions' look without the caller having
|
||||
# to know or repeat that configuration. Anything given here wins.
|
||||
for key in ("font", "font_color", "face"):
|
||||
if key in params and not isinstance(params[key], str):
|
||||
del params[key]
|
||||
if "font_size" in params:
|
||||
try:
|
||||
params["font_size"] = int(params["font_size"])
|
||||
except (TypeError, ValueError):
|
||||
del params["font_size"]
|
||||
if "bold" in params:
|
||||
params["bold"] = bool(params["bold"])
|
||||
|
||||
return (
|
||||
VoiceAction(
|
||||
|
||||
@@ -56,12 +56,20 @@ VALUE_SCALES = {
|
||||
"rate_delta": "0-1, how much the local speaking rate departs from the average",
|
||||
"pause_before": "seconds of silence immediately before the word",
|
||||
"emphasis": "0-1 combined index; high values are punch-in/highlight candidates",
|
||||
"emotion": "heuristic label from delivery: neutral, excited, tense, calm, reflective",
|
||||
"emotion_confidence": "0-1 confidence in the heuristic emotion label",
|
||||
"arousal": "0-1 vocal activation from energy/rate/pitch movement",
|
||||
"valence": "0-1 rough positive tone; lower values suggest tension/weight",
|
||||
},
|
||||
"segment": {
|
||||
"gap_before": "seconds of silence before this line",
|
||||
"take_boundary": "true when the gap is long enough that the take likely restarted here",
|
||||
"avg_energy": "0-1 mean loudness across the line",
|
||||
"peak_emphasis": "0-1 highest emphasis of any word in the line",
|
||||
"emotion": "dominant delivery emotion across the line",
|
||||
"emotion_confidence": "0-1 confidence in the dominant segment emotion",
|
||||
"arousal": "0-1 mean vocal activation across the line",
|
||||
"valence": "0-1 mean rough positive tone across the line",
|
||||
},
|
||||
}
|
||||
|
||||
@@ -92,11 +100,74 @@ def _round_word(word: dict) -> dict:
|
||||
"rate_delta": round(word.get("rate_delta", 0.0), 3),
|
||||
"pause_before": round(word.get("pause_before", 0.0), 3),
|
||||
"emphasis": round(word.get("emphasis", 0.0), 3),
|
||||
"emotion": word.get("emotion", "neutral"),
|
||||
"emotion_confidence": round(word.get("emotion_confidence", 0.0), 3),
|
||||
"arousal": round(word.get("arousal", 0.0), 3),
|
||||
"valence": round(word.get("valence", 0.5), 3),
|
||||
"energy_raw": word.get("energy"),
|
||||
"pitch_hz": word.get("pitch_hz"),
|
||||
}
|
||||
|
||||
|
||||
def _emotion_for_word(word: dict, enabled: bool, sensitivity: float) -> dict:
|
||||
"""Classify delivery emotion from normalized acoustic features.
|
||||
|
||||
This is deliberately a local heuristic rather than a claimed clinical
|
||||
emotion model. It gives the editor a useful signal about delivery shape
|
||||
while degrading predictably when acoustic extraction is unavailable.
|
||||
"""
|
||||
if not enabled:
|
||||
return {
|
||||
"emotion": "neutral",
|
||||
"emotion_confidence": 0.0,
|
||||
"arousal": 0.0,
|
||||
"valence": 0.5,
|
||||
}
|
||||
|
||||
energy = float(word.get("energy_norm", 0.0))
|
||||
pitch = float(word.get("pitch_delta", 0.0))
|
||||
rate = float(word.get("rate_delta", 0.0))
|
||||
pause = min(float(word.get("pause_before", 0.0)) / 2.0, 1.0)
|
||||
emphasis = float(word.get("emphasis", 0.0))
|
||||
|
||||
arousal = max(0.0, min(1.0, energy * 0.45 + pitch * 0.25 + rate * 0.20 + emphasis * 0.10))
|
||||
valence = max(0.0, min(1.0, 0.55 + energy * 0.15 - pause * 0.20 - rate * 0.10))
|
||||
|
||||
if arousal >= 0.68 and valence >= 0.50:
|
||||
label = "excited"
|
||||
confidence = arousal
|
||||
elif arousal >= 0.58 and valence < 0.50:
|
||||
label = "tense"
|
||||
confidence = max(arousal, 1.0 - valence)
|
||||
elif arousal <= 0.28 and pause >= 0.25:
|
||||
label = "reflective"
|
||||
confidence = max(1.0 - arousal, pause)
|
||||
elif arousal <= 0.35:
|
||||
label = "calm"
|
||||
confidence = 1.0 - arousal
|
||||
else:
|
||||
label = "neutral"
|
||||
confidence = 1.0 - abs(arousal - 0.5) * 2.0
|
||||
|
||||
confidence = max(0.0, min(1.0, confidence))
|
||||
if confidence < sensitivity:
|
||||
label = "neutral"
|
||||
return {
|
||||
"emotion": label,
|
||||
"emotion_confidence": confidence,
|
||||
"arousal": arousal,
|
||||
"valence": valence,
|
||||
}
|
||||
|
||||
|
||||
def annotate_emotions(words: Sequence[dict], enabled: bool, sensitivity: float) -> List[dict]:
|
||||
"""Attach heuristic emotion labels to enriched word rows."""
|
||||
return [
|
||||
{**w, **_emotion_for_word(w, enabled, sensitivity)}
|
||||
for w in words
|
||||
]
|
||||
|
||||
|
||||
def enrich_words(
|
||||
words: Sequence[dict],
|
||||
pitch_track: Optional[Sequence] = None,
|
||||
@@ -166,6 +237,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]
|
||||
in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end]
|
||||
energies = [w["energy_norm"] for w in in_seg]
|
||||
emphases = [w["emphasis"] for w in in_seg]
|
||||
arousals = [w.get("arousal", 0.0) for w in in_seg]
|
||||
valences = [w.get("valence", 0.5) for w in in_seg]
|
||||
emotions = [w.get("emotion", "neutral") for w in in_seg]
|
||||
dominant = max(set(emotions), key=emotions.count) if emotions else "neutral"
|
||||
emotion_confidences = [
|
||||
w.get("emotion_confidence", 0.0) for w in in_seg if w.get("emotion") == dominant
|
||||
]
|
||||
gap = max(0.0, start - previous_end)
|
||||
rows.append(
|
||||
{
|
||||
@@ -181,6 +259,13 @@ def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]
|
||||
"take_boundary": gap >= TAKE_BOUNDARY_GAP,
|
||||
"avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0,
|
||||
"peak_emphasis": round(max(emphases), 3) if emphases else 0.0,
|
||||
"emotion": dominant,
|
||||
"emotion_confidence": (
|
||||
round(sum(emotion_confidences) / len(emotion_confidences), 3)
|
||||
if emotion_confidences else 0.0
|
||||
),
|
||||
"arousal": round(sum(arousals) / len(arousals), 3) if arousals else 0.0,
|
||||
"valence": round(sum(valences) / len(valences), 3) if valences else 0.5,
|
||||
"words": [_round_word(w) for w in in_seg],
|
||||
}
|
||||
)
|
||||
@@ -425,6 +510,8 @@ def build_voice_timeline(
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
peak_percentile: float = 0.02,
|
||||
emphasis_floor: float = 0.25,
|
||||
emotion_enabled: bool = False,
|
||||
emotion_sensitivity: float = 0.5,
|
||||
progress_cb: Optional[Callable[[float, str], None]] = None,
|
||||
) -> dict:
|
||||
"""Build the consolidated voice timeline for one media file.
|
||||
@@ -445,6 +532,7 @@ def build_voice_timeline(
|
||||
|
||||
report(0.5, "Calculando ênfase...")
|
||||
words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights)
|
||||
words = annotate_emotions(words, emotion_enabled, emotion_sensitivity)
|
||||
|
||||
report(0.7, "Identificando participantes...")
|
||||
tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None
|
||||
@@ -465,6 +553,7 @@ def build_voice_timeline(
|
||||
"transcript": bool(transcript.get("words")),
|
||||
"acoustics": pitch_track is not None or energy_track is not None,
|
||||
"speakers": tracks is not None,
|
||||
"emotion": bool(emotion_enabled),
|
||||
},
|
||||
"scales": VALUE_SCALES,
|
||||
"summary": _summary(
|
||||
|
||||
+63
-2
@@ -2218,13 +2218,23 @@ class FCPXMLModifier:
|
||||
seg_start: 'TimeValue',
|
||||
seg_duration: 'TimeValue',
|
||||
) -> None:
|
||||
"""Remove markers/keywords from *clip* that fall outside the segment range.
|
||||
"""Remove markers/keywords/titles from *clip* that fall outside the segment range.
|
||||
|
||||
After ``split_clip`` deepcopy's the original clip into each segment, every
|
||||
segment inherits all child elements. Markers whose ``start`` falls outside
|
||||
``[seg_start, seg_start + seg_duration)`` are phantom duplicates and must be
|
||||
removed. Keywords that partially overlap get their ``start``/``duration``
|
||||
clamped to the segment boundaries.
|
||||
|
||||
A lane-nested ``<title>`` (a "text" voice action's on-screen callout,
|
||||
or a caption from an earlier `generate_dynamic_subtitles` pass) is
|
||||
the same kind of phantom duplicate, just keyed on ``offset`` instead
|
||||
of ``start`` — its offset lives in the same source-media coordinate
|
||||
space as a marker's ``start`` (see ``add_text_title``/``add_marker``,
|
||||
both anchored at ``parent.start``). Left unfiltered, every further
|
||||
cut (silence removal, filler removal) duplicates it into every
|
||||
resulting piece, so the same word shows up several times across the
|
||||
edited timeline instead of once where it was placed.
|
||||
"""
|
||||
seg_end = seg_start + seg_duration
|
||||
to_remove = []
|
||||
@@ -2234,6 +2244,10 @@ class FCPXMLModifier:
|
||||
child_start = TimeValue.from_timecode(child.get('start', '0s'))
|
||||
if child_start < seg_start or child_start >= seg_end:
|
||||
to_remove.append(child)
|
||||
elif tag == 'title':
|
||||
title_offset = TimeValue.from_timecode(child.get('offset', '0s'))
|
||||
if title_offset < seg_start or title_offset >= seg_end:
|
||||
to_remove.append(child)
|
||||
elif tag == 'keyword':
|
||||
kw_start = TimeValue.from_timecode(child.get('start', '0s'))
|
||||
kw_dur = TimeValue.from_timecode(child.get('duration', '0s'))
|
||||
@@ -2304,6 +2318,7 @@ class FCPXMLModifier:
|
||||
self._filter_children_for_segment(
|
||||
new_clip, current_start, segment_duration
|
||||
)
|
||||
self._reassign_text_style_ids(new_clip)
|
||||
|
||||
spine.insert(clip_index + len(new_clips), new_clip)
|
||||
new_clips.append(new_clip)
|
||||
@@ -2404,6 +2419,7 @@ class FCPXMLModifier:
|
||||
new_clip.set('start', seg_start.to_fcpxml())
|
||||
new_clip.set('duration', seg_duration.to_fcpxml())
|
||||
self._filter_children_for_segment(new_clip, seg_start, seg_duration)
|
||||
self._reassign_text_style_ids(new_clip)
|
||||
spine.insert(clip_index + len(new_clips), new_clip)
|
||||
new_clips.append(new_clip)
|
||||
current_offset = current_offset + seg_duration
|
||||
@@ -2946,6 +2962,7 @@ class FCPXMLModifier:
|
||||
('-469658744/1000000000s', '0'),
|
||||
('12328542033/1000000000s', '1'),
|
||||
)
|
||||
_TEXT_SIZE_KEY = '9999/10003/13260/3296672360/5/3296672362/3'
|
||||
|
||||
def _ensure_text_title_effect(self, resources: ET.Element) -> str:
|
||||
"""Return the resource id of the "Text" (Basic Text) effect, creating it if absent."""
|
||||
@@ -2995,6 +3012,34 @@ class FCPXMLModifier:
|
||||
self._text_style_ids.add(candidate)
|
||||
return candidate
|
||||
|
||||
def _reassign_text_style_ids(self, clip: ET.Element) -> None:
|
||||
"""Give every ``<text-style-def>`` inside a just-deepcopy'd *clip* a
|
||||
fresh document-unique id, repointing any ``<text-style ref="...">``
|
||||
in the same subtree that pointed at the old one.
|
||||
|
||||
``split_clip``/``cut_clip_ranges`` deepcopy the clip once per
|
||||
resulting segment, so a clip carrying a ``<title>`` (from a "text"
|
||||
voice action) keeps the exact same ``text-style-def id`` in every
|
||||
copy. A single cut is harmless — but the batch chain re-cuts the
|
||||
same clip at each step (silence removal, filler removal, dynamic
|
||||
subtitles), and every pass multiplies the duplicate, so the DTD
|
||||
validator eventually rejects the file with "ID ... already
|
||||
defined". Regenerating here, at the only place copies are made,
|
||||
fixes it for every caller instead of each one having to remember to.
|
||||
"""
|
||||
for style_def in clip.findall('.//text-style-def'):
|
||||
old_id = style_def.get('id')
|
||||
if not old_id:
|
||||
continue
|
||||
slug = old_id[3:] if old_id.startswith('ts_') else old_id
|
||||
slug = re.sub(r'_\d+$', '', slug) # drop a prior _<N> counter
|
||||
new_id = self._unique_text_style_id(slug)
|
||||
if new_id == old_id:
|
||||
continue
|
||||
style_def.set('id', new_id)
|
||||
for ref_el in clip.findall(f".//text-style[@ref='{old_id}']"):
|
||||
ref_el.set('ref', new_id)
|
||||
|
||||
def _make_text_title_clip(
|
||||
self,
|
||||
effect_id: str,
|
||||
@@ -3012,6 +3057,8 @@ class FCPXMLModifier:
|
||||
face: Optional[str] = None,
|
||||
kerning: Optional[float] = None,
|
||||
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
|
||||
animated: bool = True,
|
||||
size_param: Optional[float] = None,
|
||||
) -> ET.Element:
|
||||
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
|
||||
|
||||
@@ -3042,9 +3089,12 @@ class FCPXMLModifier:
|
||||
param.set('key', key)
|
||||
param.set('value', value)
|
||||
|
||||
animation_params = {'Opacity', 'Speed', 'Apply Speed'}
|
||||
for param_name, param_key, param_value in self._TEXT_TITLE_PARAMS:
|
||||
if not animated and param_name in animation_params:
|
||||
continue
|
||||
_add_param(param_name, param_key, param_value)
|
||||
if param_name == 'Speed':
|
||||
if animated and param_name == 'Speed':
|
||||
# "Custom Speed" lands between "Speed" and "Apply Speed" and
|
||||
# carries a <keyframeAnimation> child instead of a value.
|
||||
cs = ET.SubElement(elem, 'param')
|
||||
@@ -3056,6 +3106,9 @@ class FCPXMLModifier:
|
||||
kf.set('time', kf_time)
|
||||
kf.set('value', kf_value)
|
||||
|
||||
if size_param is not None:
|
||||
_add_param('Size', self._TEXT_SIZE_KEY, f"{float(size_param):g}")
|
||||
|
||||
text_el = ET.SubElement(elem, 'text')
|
||||
ts_id = self._unique_text_style_id(name)
|
||||
run = ET.SubElement(text_el, 'text-style')
|
||||
@@ -3109,6 +3162,10 @@ class FCPXMLModifier:
|
||||
font_size: int = 196,
|
||||
font_color: str = '1 1 1 1',
|
||||
bold: bool = True,
|
||||
face: Optional[str] = None,
|
||||
animated: bool = True,
|
||||
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
|
||||
size_param: Optional[float] = None,
|
||||
) -> ET.Element:
|
||||
"""Add a single static "Text" (Basic Text) title over *parent_clip*.
|
||||
|
||||
@@ -3142,6 +3199,10 @@ class FCPXMLModifier:
|
||||
font_size=font_size,
|
||||
font_color=font_color,
|
||||
bold=bold,
|
||||
face=face,
|
||||
animated=animated,
|
||||
font_scale=font_scale,
|
||||
size_param=size_param,
|
||||
)
|
||||
_dtd_insert(parent, title)
|
||||
return title
|
||||
|
||||
Reference in New Issue
Block a user