chore: atualização geral

This commit is contained in:
João Henrique
2026-08-19 16:35:29 -04:00
parent 8fca456ceb
commit e7748c2c58
66 changed files with 13037 additions and 4237 deletions
+472
View File
@@ -0,0 +1,472 @@
"""Collision detection and layout validation for dynamic-subtitle titles.
Pure functions — no I/O, no FCPXML parsing — that answer one question over and
over: given the boxes a set of titles occupy on screen, do any two titles that
are on screen at the same time intersect? And are they inside the frame, inside
the safe area, and using a font the layout actually measured?
This is the post-generation guarantee the layout engine only provides *by
construction* (``text_layout.compose_sentence`` stacks lines so their ink boxes
never touch). Re-running it over already-emitted titles catches the cases the
layout cannot see: a hand-edited position, a template whose type scales
differently than ``text_scale`` assumed, a font that fell back to an estimate,
or a word pushed off frame by a long emphasis line.
Boxes are measured in the *emitted* template space (frame pixels) — the same
numbers the writer wrote to the FCPXML (``fontSize``, ``kerning`` and
``Position`` are all already scaled by ``text_scale``), so validation re-measures
with ``measure_text``/``ink_extent`` against those same numbers and never
re-applies the scale factor. See ``writer.validate_subtitle_layout``.
"""
from dataclasses import dataclass
from math import hypot
from typing import Dict, List, Optional, Sequence
from .text_layout import (
ink_extent,
measure_text,
metrics_for,
vertical_metrics_for,
)
# Severity buckets for a spatial overlap, ordered from harmless to blocking.
# ``render_tolerance`` is the 5px the renderer can round off; ``severe`` is a
# real collision that must be fixed before export.
OVERLAP_NONE = "none"
OVERLAP_RENDER_TOLERANCE = "render_tolerance"
OVERLAP_WARNING = "warning"
OVERLAP_PROBABLE = "probable"
OVERLAP_SEVERE = "severe"
# Max fraction of the smaller box a severe collision may cover (spec 7.2).
SEVERE_OVERLAP_RATIO = 0.15
# issue types (spec 16)
SPATIAL_COLLISION = "spatial_collision"
OUTSIDE_FRAME = "outside_frame"
OUTSIDE_SAFE_AREA = "outside_safe_area"
INSUFFICIENT_SPACING = "insufficient_spacing"
EXCESSIVE_SPACING = "excessive_spacing"
FONT_MISSING = "font_missing"
FONT_TOO_SMALL = "font_too_small"
INVALID_ANCHOR = "invalid_anchor"
UNRESOLVED_TRANSFORM = "unresolved_transform"
@dataclass
class Box:
"""An axis-aligned rectangle in frame coordinates, y growing upward."""
left: float
right: float
bottom: float
top: float
@property
def width(self) -> float:
return self.right - self.left
@property
def height(self) -> float:
return self.top - self.bottom
@property
def area(self) -> float:
return self.width * self.height
def overlaps(self, other: "Box") -> bool:
"""True if the two boxes intersect (strict — touching edges do not)."""
return (
self.left < other.right
and other.left < self.right
and self.bottom < other.top
and other.bottom < self.top
)
# A boundary the writer places deliberately exact — one block's title
# duration set to literally equal the next block's start (see writer.py's
# ``block_ends``) — can still land a few float-ULPs apart by the time it
# gets here: an ``end`` re-derived as ``start + duration`` from two already-
# rounded floats isn't bit-identical to a ``start`` read as one division of
# the same exact fraction, even though both trace back to one FCPXML value.
# Found on real footage: 7 of 8 "severe" collisions in one clip were exactly
# this — same instant, off by ~1e-13s, nowhere near a real frame boundary
# (~0.04s). A tolerance many orders below one frame absorbs the artifact
# without hiding a genuine overlap.
_BOUNDARY_EPSILON = 1e-6
def temporal_overlap(
start_a: float, end_a: float, start_b: float, end_b: float
) -> bool:
"""Whether the half-open intervals ``[start, end)`` intersect (spec 7.1).
Strict on both sides, so a title that ends exactly when the next begins is
never treated as simultaneous — see ``_BOUNDARY_EPSILON`` for why "exactly"
needs a tolerance rather than bare float comparison.
"""
return (
start_a < end_b - _BOUNDARY_EPSILON
and start_b < end_a - _BOUNDARY_EPSILON
)
def overlap_metrics(a: Box, b: Box) -> Dict[str, float]:
"""Width, height, area and ratio of the intersection of ``a`` and ``b``.
``overlap_ratio`` is the shared area over the *smaller* box's area, so a
small box swallowed by a big one reads as the severe case it is.
"""
overlap_width = min(a.right, b.right) - max(a.left, b.left)
overlap_height = min(a.top, b.top) - max(a.bottom, b.bottom)
overlap_area = max(0.0, overlap_width) * max(0.0, overlap_height)
smaller = min(a.area, b.area)
ratio = overlap_area / smaller if smaller > 0 else 0.0
return {
"overlap_width": overlap_width,
"overlap_height": overlap_height,
"overlap_area": overlap_area,
"overlap_ratio": ratio,
}
def classify_overlap(metrics: Dict[str, float]) -> str:
"""Severity bucket for an overlap, following spec 7.2.
Zero area is no conflict at all; a ratio above ``SEVERE_OVERLAP_RATIO`` is
severe regardless of absolute size; otherwise the vertical penetration is
bucketed into tolerance / warning / probable / severe.
"""
height = metrics["overlap_height"]
if metrics["overlap_area"] <= 0:
return OVERLAP_NONE
if metrics["overlap_ratio"] > SEVERE_OVERLAP_RATIO:
return OVERLAP_SEVERE
if height <= 5:
return OVERLAP_RENDER_TOLERANCE
if height <= 20:
return OVERLAP_WARNING
if height <= 50:
return OVERLAP_PROBABLE
return OVERLAP_SEVERE
def distance_between(a: Box, b: Box) -> Dict[str, float]:
"""Gap between two non-overlapping boxes, per axis and euclidean (spec 8)."""
if a.right < b.left:
distance_x = b.left - a.right
elif b.right < a.left:
distance_x = a.left - b.right
else:
distance_x = 0.0
if a.top < b.bottom:
distance_y = b.bottom - a.top
elif b.top < a.bottom:
distance_y = a.bottom - b.top
else:
distance_y = 0.0
return {
"distance_x": distance_x,
"distance_y": distance_y,
"distance": hypot(distance_x, distance_y),
}
def separation_suggestion(
a: Box, b: Box, min_gap: float = 0.0
) -> Dict[str, float]:
"""The minimum translation that separates two overlapping boxes (spec 10).
Picks the smallest of the four penetrations (move left/right/up/down) and
reports that axis plus the required movement (penetration + ``min_gap``).
"""
move_left = a.right - b.left
move_right = b.right - a.left
move_down = a.top - b.bottom
move_up = b.top - a.bottom
candidates = [
("horizontal", move_left),
("horizontal", move_right),
("vertical", move_down),
("vertical", move_up),
]
axis, penetration = min(candidates, key=lambda kv: kv[1])
return {
"axis": axis,
"minimum_movement": max(0.0, penetration + min_gap),
}
def measure_title_box(
text: str,
font_size: float,
*,
x: float,
y: float,
font: Optional[str] = None,
face: Optional[str] = None,
kerning: float = 0.0,
) -> Box:
"""The on-screen box of one title, measured in the emitted template space.
``font_size``/``kerning``/``x``/``y`` are the values the writer put into the
FCPXML, so the box is comparable across every title in the document without
any further scaling. Width comes from the real advance table, vertical
extent from the real ink (accents and descenders included); the anchor is
the title's centre.
"""
width = measure_text(
text, font_size, kerning=kerning, font=font, face=face
)
top, bottom = ink_extent(text, font_size, font=font, face=face)
return Box(
left=x - width / 2,
right=x + width / 2,
bottom=y + bottom,
top=y + top,
)
def _is_font_measured(font: Optional[str], face: Optional[str]) -> bool:
"""Whether both advance and vertical metrics for ``font``/``face`` exist."""
if metrics_for(font, face) is None:
return False
_, measured = vertical_metrics_for(font, face)
return measured
def _box_within(box: Box, limits: Box) -> bool:
return (
box.left >= limits.left
and box.right <= limits.right
and box.bottom >= limits.bottom
and box.top <= limits.top
)
def _issue(severity: str, type_: str, **fields) -> Dict:
return {"severity": severity, "type": type_, **fields}
def validate_titles(
titles: Sequence[Dict],
frame_width: float,
frame_height: float,
*,
safe_margin_x: float = 0.05,
safe_margin_y: float = 0.05,
min_font_size: Optional[float] = None,
min_distance: Optional[float] = None,
max_distance: Optional[float] = None,
) -> Dict:
"""Validate a set of already-positioned titles and return a report.
Each title dict must carry the emitted values:
- ``text`` (str)
- ``font_size`` (float), ``kerning`` (float), ``x``/``y`` (floats)
- ``font`` (str) and ``face`` (str|None)
- ``start``/``end`` (seconds) for temporal overlap
- ``group`` (hashable) for spacing checks: titles sharing a group are one
block, expected to sit near each other (spec 14). Optional.
Returns ``{"severity", "issues", "summary"}`` where ``severity`` is the
worst bucket seen and ``issues`` are the spec-16-shaped occurrences.
"""
frame_left = -frame_width / 2
frame_right = frame_width / 2
frame_bottom = -frame_height / 2
frame_top = frame_height / 2
frame_box = Box(frame_left, frame_right, frame_bottom, frame_top)
safe_box = Box(
left=frame_left + safe_margin_x * frame_width,
right=frame_right - safe_margin_x * frame_width,
bottom=frame_bottom + safe_margin_y * frame_height,
top=frame_top - safe_margin_y * frame_height,
)
issues: List[Dict] = []
boxes: List[Box] = []
measured_flags: List[bool] = []
for title in titles:
text = str(title.get("text", "") or "")
font = title.get("font") or None
face = title.get("face") or None
box = measure_title_box(
text,
float(title.get("font_size", 0.0)),
x=float(title.get("x", 0.0)),
y=float(title.get("y", 0.0)),
font=font,
face=face,
kerning=float(title.get("kerning", 0.0)),
)
boxes.append(box)
measured_flags.append(_is_font_measured(font, face))
if not _is_font_measured(font, face):
issues.append(
_issue(
"warning",
FONT_MISSING,
title=text,
font=font,
face=face,
message=(
f"Font '{font or '?'}"
+ (f" {face}" if face else "")
+ "' has no embedded metrics; widths are estimated"
),
)
)
if min_font_size is not None and float(title.get("font_size", 0.0)) < min_font_size:
issues.append(
_issue(
"warning",
FONT_TOO_SMALL,
title=text,
font_size=float(title.get("font_size", 0.0)),
minimum=min_font_size,
)
)
if not _box_within(box, frame_box):
issues.append(
_issue(
"error",
OUTSIDE_FRAME,
title=text,
left=box.left,
right=box.right,
bottom=box.bottom,
top=box.top,
)
)
elif not _box_within(box, safe_box):
issues.append(
_issue(
"warning",
OUTSIDE_SAFE_AREA,
title=text,
left=box.left,
right=box.right,
bottom=box.bottom,
top=box.top,
)
)
# Spatial collisions between temporally overlapping titles.
for i in range(len(titles)):
for j in range(i + 1, len(titles)):
a, b = titles[i], titles[j]
if not temporal_overlap(
float(a.get("start", 0.0)), float(a.get("end", 0.0)),
float(b.get("start", 0.0)), float(b.get("end", 0.0)),
):
continue
box_a, box_b = boxes[i], boxes[j]
if not box_a.overlaps(box_b):
continue
metrics = overlap_metrics(box_a, box_b)
severity = classify_overlap(metrics)
issues.append(
_issue(
severity,
SPATIAL_COLLISION,
first_title=str(a.get("text", "")),
second_title=str(b.get("text", "")),
time_start=float(a.get("start", 0.0)),
time_end=float(b.get("end", 0.0)),
overlap_width=metrics["overlap_width"],
overlap_height=metrics["overlap_height"],
overlap_area=metrics["overlap_area"],
overlap_ratio=metrics["overlap_ratio"],
suggested_correction=separation_suggestion(box_a, box_b),
)
)
# Spacing within a block (spec 8/14). Only when the caller asked for it —
# a generic minimum can fire on the reference look's own tight stacking.
if min_distance is not None or max_distance is not None:
groups: Dict = {}
for index, title in enumerate(titles):
groups.setdefault(title.get("group", index), []).append(index)
for members in groups.values():
for m in range(len(members)):
for n in range(m + 1, len(members)):
i, j = members[m], members[n]
box_a, box_b = boxes[i], boxes[j]
if box_a.overlaps(box_b):
continue
gap = distance_between(box_a, box_b)["distance"]
if min_distance is not None and gap < min_distance:
issues.append(
_issue(
"warning",
INSUFFICIENT_SPACING,
first_title=str(titles[i].get("text", "")),
second_title=str(titles[j].get("text", "")),
distance=gap,
minimum=min_distance,
)
)
if max_distance is not None and gap > max_distance:
issues.append(
_issue(
"warning",
EXCESSIVE_SPACING,
first_title=str(titles[i].get("text", "")),
second_title=str(titles[j].get("text", "")),
distance=gap,
maximum=max_distance,
)
)
_rank = {
OVERLAP_NONE: 0,
OVERLAP_RENDER_TOLERANCE: 1,
OVERLAP_WARNING: 2,
OVERLAP_PROBABLE: 3,
OVERLAP_SEVERE: 4,
}
severities = [issue["severity"] for issue in issues]
worst = max(severities, key=lambda s: _rank.get(s, 0), default=OVERLAP_NONE)
return {
"severity": worst,
"issues": issues,
"summary": {
"title_count": len(titles),
"issue_count": len(issues),
"spatial_collision": sum(
1 for i in issues if i["type"] == SPATIAL_COLLISION
),
"outside_frame": sum(
1 for i in issues if i["type"] == OUTSIDE_FRAME
),
"outside_safe_area": sum(
1 for i in issues if i["type"] == OUTSIDE_SAFE_AREA
),
"font_missing": sum(
1 for i in issues if i["type"] == FONT_MISSING
),
"font_too_small": sum(
1 for i in issues if i["type"] == FONT_TOO_SMALL
),
"insufficient_spacing": sum(
1 for i in issues if i["type"] == INSUFFICIENT_SPACING
),
"excessive_spacing": sum(
1 for i in issues if i["type"] == EXCESSIVE_SPACING
),
},
}
def blocking(severity: str) -> bool:
"""Whether a validation severity should block export (spec 16)."""
return severity in (OVERLAP_SEVERE, OVERLAP_PROBABLE)
+31 -1
View File
@@ -37,6 +37,36 @@ def diarization_capability(token: Optional[str]) -> Tuple[bool, str]:
return True, "Identificação de participantes disponível."
def _load_waveform(path: str) -> Optional[dict]:
"""Decode ``path`` ourselves into the waveform dict pyannote accepts.
pyannote 4.x decodes audio through torchcodec, which links against a
specific FFmpeg major version and fails outright when the installed one
differs (``libavutil.56.dylib`` not found) — taking diarization down on
an otherwise working machine. Handing it an already-decoded waveform
skips that path entirely and reuses the ffmpeg extraction the acoustic
analysis already relies on, so video containers work too.
Returns ``None`` when decoding is not possible, letting the caller fall
back to passing the path and whatever pyannote can do with it.
"""
try:
import soundfile
import torch
from .voice_features import decodable_audio
with decodable_audio(path) as audio_path:
if audio_path is None:
return None
data, sample_rate = soundfile.read(audio_path, dtype="float32", always_2d=True)
# soundfile gives (samples, channels); pyannote wants (channels, samples)
return {"waveform": torch.from_numpy(data.T), "sample_rate": int(sample_rate)}
except Exception:
logger.info("could not pre-decode %s for diarization", path)
return None
def diarize(
path: str,
token: Optional[str],
@@ -67,7 +97,7 @@ def diarize(
n = str(num_speakers or "").strip()
if n.isdigit() and int(n) > 0:
kwargs["num_speakers"] = int(n)
result = pipe(path, **kwargs)
result = pipe(_load_waveform(path) or path, **kwargs)
# pyannote.audio >= 4.0 wraps the annotation; normalize to the raw one.
if hasattr(result, "exclusive_speaker_diarization"):
result = result.exclusive_speaker_diarization
+133
View File
@@ -0,0 +1,133 @@
"""Emphasis index — how much a spoken word "pops" acoustically.
Pure functions over already-extracted per-word features (energy, pitch
delta, rate delta, pause before, duration); no I/O, no external dependency.
Combines them into a single ``[0, 1]`` score, configurable via
:class:`EmphasisWeights` so the weighting can be tuned (and persisted,
see ``model_manager.load_voice_analysis_config``) without touching code.
"""
from dataclasses import dataclass
from typing import List, Sequence
_FIELDS = ("energy", "pitch_variation", "rate_variation", "pause_before", "duration")
@dataclass
class EmphasisWeights:
energy: float = 0.30
pitch_variation: float = 0.25
rate_variation: float = 0.20
pause_before: float = 0.15
duration: float = 0.10
def as_dict(self) -> dict:
return {field: getattr(self, field) for field in _FIELDS}
@classmethod
def from_dict(cls, data: dict) -> "EmphasisWeights":
defaults = cls()
return cls(**{field: float(data.get(field, getattr(defaults, field))) for field in _FIELDS})
def _clamp01(x: float) -> float:
return max(0.0, min(1.0, x))
def pause_weight(
pause_before: float, max_pause: float = 1.5, ignore_above: float = 3.0
) -> float:
"""How much a preceding silence counts as emphasis, in ``[0, 1]``.
A short beat before a word is real emphasis: the speaker is setting it
up. A *long* gap is not — it is an edit point, a B-roll insert, or the
other person in the room talking. Measured on real footage, gaps of
6-9s were scoring as the most emphatic moments in the recording purely
because the scale saturated, ranking a scene change above a word the
speaker actually hit hard.
So the contribution rises up to ``max_pause`` and then drops to zero
past ``ignore_above``, instead of saturating. Set ``ignore_above`` to
``0`` to disable the cutoff and keep the old saturating behaviour.
"""
if pause_before <= 0 or max_pause <= 0:
return 0.0
if ignore_above > 0 and pause_before > ignore_above:
return 0.0
return _clamp01(pause_before / max_pause)
def compute_emphasis(
energy: float,
pitch_delta: float,
rate_delta: float,
pause_before: float,
word_duration: float,
weights: EmphasisWeights = EmphasisWeights(),
*,
max_pause: float = 1.5,
max_duration: float = 1.0,
pause_ignore_above: float = 3.0,
) -> float:
"""Emphasis score in ``[0, 1]`` for one word.
``energy``/``pitch_delta``/``rate_delta`` are expected already
normalized to roughly ``[0, 1]`` (deltas may be negative — only their
magnitude counts as emphasis). ``pause_before``/``word_duration`` are
raw seconds; duration saturates at ``max_duration``, while the pause
contribution is shaped by :func:`pause_weight`.
"""
energy_n = _clamp01(energy)
pitch_n = _clamp01(abs(pitch_delta))
rate_n = _clamp01(abs(rate_delta))
pause_n = pause_weight(pause_before, max_pause, pause_ignore_above)
duration_n = _clamp01(word_duration / max_duration) if max_duration > 0 else 0.0
total_weight = sum(getattr(weights, field) for field in _FIELDS)
if total_weight <= 0:
return 0.0
score = (
weights.energy * energy_n
+ weights.pitch_variation * pitch_n
+ weights.rate_variation * rate_n
+ weights.pause_before * pause_n
+ weights.duration * duration_n
)
return _clamp01(score / total_weight)
def annotate_emphasis(
words: Sequence[dict],
weights: EmphasisWeights = EmphasisWeights(),
*,
max_pause: float = 1.5,
max_duration: float = 1.0,
pause_ignore_above: float = 3.0,
) -> List[dict]:
"""Return copies of ``words`` with an ``"emphasis"`` key added.
Each word dict is expected to carry ``energy``, ``pitch_delta``,
``rate_delta``, ``pause_before`` (all pre-computed, e.g. by
``voice_features.py``), plus ``start``/``end`` — or an explicit
``duration`` — to derive word length.
"""
out: List[dict] = []
for w in words:
ww = dict(w)
duration = ww.get("duration")
if duration is None:
duration = max(0.0, float(ww.get("end", 0.0)) - float(ww.get("start", 0.0)))
ww["emphasis"] = compute_emphasis(
energy=float(ww.get("energy") or 0.0),
pitch_delta=float(ww.get("pitch_delta") or 0.0),
rate_delta=float(ww.get("rate_delta") or 0.0),
pause_before=float(ww.get("pause_before") or 0.0),
word_duration=float(duration),
weights=weights,
max_pause=max_pause,
max_duration=max_duration,
pause_ignore_above=pause_ignore_above,
)
out.append(ww)
return out
+288
View File
@@ -359,3 +359,291 @@ def save_num_speakers(num: str) -> str:
data["num_speakers"] = val
_write_config(data)
return val
DEFAULT_VOICE_ANALYSIS_CONFIG: dict = {
"energy_threshold": 0.5,
"emphasis_weights": {
"energy": 0.30,
"pitch_variation": 0.25,
"rate_variation": 0.20,
"pause_before": 0.15,
"duration": 0.10,
},
# Peaks are selected RELATIVELY — the top slice of the distribution —
# because the emphasis index is a weighted average whose real range
# depends on the material. Measured on a 17-minute interview the index
# never passed 0.55, so any absolute cutoff near the spec's 0.85 selects
# nothing; on punchier material the same cutoff would flood the edit.
# 2% of words is roughly one highlight every 50 words.
"peak_percentile": 0.02,
# Guard for genuinely flat audio, where even the top of the distribution
# carries no emphasis worth cutting on.
"emphasis_floor": 0.25,
"emotion_enabled": False,
"emotion_sensitivity": 0.5,
}
def load_voice_analysis_config() -> dict:
"""The persisted voice-analysis thresholds/weights, merged over defaults.
Backs the "Análise de Voz" settings screen: energy threshold (how loud
counts as "high energy"), the emphasis-index weights (see
``emphasis.EmphasisWeights``), the punch-in emphasis cutoff, and the
emotion-detection toggle/sensitivity. Unknown/malformed stored values
fall back to the default rather than raising, so a hand-edited or
partially-written config.json never breaks the settings screen.
"""
cfg = {
**DEFAULT_VOICE_ANALYSIS_CONFIG,
"emphasis_weights": dict(DEFAULT_VOICE_ANALYSIS_CONFIG["emphasis_weights"]),
}
stored = _load_config().get("voice_analysis")
if not isinstance(stored, dict):
return cfg
for key in ("energy_threshold", "peak_percentile", "emphasis_floor", "emotion_sensitivity"):
if key in stored:
try:
cfg[key] = max(0.0, min(1.0, float(stored[key])))
except (TypeError, ValueError):
pass
if "emotion_enabled" in stored:
cfg["emotion_enabled"] = bool(stored["emotion_enabled"])
weights = stored.get("emphasis_weights")
if isinstance(weights, dict):
for key in cfg["emphasis_weights"]:
if key in weights:
try:
cfg["emphasis_weights"][key] = max(0.0, float(weights[key]))
except (TypeError, ValueError):
pass
return cfg
def save_voice_analysis_config(
energy_threshold: float | None = None,
emphasis_weights: dict | None = None,
peak_percentile: float | None = None,
emphasis_floor: float | None = None,
emotion_enabled: bool | None = None,
emotion_sensitivity: float | None = None,
) -> dict:
"""Persist voice-analysis thresholds/weights. Only given fields change.
Returns the full merged config (same shape as
:func:`load_voice_analysis_config`) so callers can render it back
immediately without a second round-trip.
"""
cfg = load_voice_analysis_config()
if energy_threshold is not None:
cfg["energy_threshold"] = max(0.0, min(1.0, float(energy_threshold)))
if peak_percentile is not None:
cfg["peak_percentile"] = max(0.0, min(1.0, float(peak_percentile)))
if emphasis_floor is not None:
cfg["emphasis_floor"] = max(0.0, min(1.0, float(emphasis_floor)))
if emotion_enabled is not None:
cfg["emotion_enabled"] = bool(emotion_enabled)
if emotion_sensitivity is not None:
cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity)))
if emphasis_weights is not None:
for key, value in emphasis_weights.items():
if key in cfg["emphasis_weights"] and value is not None:
cfg["emphasis_weights"][key] = max(0.0, float(value))
data = _load_config()
data["voice_analysis"] = cfg
_write_config(data)
return cfg
# Mirrors the "Legendas Dinâmicas" tab's own defaults (MacApp/Sources/
# CaptionsView.swift), so a fresh install shows the same look in the UI and
# in what generate_dynamic_subtitles renders when no override is passed.
DEFAULT_DYNAMIC_SUBTITLE_CONFIG: dict = {
"band_height": 0.22,
"block_center_y": -167.0,
"line_gap": 8.0,
"font": "Helvetica Neue",
"font_size": 104,
"emphasis_font": "Playfair Display",
"emphasis_face": "Medium Italic",
"emphasis_size": 265,
"active_color": "1 1 1 1",
"emphasis_color": "1 1 1 1",
"text_scale": 2.0,
}
def load_dynamic_subtitle_config() -> dict:
"""The persisted dynamic-subtitle style, merged over defaults.
Backs the "Legendas Dinâmicas" settings screen AND is the fallback
``generate_dynamic_subtitles`` reads for any field the caller doesn't
explicitly override — so the style configured in the UI is what actually
renders, without the app having to thread every field through each call.
Unknown/malformed stored values fall back to the default, same as
:func:`load_voice_analysis_config`.
"""
cfg = dict(DEFAULT_DYNAMIC_SUBTITLE_CONFIG)
stored = _load_config().get("dynamic_subtitles")
if not isinstance(stored, dict):
return cfg
for key in ("band_height", "block_center_y", "line_gap", "text_scale"):
if key in stored:
try:
cfg[key] = float(stored[key])
except (TypeError, ValueError):
pass
for key in ("font_size", "emphasis_size"):
if key in stored:
try:
cfg[key] = int(stored[key])
except (TypeError, ValueError):
pass
for key in ("font", "emphasis_font", "emphasis_face", "active_color", "emphasis_color"):
if key in stored and isinstance(stored[key], str) and stored[key]:
cfg[key] = stored[key]
return cfg
def save_dynamic_subtitle_config(**fields) -> dict:
"""Persist dynamic-subtitle style fields. Only given fields change.
Accepts the same keys as :data:`DEFAULT_DYNAMIC_SUBTITLE_CONFIG`; unknown
keys are ignored so a newer app talking to an older config shape degrades
quietly. Returns the full merged config, mirroring
:func:`save_voice_analysis_config`.
"""
cfg = load_dynamic_subtitle_config()
for key, value in fields.items():
if key not in DEFAULT_DYNAMIC_SUBTITLE_CONFIG or value is None:
continue
if isinstance(DEFAULT_DYNAMIC_SUBTITLE_CONFIG[key], float):
try:
cfg[key] = float(value)
except (TypeError, ValueError):
continue
elif isinstance(DEFAULT_DYNAMIC_SUBTITLE_CONFIG[key], int):
try:
cfg[key] = int(value)
except (TypeError, ValueError):
continue
else:
cfg[key] = str(value)
data = _load_config()
data["dynamic_subtitles"] = cfg
_write_config(data)
return cfg
# Mirrors the silence thresholds the detection/removal handlers use when no
# argument is passed (server_tools/qc.py). Persisted so the app's slider and
# any later run agree without threading three fields through every call.
DEFAULT_SILENCE_CONFIG: dict = {
# dBFS below which audio counts as silence.
"noise_db": -30.0,
# Seconds a quiet stretch must last before it's a cut candidate.
"min_silence": 0.5,
# Seconds left inside each cut so speech never gets clipped at the edges.
"padding": 0.05,
}
def load_silence_config() -> dict:
"""The persisted silence-detection thresholds, merged over defaults.
Read by ``detect_media_silence``/``remove_media_silence`` as their
fallback, so the tolerance chosen in the app is what actually runs.
Malformed stored values fall back to the default rather than raising,
matching :func:`load_voice_analysis_config`.
"""
cfg = dict(DEFAULT_SILENCE_CONFIG)
stored = _load_config().get("silence")
if not isinstance(stored, dict):
return cfg
for key in cfg:
if key in stored:
try:
cfg[key] = float(stored[key])
except (TypeError, ValueError):
pass
return cfg
def save_silence_config(
noise_db: float | None = None,
min_silence: float | None = None,
padding: float | None = None,
) -> dict:
"""Persist silence thresholds. Only the given fields change.
Values are clamped to the same ranges the handlers validate against, so
a bad write here can't produce a config the tools would later reject.
"""
cfg = load_silence_config()
if noise_db is not None:
try:
cfg["noise_db"] = max(-120.0, min(0.0, float(noise_db)))
except (TypeError, ValueError):
pass
if min_silence is not None:
try:
cfg["min_silence"] = max(0.01, min(3600.0, float(min_silence)))
except (TypeError, ValueError):
pass
if padding is not None:
try:
cfg["padding"] = max(0.0, min(5.0, float(padding)))
except (TypeError, ValueError):
pass
data = _load_config()
data["silence"] = cfg
_write_config(data)
return cfg
# Last project worked on, so the app reopens where the user left off instead of
# making them pick the folder again every launch. Only paths that still exist
# are handed back — a project on an unmounted volume degrades to "none selected"
# rather than to a dead path the tools would later fail on.
DEFAULT_PROJECT_CONFIG: dict = {
# Folder every generated file (transcript .json, XML, SRT) is written to.
"folder": "",
# The .fcpxml/.fcpxmld that was loaded from it.
"file": "",
}
def load_project_config() -> dict:
"""The persisted last project (folder + file), merged over defaults.
Paths that no longer exist on disk come back empty, matching what the app
shows for "nothing selected". Malformed stored values fall back to the
default rather than raising, same as :func:`load_voice_analysis_config`.
"""
cfg = dict(DEFAULT_PROJECT_CONFIG)
stored = _load_config().get("project")
if not isinstance(stored, dict):
return cfg
for key in cfg:
value = stored.get(key)
if isinstance(value, str) and value and Path(value).exists():
cfg[key] = value
return cfg
def save_project_config(folder: str | None = None, file: str | None = None) -> dict:
"""Persist the last project folder/file. Only the given fields change.
Passing an empty string clears a field (the app does this when the user
deselects), while ``None`` leaves it untouched.
"""
cfg = load_project_config()
for key, value in (("folder", folder), ("file", file)):
if value is None:
continue
cfg[key] = str(Path(value).expanduser()) if str(value).strip() else ""
data = _load_config()
data["project"] = cfg
_write_config(data)
return cfg
+18
View File
@@ -13,6 +13,8 @@ from functools import total_ordering
from math import gcd
from typing import Any, Callable, Dict, List, Optional, Tuple
from .text_layout import REFERENCE_BLOCK_LINE_GAP, TEXT_TEMPLATE_FONT_SCALE
# ============================================================================
# ENUMS
# ============================================================================
@@ -1071,3 +1073,19 @@ class DynamicSubtitleConfig:
# grouped, the key word alone and large (the reference look). "word": one
# title per word, the earlier rhythm.
granularity: str = "phrase"
# Ratio between the template's fontSize space and the canvas-point space
# its Position uses. See text_layout.TEXT_TEMPLATE_FONT_SCALE: the "Text"
# (Text.moti) template sizes type in frame pixels, so a size chosen in
# points renders half as large unless it is converted on the way out.
text_scale: float = TEXT_TEMPLATE_FONT_SCALE
# Vertical air between stacked lines, in canvas points. Negative values
# deliberately overlap the lines — the display italic tucking under the
# line above is a real editorial look, and the stacking arithmetic places
# ink boxes edge to edge, so a negative gap moves them by exactly that
# much rather than colliding unpredictably.
line_gap: float = REFERENCE_BLOCK_LINE_GAP
# Run the post-generation collision validation (collision.validate_titles)
# and refuse to emit when it reports a blocking overlap. Off by default so
# generation stays byte-identical to before this flag existed; flip it on
# for a guaranteed no-collision export.
validate: bool = False
+91 -16
View File
@@ -122,6 +122,33 @@ REFERENCE_BLOCK_LINE_GAP = 8.0
# emphasis line's edge; the reference leaves a little air.
REFERENCE_STAGGER_RATIO = 0.8
# Extra gap, as a fraction of the emphasis line's font size, added only to
# the boundary right below it. The display italic's slant leans its stems
# past the vertical ink box the metrics measure, so a body line directly
# under the emphasis line reads tighter than the same nominal gap anywhere
# else in the stack — this cushion (~14pt at the 230pt reference size)
# closes that optical gap without touching the user's `line_gap` elsewhere.
_EMPHASIS_ITALIC_CUSHION_RATIO = 0.06
# The numbers above were read off a hand export that used the "Essencial -
# Título" template. That template never rendered when we generated it (see
# Engine/docs/05_EXPERIENCIAS.md, 2026-08-17), so the writer switched to FCP's
# own "Basic Text > Text" (Text.moti) — whose coordinate space is the FRAME
# ITSELF (2160x3840), not the half-scale point canvas the numbers above were
# measured in. Everything the template reads is in that space: fontSize,
# kerning AND Position alike.
#
# Getting this half-right is worse than getting it wrong. Scaling only the type
# left the block at the old spread with twice the type in it, so the lines
# collided; scaling only the positions would spread a block of half-size type
# across the frame. The layout keeps measuring in canvas points — every
# constant above depends on that — and this single factor converts the whole
# result on the way out, which is the only way the two stay in step.
#
# Exposed as `text_scale` on DynamicSubtitleConfig for a template authored
# against a different space.
TEXT_TEMPLATE_FONT_SCALE = 2.0
def metrics_for(font: Optional[str], face: Optional[str] = None) -> Optional[Dict]:
"""The embedded advance table for *font*/*face*, or None if uncovered.
@@ -296,9 +323,14 @@ class PlacedWord:
and other.bottom < self.top
)
def position_param(self) -> str:
"""The value for the title's "Posição" param, as FCP writes it."""
return f"{self.x:g} {self.y:g}"
def position_param(self, scale: float = 1.0) -> str:
"""The value for the title's "Posição" param, as FCP writes it.
*scale* converts from canvas points to the template's own space; see
TEXT_TEMPLATE_FONT_SCALE. It must be the same factor the emitted
fontSize uses, or the type and the spacing drift apart.
"""
return f"{self.x * scale:g} {self.y * scale:g}"
# The rest of this block is the interface a placed unit shares with
# PlacedBlock, so the writer emits titles from either without caring
@@ -649,9 +681,14 @@ class PlacedBlock:
"""When this block finishes being spoken, in seconds."""
return max(float(w.get('end', 0.0)) for w in self.words)
def position_param(self) -> str:
"""The value for the title's "Posição" param, as FCP writes it."""
return f"{self.x:g} {self.y:g}"
def position_param(self, scale: float = 1.0) -> str:
"""The value for the title's "Posição" param, as FCP writes it.
*scale* converts from canvas points to the template's own space; see
TEXT_TEMPLATE_FONT_SCALE. It must be the same factor the emitted
fontSize uses, or the type and the spacing drift apart.
"""
return f"{self.x * scale:g} {self.y * scale:g}"
def overlaps(self, other: 'PlacedBlock') -> bool:
return (
@@ -742,10 +779,33 @@ def compose_sentence(
for run in body_lines(entries[emphasis_index + 1:]):
lines.append((run, body_look, False))
# A body run wraps onto a new line when it doesn't fit — but the
# emphasis line is always exactly one word, so it can't wrap, and
# nothing capped its size against the box. A long or all-caps word (an
# emphasis pass sometimes upper-cases its pick) could run past both
# edges of the frame — found on real footage, wide enough to spill off
# BOTH sides while centred. Shrinking it back to the box scales its
# font_size and kerning by the same factor, so the ink height used for
# stacking below shrinks with it too — restoring the vertical
# non-overlap the rest of this function already guarantees by
# construction. Never shrunk below the body size: emphasis smaller
# than body text isn't emphasis anymore, it's just a different font.
def fit_emphasis(text: str, look) -> tuple:
size, kerning, width = measure(text, look)
if width <= box.width:
return size, kerning, width
floor = float(body_look.font_size) * scale
fit = max(box.width / width, floor / size) if size > 0 else 1.0
fit = min(fit, 1.0)
return size * fit, kerning * fit, width * fit
measured = []
for run, look, is_emphasis in lines:
text = ' '.join(t for _, t in run)
size, kerning, width = measure(text, look)
if is_emphasis:
size, kerning, width = fit_emphasis(text, look)
else:
size, kerning, width = measure(text, look)
# Stack on the real ink each line contains, not on a nominal
# cap-height: the display italic's accents and descenders run well
# past it, and a nominal box lets them collide with the neighbour.
@@ -764,14 +824,29 @@ def compose_sentence(
# loses the whole point of the look — so if it does not fit, everything
# from the emphasis on overflows together.
gap = line_gap * scale
# The emphasis line's italic slant carries visual weight below its own
# ink box — Playfair's stems lean past what the vertical metrics measure
# — so a body line sitting right under it reads tighter than the same
# nominal gap elsewhere, even though the ink boxes themselves never
# touch. Add a size-proportional cushion only to the boundary right
# after the emphasis line; every other pair keeps exactly the caller's
# ``line_gap``.
def pair_gap(prev_line: dict) -> float:
if prev_line['emphasis']:
return gap + _EMPHASIS_ITALIC_CUSHION_RATIO * prev_line['font_size']
return gap
kept = 0
total = 0.0
prev = None
for line in measured:
advance = line['height'] if not kept else line['height'] + gap
if kept and total + advance > box.height:
advance = line['height'] if prev is None else line['height'] + pair_gap(prev)
if prev is not None and total + advance > box.height:
break
total += advance
kept += 1
prev = line
kept = max(kept, 1)
if not any(line['emphasis'] for line in measured[:kept]):
kept = min(kept, next(
@@ -783,12 +858,12 @@ def compose_sentence(
result.overflow.extend(w for w, _ in line['run'])
visible = measured[:kept]
# Stack the ink boxes edge to edge with exactly *gap* between them, then
# centre the whole stack on the band. Because the boxes are the real ink,
# "no overlap" is a property of the arithmetic, not of a safety factor.
stack_height = (
sum(line['height'] for line in visible) + gap * (len(visible) - 1)
)
# Stack the ink boxes edge to edge with exactly *gap* between them (plus
# the emphasis cushion where it applies), then centre the whole stack on
# the band. Because the boxes are the real ink, "no overlap" is a
# property of the arithmetic, not of a safety factor.
gaps = [pair_gap(visible[i - 1]) for i in range(1, len(visible))]
stack_height = sum(line['height'] for line in visible) + sum(gaps)
edge = box.center_y + stack_height / 2
# Body lines hang off the emphasis line's edges, alternating sides in
@@ -797,7 +872,7 @@ def compose_sentence(
side = -1
for index, line in enumerate(visible):
if index:
edge -= gap
edge -= gaps[index - 1]
cursor_y = edge - line['ink_top']
edge = cursor_y + line['ink_bottom']
if line['emphasis']:
+11 -2
View File
@@ -16,7 +16,7 @@ import logging
import os
import re
from pathlib import Path
from typing import List, Optional, Sequence, Tuple
from typing import Callable, List, Optional, Sequence, Tuple
logger = logging.getLogger(__name__)
@@ -118,7 +118,10 @@ def invert_ranges(
def transcribe(
path: str, model_size: str = "base", language: Optional[str] = None
path: str,
model_size: str = "base",
language: Optional[str] = None,
progress_cb: Optional[Callable[[float], None]] = None,
) -> Optional[dict]:
"""Transcribe an audio/video file locally with word-level timestamps.
@@ -170,6 +173,10 @@ def transcribe(
)
segments: List[dict] = []
words: List[dict] = []
# `info.duration` is known upfront (from the container), so each
# segment's end time — yielded lazily as faster-whisper decodes —
# gives real, granular progress instead of a single before/after step.
total_duration = float(info.duration) if info.duration else 0.0
for seg in segments_iter:
start = float(seg.start)
end = float(seg.end)
@@ -182,6 +189,8 @@ def transcribe(
"end_fmt": format_timestamp(end),
}
)
if progress_cb is not None and total_duration > 0:
progress_cb(min(end / total_duration, 1.0))
for w in seg.words or []:
ws = float(w.start)
we = float(w.end)
+248
View File
@@ -0,0 +1,248 @@
"""Voice actions — the editing decisions produced from a voice timeline.
This is the contract between *deciding* and *applying*. Whoever makes the
editorial call — the deterministic rules engine, or a model reading the
voice timeline JSON — emits the same list of actions, and one applier turns
it into FCPXML. Nothing that produces actions ever touches XML.
Every action's ``start``/``end`` is in **original source seconds**, matching
the voice timeline. That matters: cuts shift everything after them, so if
decisions were expressed in post-cut time they would silently land in the
wrong place the moment a cut was added. Keeping one origin and resolving the
shift at apply time (:func:`shift_after_cuts`) removes that whole class of bug.
Actions arriving from a model are untrusted input: :func:`parse_actions`
validates and reports what it rejected rather than raising, so one malformed
row never discards a whole edit.
"""
from dataclasses import dataclass, field
from typing import Any, List, Optional, Sequence, Tuple
# What an action can ask for. Deliberately small — each maps onto one
# existing writer capability, so no new XML knowledge lives here.
ACTION_KINDS = ("cut", "zoom", "text", "marker")
# Bounds for a zoom's scale factor. Below 1.0 is a pull-back, not a punch-in;
# above 3x the image falls apart on any normal footage.
MIN_ZOOM_SCALE = 1.0
MAX_ZOOM_SCALE = 3.0
MAX_TEXT_LENGTH = 120
@dataclass
class VoiceAction:
"""One editing decision, in original source time."""
kind: str
start: float
end: float
params: dict = field(default_factory=dict)
reason: str = ""
speaker: str = ""
@property
def duration(self) -> float:
return max(0.0, self.end - self.start)
def as_dict(self) -> dict:
return {
"kind": self.kind,
"start": round(self.start, 3),
"end": round(self.end, 3),
"params": self.params,
"reason": self.reason,
"speaker": self.speaker,
}
def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]:
"""Turn one raw row into a VoiceAction, or explain why it can't be."""
where = f"action[{index}]"
if not isinstance(raw, dict):
return None, f"{where}: expected an object, got {type(raw).__name__}"
kind = str(raw.get("kind", "")).strip().lower()
if kind not in ACTION_KINDS:
return None, f"{where}: unknown kind {raw.get('kind')!r} (expected one of {', '.join(ACTION_KINDS)})"
try:
start = float(raw.get("start"))
end = float(raw.get("end"))
except (TypeError, ValueError):
return None, f"{where}: start/end must be numbers (seconds)"
if start < 0:
return None, f"{where}: start is negative ({start})"
if end <= start:
return None, f"{where}: end ({end}) must be after start ({start})"
params = raw.get("params")
params = dict(params) if isinstance(params, dict) else {}
if kind == "zoom":
try:
scale = float(params.get("scale", 1.3))
except (TypeError, ValueError):
return None, f"{where}: zoom scale must be a number"
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
return None, (
f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}"
)
params["scale"] = scale
if kind == "text":
content = str(params.get("content", "")).strip()
if not content:
return None, f"{where}: text action needs params.content"
params["content"] = content[:MAX_TEXT_LENGTH]
return (
VoiceAction(
kind=kind,
start=start,
end=end,
params=params,
reason=str(raw.get("reason", "")),
speaker=str(raw.get("speaker", "")),
),
"",
)
def parse_actions(data: Any) -> Tuple[List[VoiceAction], List[str]]:
"""Validate a decision list into actions, collecting rejections.
Accepts either a bare list of actions or ``{"actions": [...]}`` — the
shape a model is most likely to return. Returns ``(actions, errors)``;
a row that fails validation is reported and skipped, never fatal.
"""
if isinstance(data, dict):
data = data.get("actions", [])
if not isinstance(data, Sequence) or isinstance(data, (str, bytes)):
return [], ["expected a list of actions, or an object with an 'actions' list"]
actions: List[VoiceAction] = []
errors: List[str] = []
for i, raw in enumerate(data):
action, error = _validate_one(raw, i)
if action is not None:
actions.append(action)
else:
errors.append(error)
return actions, errors
def speaker_cut_actions(
timeline: dict,
speaker_ids: Sequence[str],
padding: float = 0.15,
) -> List[VoiceAction]:
"""Cut actions removing everything the given speakers say.
The everyday case on a testimonial shoot: an interviewer or a crew
member talks over the take, and only the subject should survive the
edit. ``padding`` trims slightly *inside* each segment rather than
around it — speech boundaries from a transcript are approximate, and
eating into the neighbouring silence is far safer than clipping the
first syllable of the person being kept.
"""
wanted = {str(s) for s in speaker_ids}
actions: List[VoiceAction] = []
for segment in timeline.get("segments", []):
if str(segment.get("speaker", "")) not in wanted:
continue
start = float(segment.get("start", 0.0)) + padding
end = float(segment.get("end", 0.0)) - padding
if end <= start:
continue
actions.append(
VoiceAction(
kind="cut",
start=start,
end=end,
reason=f"fala de {segment.get('speaker')}",
speaker=str(segment.get("speaker", "")),
)
)
return actions
def merge_cut_ranges(actions: Sequence[VoiceAction]) -> List[Tuple[float, float]]:
"""The cut actions as merged, sorted, non-overlapping source ranges."""
cuts = sorted((a.start, a.end) for a in actions if a.kind == "cut")
merged: List[Tuple[float, float]] = []
for start, end in cuts:
if merged and start <= merged[-1][1]:
merged[-1] = (merged[-1][0], max(merged[-1][1], end))
else:
merged.append((start, end))
return merged
def shift_after_cuts(
time: float, cuts: Sequence[Tuple[float, float]]
) -> Optional[float]:
"""Where source ``time`` lands once ``cuts`` are removed.
Returns ``None`` when the time falls *inside* a cut — the material it
referred to no longer exists, so the action that pointed at it must be
dropped rather than silently slid onto neighbouring content.
``cuts`` must be merged and sorted (see :func:`merge_cut_ranges`).
"""
shift = 0.0
for start, end in cuts:
if time < start:
break
if time < end:
return None
shift += end - start
return time - shift
def resolve_actions(
actions: Sequence[VoiceAction],
) -> Tuple[List[Tuple[float, float]], List[VoiceAction], List[VoiceAction]]:
"""Split a decision list into what to cut and what to place afterwards.
Returns ``(cut_ranges, placed, dropped)``. Non-cut actions are moved onto
their post-cut times; any that pointed into removed material land in
``dropped`` so the caller can report them instead of losing them quietly.
"""
cut_ranges = merge_cut_ranges(actions)
placed: List[VoiceAction] = []
dropped: List[VoiceAction] = []
for action in actions:
if action.kind == "cut":
continue
new_start = shift_after_cuts(action.start, cut_ranges)
if new_start is None:
dropped.append(action)
continue
if action.kind == "marker":
# A marker is a point, not a span: it survives as long as its own
# instant does. Requiring its nominal end to survive too would
# drop exactly the markers worth keeping — the ones flagging a
# join, which sit right against a cut edge by definition.
new_end = new_start + action.duration
else:
new_end = shift_after_cuts(action.end, cut_ranges)
if new_end is None:
dropped.append(action)
continue
if new_end <= new_start:
dropped.append(action)
continue
placed.append(
VoiceAction(
kind=action.kind,
start=new_start,
end=new_end,
params=action.params,
reason=action.reason,
speaker=action.speaker,
)
)
return cut_ranges, placed, dropped
+220
View File
@@ -0,0 +1,220 @@
"""Acoustic features for voice analysis — pitch, energy, rate, pauses.
Mirrors the ``media_intel.py`` contract: librosa is an optional dependency
(``pip install 'fcp-mcp-server[intelligence]'``, already required by beat
detection), imported lazily, and every extractor degrades to ``None`` when
the library is missing or the file cannot be analyzed — never crashes.
``compute_speech_rate``/``compute_pauses`` are pure functions over
word-timestamp dicts (the shape ``transcribe.py`` already produces) and need
no audio file at all.
"""
import contextlib
import logging
import shutil
import subprocess
import tempfile
from pathlib import Path
from typing import Iterator, List, Optional, Sequence, Tuple
logger = logging.getLogger(__name__)
# Human voice fundamental frequency range (covers low male to high female/child).
PITCH_FMIN_HZ = 65.0
PITCH_FMAX_HZ = 1000.0
# Formats librosa reads directly through soundfile. Anything else — notably
# the .mov/.mp4 that source footage actually arrives in — must be decoded by
# ffmpeg first, or analysis fails outright.
NATIVE_AUDIO_SUFFIXES = {".wav", ".aif", ".aiff", ".flac"}
# Voice analysis only needs the speech band: 16 kHz mono is well above the
# Nyquist limit for our 1 kHz pitch ceiling, and keeps the extracted file
# small and fast to decode even for hour-long footage.
EXTRACT_SAMPLE_RATE = 16000
EXTRACT_TIMEOUT_SECONDS = 600
@contextlib.contextmanager
def decodable_audio(path: str) -> Iterator[Optional[str]]:
"""Yield a path librosa can read, extracting the audio track if needed.
Audio files pass straight through. Video containers are decoded to a
temporary mono WAV with ffmpeg and cleaned up on exit. Yields ``None``
when the audio cannot be obtained (no ffmpeg, no audio track, failure),
keeping the graceful-degradation contract of this module.
"""
file_path = Path(path)
if file_path.suffix.lower() in NATIVE_AUDIO_SUFFIXES:
yield str(file_path)
return
if shutil.which("ffmpeg") is None:
logger.info("ffmpeg not found on PATH; cannot extract audio from %s", file_path)
yield None
return
tmp_dir = tempfile.mkdtemp(prefix="fcp_voice_")
wav_path = Path(tmp_dir) / "audio.wav"
try:
result = subprocess.run(
[
"ffmpeg", "-hide_banner", "-nostdin", "-y",
"-i", str(file_path),
"-vn", # audio only: decoding video would dominate the runtime
"-ac", "1",
"-ar", str(EXTRACT_SAMPLE_RATE),
str(wav_path),
],
capture_output=True,
text=True,
timeout=EXTRACT_TIMEOUT_SECONDS,
)
if result.returncode != 0 or not wav_path.is_file():
logger.warning("ffmpeg could not extract audio from %s", file_path)
yield None
else:
yield str(wav_path)
except (OSError, subprocess.TimeoutExpired):
logger.warning("audio extraction failed for %s", file_path)
yield None
finally:
shutil.rmtree(tmp_dir, ignore_errors=True)
def features_capability() -> Tuple[bool, str]:
"""Whether pitch/energy extraction is available (librosa installed)."""
try:
import librosa # noqa: F401
except Exception:
return False, "Análise acústica indisponível: componente librosa ausente."
return True, "Análise acústica disponível."
def extract_pitch(
path: str, hop_length: int = 512, max_analysis_seconds: float = 1200.0
) -> Optional[List[Tuple[float, float]]]:
"""Frame-level pitch (F0) track via librosa's ``pyin``.
Returns ``[(time_seconds, hz), ...]`` for voiced frames only (unvoiced
frames, where ``pyin`` reports no pitch, are dropped), or ``None`` when
librosa is unavailable or the file cannot be analyzed.
"""
file_path = Path(path)
if not file_path.is_file():
return None
try:
import librosa
except ImportError:
logger.info("librosa not installed; pitch extraction unavailable")
return None
try:
with decodable_audio(str(file_path)) as audio_path:
if audio_path is None:
return None
y, sr = librosa.load(audio_path, sr=None, mono=True, duration=max_analysis_seconds)
f0, voiced_flag, _voiced_prob = librosa.pyin(
y, fmin=PITCH_FMIN_HZ, fmax=PITCH_FMAX_HZ, sr=sr, hop_length=hop_length
)
times = librosa.times_like(f0, sr=sr, hop_length=hop_length)
except Exception:
logger.warning("librosa pitch analysis failed for %s", file_path)
return None
return [
(float(t), float(hz))
for t, hz, voiced in zip(times, f0, voiced_flag)
if voiced and hz == hz # ``hz == hz`` filters NaN without importing math/numpy here
]
def extract_energy(
path: str, hop_length: int = 512, max_analysis_seconds: float = 1200.0
) -> Optional[List[Tuple[float, float]]]:
"""Frame-level RMS energy track via librosa.
Returns ``[(time_seconds, rms), ...]``, or ``None`` when librosa is
unavailable or the file cannot be analyzed.
"""
file_path = Path(path)
if not file_path.is_file():
return None
try:
import librosa
except ImportError:
logger.info("librosa not installed; energy extraction unavailable")
return None
try:
with decodable_audio(str(file_path)) as audio_path:
if audio_path is None:
return None
y, sr = librosa.load(audio_path, sr=None, mono=True, duration=max_analysis_seconds)
rms = librosa.feature.rms(y=y, hop_length=hop_length)[0]
times = librosa.times_like(rms, sr=sr, hop_length=hop_length)
except Exception:
logger.warning("librosa energy analysis failed for %s", file_path)
return None
return [(float(t), float(r)) for t, r in zip(times, rms)]
def _window_average(track: Sequence[Tuple[float, float]], start: float, end: float) -> Optional[float]:
"""Average of ``track`` values whose timestamp falls in ``[start, end]``."""
values = [v for t, v in track if start <= t <= end]
if not values:
return None
return sum(values) / len(values)
def word_pitch_energy(
words: Sequence[dict],
pitch_track: Optional[Sequence[Tuple[float, float]]],
energy_track: Optional[Sequence[Tuple[float, float]]],
) -> List[dict]:
"""Attach average pitch/energy over each word's ``[start, end]`` span.
Words carry ``pitch_hz``/``energy`` (``None`` when the span has no
voiced frames or a track is unavailable). Both tracks are the output of
:func:`extract_pitch`/:func:`extract_energy`.
"""
out: List[dict] = []
for w in words:
ww = dict(w)
start = float(w.get("start", 0.0))
end = float(w.get("end", start))
ww["pitch_hz"] = _window_average(pitch_track, start, end) if pitch_track else None
ww["energy"] = _window_average(energy_track, start, end) if energy_track else None
out.append(ww)
return out
def compute_speech_rate(words: Sequence[dict], window_seconds: float = 3.0) -> List[float]:
"""Local speech rate (words/second) around each word.
For word *i*, counts every word whose start falls within
``[start_i - window_seconds, start_i]`` and divides by
``window_seconds`` — a trailing local rate, cheap to compute and stable
against a single long/short word skewing the whole utterance's average.
"""
starts = [float(w.get("start", 0.0)) for w in words]
rates: List[float] = []
for i, s in enumerate(starts):
lo = s - window_seconds
count = sum(1 for t in starts[: i + 1] if t >= lo)
rates.append(count / window_seconds if window_seconds > 0 else 0.0)
return rates
def compute_pauses(words: Sequence[dict]) -> List[float]:
"""Silence (seconds) immediately before each word.
The first word's "pause before" is the time from the start of the audio
to its own start; every other word measures the gap since the previous
word's end (clamped to ``0`` for overlapping/adjacent words).
"""
pauses: List[float] = []
prev_end = 0.0
for w in words:
start = float(w.get("start", 0.0))
pauses.append(max(0.0, start - prev_end))
prev_end = float(w.get("end", start))
return pauses
+505
View File
@@ -0,0 +1,505 @@
"""Voice timeline — the consolidated, AI-readable view of how a video is spoken.
This is the *source of truth* between analysis and editing: it merges what
was said (transcript), who said it (diarization), and how it was said
(pitch/energy/rate/pauses → emphasis) into one JSON document, decoupling the
audio analysis from FCPXML generation entirely.
The shape is designed to be handed to a language model so it can reason about
the narrative — which beats carry weight, where a speaker changes, where the
delivery peaks — and decide how to direct the edit. Two design choices serve
that goal:
* **Layered, not flat.** A ``summary`` gives the whole picture in a few
numbers, ``segments`` group words into utterances with their own
aggregates, and ``words`` hold the fine detail. A model can reason from
the top layer and only descend where it matters, instead of parsing
thousands of word rows to find the shape of the piece.
* **Normalized, self-describing values.** Every acoustic value is 0–1 and
relative to *this* recording (a quiet podcast and a shouted ad both use
the full range), and ``scales`` documents that contract inline, so the
numbers are interpretable without external context.
"""
import json
import logging
from pathlib import Path
from typing import Callable, List, Optional, Sequence, Tuple
from .diarize import DEFAULT_SPEAKER, assign_speakers, build_speakers, diarize
from .emphasis import EmphasisWeights, annotate_emphasis
from .voice_features import (
compute_pauses,
compute_speech_rate,
extract_energy,
extract_pitch,
word_pitch_energy,
)
logger = logging.getLogger(__name__)
VOICE_TIMELINE_VERSION = "1.0"
# Silence long enough to mean the take stopped rather than the speaker paused.
# On real footage, boundaries between retakes showed gaps of 3.6-19.8s while
# dramatic beats inside a delivered line stayed under ~2s.
TAKE_BOUNDARY_GAP = 3.0
# How to read the values in this document, split by the level they live on.
# Embedded in the output so a model consuming the JSON needs no external
# documentation — and kept honest: a metric listed under "word" must exist on
# every word row, and one under "segment" on every segment row.
VALUE_SCALES = {
"word": {
"energy": "0-1, loudness relative to the loudest moment of this recording",
"pitch_delta": "0-1, how far this word's pitch sits from the speaker's average",
"rate_delta": "0-1, how much the local speaking rate departs from the average",
"pause_before": "seconds of silence immediately before the word",
"emphasis": "0-1 combined index; high values are punch-in/highlight candidates",
},
"segment": {
"gap_before": "seconds of silence before this line",
"take_boundary": "true when the gap is long enough that the take likely restarted here",
"avg_energy": "0-1 mean loudness across the line",
"peak_emphasis": "0-1 highest emphasis of any word in the line",
},
}
def _normalize(value: Optional[float], maximum: float) -> float:
"""Scale ``value`` into 0-1 against ``maximum`` (0.0 when unavailable)."""
if value is None or maximum <= 0:
return 0.0
return max(0.0, min(1.0, value / maximum))
def _round_word(word: dict) -> dict:
"""One word row, rounded to a size a model can read without noise.
The raw ``energy_raw``/``pitch_hz`` ride along beside the normalized
values so the document can be re-analyzed over a subset later. That
matters after cutting: every normalized value is relative to the
loudest moment of the *whole* recording, and if that moment gets cut
the survivors are scored against something that no longer exists.
"""
return {
"text": word.get("word", ""),
"start": round(float(word.get("start", 0.0)), 3),
"end": round(float(word.get("end", 0.0)), 3),
"speaker": word.get("speaker_id", DEFAULT_SPEAKER),
"energy": round(word.get("energy_norm", 0.0), 3),
"pitch_delta": round(word.get("pitch_delta", 0.0), 3),
"rate_delta": round(word.get("rate_delta", 0.0), 3),
"pause_before": round(word.get("pause_before", 0.0), 3),
"emphasis": round(word.get("emphasis", 0.0), 3),
"energy_raw": word.get("energy"),
"pitch_hz": word.get("pitch_hz"),
}
def enrich_words(
words: Sequence[dict],
pitch_track: Optional[Sequence] = None,
energy_track: Optional[Sequence] = None,
weights: EmphasisWeights = EmphasisWeights(),
already_measured: bool = False,
) -> List[dict]:
"""Attach normalized acoustic features + the emphasis index to each word.
Normalization is per-recording: energy against the loudest word, pitch
against the spread around this recording's average, rate against the
largest local departure. That makes the numbers comparable within a
piece regardless of how it was recorded.
Set ``already_measured`` when the words already carry ``energy`` and
``pitch_hz`` from a previous pass — re-analyzing a subset, say. The
frame tracks are then unnecessary, and sampling them again would
overwrite good values with ``None``.
"""
if not words:
return []
if not already_measured:
words = word_pitch_energy(words, pitch_track, energy_track)
rates = compute_speech_rate(words)
pauses = compute_pauses(words)
energies = [w["energy"] for w in words if w.get("energy") is not None]
max_energy = max(energies) if energies else 0.0
pitches = [w["pitch_hz"] for w in words if w.get("pitch_hz") is not None]
avg_pitch = sum(pitches) / len(pitches) if pitches else 0.0
pitch_span = (max(pitches) - min(pitches)) if len(pitches) > 1 else 0.0
avg_rate = sum(rates) / len(rates) if rates else 0.0
max_rate = max(rates) if rates else 0.0
enriched: List[dict] = []
for i, w in enumerate(words):
ww = dict(w)
ww["energy_norm"] = _normalize(w.get("energy"), max_energy)
pitch = w.get("pitch_hz")
ww["pitch_delta"] = (
_normalize(abs(pitch - avg_pitch), pitch_span) if pitch is not None else 0.0
)
ww["rate_delta"] = _normalize(abs(rates[i] - avg_rate), max_rate)
ww["pause_before"] = pauses[i]
enriched.append(ww)
annotated = annotate_emphasis(
[{**w, "energy": w["energy_norm"]} for w in enriched], weights=weights
)
for word, scored in zip(enriched, annotated):
word["emphasis"] = scored["emphasis"]
return enriched
def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]:
"""Group enriched words under their segment, with per-segment aggregates.
The aggregates are what let a model judge a whole utterance ("this line
is delivered hot, that one trails off") without reading every word.
"""
rows: List[dict] = []
previous_end = 0.0
for seg in segments:
start = float(seg.get("start", 0.0))
end = float(seg.get("end", 0.0))
in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end]
energies = [w["energy_norm"] for w in in_seg]
emphases = [w["emphasis"] for w in in_seg]
gap = max(0.0, start - previous_end)
rows.append(
{
"start": round(start, 3),
"end": round(end, 3),
"speaker": seg.get("speaker_id", DEFAULT_SPEAKER),
"text": (seg.get("text") or "").strip(),
# Silence before this line. Long gaps are where the camera
# stopped or the take restarted, so this is the structural
# hint for splitting a recording into takes — the same signal
# that is *noise* for emphasis (see emphasis.pause_weight).
"gap_before": round(gap, 3),
"take_boundary": gap >= TAKE_BOUNDARY_GAP,
"avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0,
"peak_emphasis": round(max(emphases), 3) if emphases else 0.0,
"words": [_round_word(w) for w in in_seg],
}
)
previous_end = end
return rows
# Words too common to ever be the point of a punch-in. A zoom lands on what a
# sentence is *about*, and an article spoken loudly is still an article.
_FUNCTION_WORDS = {
"a", "o", "e", "de", "da", "do", "que", "é", "em", "um", "uma", "as", "os",
"no", "na", "com", "pra", "para", "por", "se", "mais", "isso", "aí", "tudo",
"ao", "à", "dos", "das", "nos", "nas", "ou", "mas", "já", "ele", "ela",
"eu", "você", "seu", "sua", "meu", "minha", "esse", "essa", "aquele",
}
def _survives(start: float, end: float, cuts: Sequence[Tuple[float, float]]) -> bool:
"""Whether a span lies entirely outside every removed range."""
return all(end <= cut_start or start >= cut_end for cut_start, cut_end in cuts)
def restrict_to_kept(
timeline: dict,
cut_ranges: Sequence[Tuple[float, float]],
weights: EmphasisWeights = EmphasisWeights(),
peak_percentile: float = 0.02,
emphasis_floor: float = 0.25,
) -> dict:
"""Re-analyze a timeline over only the material that survives ``cut_ranges``.
Emphasis is *relative*: energy is scored against the loudest word,
pitch against the spread of the recording. Cut the loudest moment out —
a laugh, an aside to the crew — and every remaining score is measured
against something the viewer will never see. Re-running the
normalization over just the survivors is what makes "the most emphatic
line of the final video" a meaningful question.
Returns a timeline of the same shape, with times still in original
source seconds so the result can be fed straight back as actions.
"""
kept_words = [
w
for segment in timeline.get("segments", [])
for w in segment.get("words", [])
if _survives(w["start"], w["end"], cut_ranges)
]
# enrich_words expects the raw analysis keys, not the normalized ones.
raw = [
{
"word": w["text"],
"start": w["start"],
"end": w["end"],
"speaker_id": w.get("speaker", DEFAULT_SPEAKER),
"energy": w.get("energy_raw"),
"pitch_hz": w.get("pitch_hz"),
}
for w in kept_words
]
enriched = enrich_words(raw, weights=weights, already_measured=True)
kept_segments = [
{**s, "words": [w for w in s.get("words", []) if _survives(w["start"], w["end"], cut_ranges)]}
for s in timeline.get("segments", [])
]
kept_segments = [s for s in kept_segments if s["words"]]
rows = _segment_rows(
[{"text": s["text"], "start": s["start"], "end": s["end"],
"speaker_id": s.get("speaker", DEFAULT_SPEAKER)} for s in kept_segments],
enriched,
)
duration = sum(s["end"] - s["start"] for s in rows)
return {
**timeline,
"summary": _summary(enriched, rows, timeline.get("speakers", []),
duration, peak_percentile, emphasis_floor),
"segments": rows,
}
def sentence_end(segments: Sequence[dict], index: int) -> float:
"""Where the sentence starting at ``segments[index]`` actually finishes.
Transcription segments break on breath and timing, not on grammar — a
sentence routinely spans two or three of them ("…que dá aquele ar" /
"de elegância, isso é desejo de muitas mulheres, né?"). A zoom that
ends on a segment boundary would therefore release mid-thought, so the
window is extended until a segment closes with terminal punctuation.
"""
last = float(segments[index]["end"])
for offset, segment in enumerate(segments[index:]):
# A long gap means the take stopped; never run a zoom across that.
# Checked before adopting the end, or the boundary segment's own
# end would already have been taken.
if offset > 0 and segment.get("take_boundary"):
break
last = float(segment["end"])
if (segment.get("text") or "").strip().endswith((".", "!", "?", "…")):
break
return last
def suggest_zoom_windows(
timeline: dict,
min_gap: float = 8.0,
max_zooms: Optional[int] = None,
) -> List[dict]:
"""Propose punch-in windows over a timeline's strongest lines.
One zoom per line at most, taken from the line's most emphatic
*content* word — a loudly spoken "a" is still an article, so function
words are skipped. The window runs from that word to the end of its
line, which is the shape the edit wants: the move lands with the word
and holds through the rest of the phrase.
``min_gap`` keeps successive zooms apart; effects stacked close
together read as nervous editing rather than emphasis.
"""
segments = timeline.get("segments", [])
candidates: List[dict] = []
for i, segment in enumerate(segments):
content = [
w for w in segment.get("words", [])
if w["text"].strip(",.!?;:").lower() not in _FUNCTION_WORDS
]
if not content:
continue
best = max(content, key=lambda w: w["emphasis"])
candidates.append({
"start": best["start"],
# Hold through to the end of the sentence, not of the segment —
# releasing mid-thought is what makes a punch-in feel arbitrary.
"end": sentence_end(segments, i),
"word": best["text"],
"emphasis": best["emphasis"],
"line": segment["text"],
})
chosen: List[dict] = []
for candidate in sorted(candidates, key=lambda c: c["emphasis"], reverse=True):
if max_zooms is not None and len(chosen) >= max_zooms:
break
if any(abs(candidate["start"] - c["start"]) < min_gap for c in chosen):
continue
chosen.append(candidate)
return sorted(chosen, key=lambda c: c["start"])
def speaker_profiles(segments: Sequence[dict], duration: float) -> List[dict]:
"""Per-speaker statistics and sample lines, so a person can tell who is who.
A bare ``SPEAKER_00`` label is useless for deciding whose audio to cut.
What identifies a role is *how* someone participates: an interviewer or
a crew member asks short questions and holds little of the runtime,
while the subject speaks in long stretches. ``avg_segment`` and
``share`` capture exactly that contrast, and the sample lines confirm
it in the person's own words.
"""
by_speaker: dict = {}
for seg in segments:
sid = seg.get("speaker", seg.get("speaker_id", DEFAULT_SPEAKER))
length = max(0.0, float(seg.get("end", 0.0)) - float(seg.get("start", 0.0)))
entry = by_speaker.setdefault(sid, {"seconds": 0.0, "segments": [], "words": 0})
entry["seconds"] += length
entry["words"] += len(seg.get("words", []))
entry["segments"].append(seg)
profiles: List[dict] = []
for i, (sid, entry) in enumerate(
sorted(by_speaker.items(), key=lambda kv: kv[1]["seconds"], reverse=True)
):
count = len(entry["segments"])
# Longest lines identify a role far better than the first ones: a
# question and an answer look alike at the start of a recording.
longest = sorted(
entry["segments"],
key=lambda s: float(s.get("end", 0)) - float(s.get("start", 0)),
reverse=True,
)[:3]
profiles.append({
"id": sid,
"name": f"Speaker {i + 1}",
"speaking_seconds": round(entry["seconds"], 2),
"share": round(entry["seconds"] / duration, 3) if duration > 0 else 0.0,
"segment_count": count,
"avg_segment": round(entry["seconds"] / count, 2) if count else 0.0,
"word_count": entry["words"],
"samples": [(s.get("text") or "").strip()[:160] for s in longest],
})
return profiles
def select_peaks(
words: Sequence[dict], percentile: float, floor: float
) -> List[dict]:
"""The most emphatic words: the top ``percentile`` fraction, above ``floor``.
Selection is relative on purpose. The emphasis index is a weighted
average whose real range depends entirely on the material — a measured
interview peaks around 0.5 while an energetic ad reaches much higher —
so any fixed cutoff either floods one and selects nothing in the other.
Asking for "the top 2%" instead yields a usable handful either way.
``floor`` is only a sanity guard for genuinely flat audio, where even
the top of the distribution carries no emphasis worth cutting on.
"""
ranked = sorted(words, key=lambda w: w["emphasis"], reverse=True)
keep = max(1, round(len(ranked) * percentile)) if ranked else 0
return [w for w in ranked[:keep] if w["emphasis"] >= floor]
def _summary(words: Sequence[dict], segments: Sequence[dict], speakers: Sequence[dict],
duration: float, peak_percentile: float, emphasis_floor: float) -> dict:
"""The top layer: the shape of the piece in a handful of numbers."""
emphases = [w["emphasis"] for w in words]
peaks = select_peaks(words, peak_percentile, emphasis_floor)
return {
"duration": round(duration, 3),
"speaker_count": len(speakers),
"segment_count": len(segments),
"word_count": len(words),
"avg_emphasis": round(sum(emphases) / len(emphases), 3) if emphases else 0.0,
"peak_selection": f"top {peak_percentile:.0%} of words, minimum emphasis {emphasis_floor:.2f}",
"peak_count": len(peaks),
"peak_moments": [
{
"time": round(float(w.get("start", 0.0)), 3),
"text": w.get("word", ""),
"speaker": w.get("speaker_id", DEFAULT_SPEAKER),
"emphasis": round(w["emphasis"], 3),
}
for w in sorted(peaks, key=lambda w: w["emphasis"], reverse=True)[:20]
],
}
def build_voice_timeline(
media_path: str,
transcript: dict,
hf_token: Optional[str] = None,
num_speakers: str = "",
weights: EmphasisWeights = EmphasisWeights(),
peak_percentile: float = 0.02,
emphasis_floor: float = 0.25,
progress_cb: Optional[Callable[[float, str], None]] = None,
) -> dict:
"""Build the consolidated voice timeline for one media file.
Every analysis layer is optional and degrades independently: without
librosa the acoustic values are ``0.0``; without a diarization token
every word belongs to ``SPEAKER_00``. The document's shape never
changes, so downstream consumers (the rules engine, or a model reading
the JSON) can rely on it.
"""
def report(fraction: float, stage: str) -> None:
if progress_cb:
progress_cb(fraction, stage)
report(0.1, "Analisando tom e energia...")
pitch_track = extract_pitch(media_path)
energy_track = extract_energy(media_path)
report(0.5, "Calculando ênfase...")
words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights)
report(0.7, "Identificando participantes...")
tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None
segments, words = assign_speakers(transcript.get("segments", []), words, tracks)
speakers = build_speakers(segments)
report(0.9, "Montando linha do tempo...")
duration = float(transcript.get("duration", 0.0))
segment_rows = _segment_rows(segments, words)
return {
"version": VOICE_TIMELINE_VERSION,
"source": Path(media_path).name,
"language": transcript.get("language", ""),
# What actually ran, not what was installed — a consumer must be able
# to tell "this speech is flat" from "the acoustics never loaded",
# since both leave the same zeros in the data.
"layers": {
"transcript": bool(transcript.get("words")),
"acoustics": pitch_track is not None or energy_track is not None,
"speakers": tracks is not None,
},
"scales": VALUE_SCALES,
"summary": _summary(
words, segments, speakers, duration, peak_percentile, emphasis_floor
),
"speakers": speaker_profiles(segment_rows, duration),
"segments": segment_rows,
}
def voice_timeline_path(media_path: str, output_dir: Optional[str] = None) -> Path:
"""Where the ``_voice_timeline.json`` for ``media_path`` lives.
Mirrors ``_transcript.json``: next to the media, or in the chosen
project folder when one is set.
"""
p = Path(media_path)
if output_dir:
directory = Path(output_dir).expanduser()
directory.mkdir(parents=True, exist_ok=True)
return directory / f"{p.stem}_voice_timeline.json"
return p.with_name(p.stem + "_voice_timeline.json")
def save_voice_timeline(timeline: dict, path: Path) -> None:
"""Write the timeline as UTF-8 JSON (accented transcripts stay readable)."""
with open(path, "w", encoding="utf-8") as f:
json.dump(timeline, f, ensure_ascii=False, indent=2)
def load_voice_timeline(path: Path) -> Optional[dict]:
"""Read a cached voice timeline, or ``None`` when absent/unreadable."""
try:
with open(path, encoding="utf-8") as f:
data = json.load(f)
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
return None
return data if isinstance(data, dict) and "segments" in data else None
+355 -39
View File
@@ -35,9 +35,11 @@ import unicodedata
import uuid
import xml.etree.ElementTree as ET
from datetime import datetime
from fractions import Fraction
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
from .collision import blocking, validate_titles
from .models import (
_FCPXML_STANDARD_TIMEBASES,
DynamicSubtitleConfig,
@@ -50,7 +52,12 @@ from .models import (
ValidationIssue,
ValidationIssueType,
)
from .text_layout import LayoutBox, compose_sentence, layout_sentence
from .text_layout import (
TEXT_TEMPLATE_FONT_SCALE,
LayoutBox,
compose_sentence,
layout_sentence,
)
from .transcribe import group_words_by_segment
# Maximum lengths for XML attribute values to prevent memory abuse
@@ -154,6 +161,23 @@ _ASSET_CLIP_CHILD_ORDER = [
_CHILD_ORDER_INDEX = {tag: i for i, tag in enumerate(_ASSET_CLIP_CHILD_ORDER)}
# How close to the end of a clip a zoom must finish for the return to be
# skipped. Within this margin the cut arrives before the eye registers the
# move back, so the return reads as a twitch rather than a resolution.
HOLD_AT_CUT_THRESHOLD = 1.0
# How close to the start of a clip a zoom must begin for the ramp-in to be
# skipped and the shot to simply open already zoomed. Tighter than the end
# margin on purpose: at the end the cut hides an unfinished return, but at
# the start a ramp is visible from frame one and reads as the shot settling.
START_AT_CUT_THRESHOLD = 0.5
def _fmt_scale(value: float) -> str:
"""Format a scale factor without trailing float noise (1.0 -> "1")."""
return f"{value:.6f}".rstrip("0").rstrip(".") or "0"
def _dtd_insert(parent: ET.Element, child: ET.Element) -> ET.Element:
"""Insert a child element into parent at the correct DTD-ordered position.
@@ -541,10 +565,43 @@ def _check_timebases(root: ET.Element) -> List[ValidationIssue]:
return issues
def _document_frame_duration(root: ET.Element) -> Optional[Fraction]:
"""The sequence's exact ``frameDuration`` as a fraction, if declared.
Read from the format the ``<sequence>`` references (falling back to the
first declared format), so the value is the document's own timebase
rather than an assumed rate.
"""
formats = {f.get('id'): f for f in root.findall('.//format') if f.get('id')}
sequence = root.find('.//sequence')
fmt = formats.get(sequence.get('format')) if sequence is not None else None
if fmt is None:
fmt = next(iter(formats.values()), None)
if fmt is None:
return None
raw = fmt.get('frameDuration', '')
if not (raw.endswith('s') and '/' in raw):
return None
numerator, denominator = raw[:-1].split('/', 1)
try:
value = Fraction(int(numerator), int(denominator))
except (ValueError, ZeroDivisionError):
return None
return value if value > 0 else None
def _check_frame_alignment(root: ET.Element, fps: float = 24.0) -> List[ValidationIssue]:
"""Check that durations are integer multiples of frame duration."""
"""Check that durations are integer multiples of the frame duration.
Uses the document's exact ``frameDuration`` fraction and rational
arithmetic. Comparing against an integer fps instead would flag every
NTSC project as broken: at 1001/24000s (23.976fps) a perfectly aligned
duration is not an integer number of "24fps" frames, so whole timelines
would be reported misaligned when nothing is wrong.
"""
issues = []
fps_int = int(fps)
frame_duration = _document_frame_duration(root)
label = f"{1 / float(frame_duration):.3f}".rstrip('0').rstrip('.') if frame_duration else str(fps)
for elem in root.iter():
dur_str = elem.get('duration')
if not dur_str or not dur_str.endswith('s'):
@@ -553,14 +610,19 @@ def _check_frame_alignment(root: ET.Element, fps: float = 24.0) -> List[Validati
continue
try:
tv = TimeValue.from_timecode(dur_str)
frames = tv.to_seconds() * fps_int
if abs(frames - round(frames)) > 0.01:
if frame_duration is not None:
frames = Fraction(tv.numerator, tv.denominator) / frame_duration
aligned = frames.denominator == 1
else:
approx = tv.to_seconds() * fps
aligned = abs(approx - round(approx)) <= 0.01
if not aligned:
issues.append(ValidationIssue(
issue_type=ValidationIssueType.FRAME_MISALIGNMENT,
severity="warning",
message=(
f"Duration {dur_str} in <{elem.tag}> "
f"'{elem.get('name', '')}' is not frame-aligned at {fps_int}fps."
f"'{elem.get('name', '')}' is not frame-aligned at {label}fps."
),
clip_name=elem.get('name'),
))
@@ -1030,13 +1092,21 @@ class FCPXMLModifier:
elem.set(attr, val)
return elem
def _require_clip(self, clip_id: str) -> ET.Element:
def _require_clip(self, clip_id: 'str | ET.Element') -> ET.Element:
"""Look up a clip by ID/name, raising if not found.
Centralises the get-or-raise pattern used by every clip-mutating
method so the error message stays consistent and future
enhancements (fuzzy matching, suggestions) only need one site.
An Element is returned as-is. That matters after ``split_clip`` or
``cut_clip_ranges``: the resulting pieces all carry the *same* name,
so a name lookup would always resolve to the first one and silently
put the edit on the wrong piece. Callers holding the exact element
pass it directly.
"""
if isinstance(clip_id, ET.Element):
return clip_id
clip = self.clips.get(clip_id)
if clip is None:
raise ValueError(f"Clip not found: {clip_id}")
@@ -1421,7 +1491,7 @@ class FCPXMLModifier:
def add_marker(
self,
clip_id: str,
clip_id: 'str | ET.Element',
timecode: str,
name: str,
marker_type: "MarkerType | str" = MarkerType.STANDARD,
@@ -1937,35 +2007,50 @@ class FCPXMLModifier:
def add_zoom(
self,
clip_id: str,
clip_id: 'str | ET.Element',
start: float,
end: float,
scale: float = 1.3,
ease: float = 0.3,
ease: float = 0.25,
position: str = "0 0",
ease_out: Optional[float] = None,
hold_at_end: Optional[bool] = None,
start_at_peak: Optional[bool] = None,
) -> ET.Element:
"""Add a smooth ease-in/ease-out punch-in zoom to a clip.
"""Add a punch-in zoom to a clip, snapping back to its framing at the end.
Animates ``<adjust-transform>``'s ``scale`` param (per the FCPXML
DTD: ``<param>`` + ``<keyframeAnimation>`` of ``<keyframe>``
elements, ``interp="ease"``) from 100% up to *scale* and back down
to 100%, entirely within ``[start, end]`` — clip-relative seconds
(seconds from the clip's own head, same convention as
``cut_clip_ranges``). The ease portions each last *ease* seconds;
the zoom holds at *scale* in between.
Animates ``<adjust-transform>``'s ``scale`` param (``<param>`` +
``<keyframeAnimation>`` of ``<keyframe>``) from the clip's current
scale up to *scale* times it, holds, then returns — all within
``[start, end]`` — clip-relative seconds (same convention as
``cut_clip_ranges``).
The two ends are deliberately asymmetric. *ease* ramps the zoom
**in** over half a second by default, fast enough to land with the
emphasised word. The way **out** is instant — a single frame — so
the moment the impact phrase ends the shot is simply back to its
normal framing and the video resumes its flow, with no drift
drawing attention to itself. Pass *ease_out* to ramp the return
gradually instead.
*hold_at_end* keeps the peak instead of returning, and
*start_at_peak* opens already zoomed with no ramp. Left as ``None``
both decide on their own from how close the window sits to the
clip's edges: a cut is itself the transition, so ramping away from
one — or back toward one — is motion the viewer reads as a wobble
rather than as emphasis.
"""
if end <= start:
raise ValueError(f"end ({end}) must be greater than start ({start})")
if ease <= 0:
raise ValueError(f"ease must be positive, got {ease}")
if ease * 2 > (end - start):
raise ValueError(
f"ease ({ease}s x2 = {ease * 2}s) doesn't fit in the zoom "
f"window ({end - start}s) — shorten ease or widen start/end"
)
if scale <= 0:
raise ValueError(f"scale must be positive, got {scale}")
frame = float(self.frame_duration_fraction())
ramp_out = frame if ease_out is None else ease_out
if ramp_out <= 0:
raise ValueError(f"ease_out must be positive, got {ease_out}")
clip = self._require_clip(clip_id)
clip_duration = self._parse_time(clip.get('duration', '0s')).to_seconds()
if start < 0 or end > clip_duration:
@@ -1974,26 +2059,146 @@ class FCPXMLModifier:
f"duration (0 to {clip_duration:.3f}s)"
)
# Replace rather than stack a prior zoom on the same clip.
# Replace a prior zoom, but never the clip's framing. A clip can
# already carry an <adjust-transform> holding the editor's own
# reframe — rotation for footage shot sideways, position, a scale
# that makes the shot work at all. Dropping it outright (the old
# behaviour) silently destroyed that framing; on real footage the
# zoomed section came back rotated. So: keep the static attributes,
# and animate *relative to* the existing scale.
base_x, base_y = 1.0, 1.0
carried: dict = {}
old_keyframes: list = []
for stale in clip.findall('adjust-transform'):
carried = {k: v for k, v in stale.attrib.items() if k != 'scale'}
parts = (stale.get('scale') or '').split()
if len(parts) == 2:
try:
base_x, base_y = float(parts[0]), float(parts[1])
except ValueError:
base_x, base_y = 1.0, 1.0
else:
# No static attribute — a PRIOR zoom on this same clip left
# an animated <param name="scale"> instead, and the true
# resting framing lives in its keyframes, not in 1.0.
# Reading it as 1.0 here doesn't just miss the framing: it
# replaces the earlier zoom's whole animation with a wrong
# one, since this loop unconditionally removes `stale`
# right after. The rest value is recoverable without
# knowing which keyframe it is: MIN_ZOOM_SCALE == 1.0 means
# every keyframed value is >= the rest scale, so the
# smallest one keyframed is the rest value, peak or not.
for old_param in stale.findall("param[@name='scale']"):
xs, ys = [], []
for kf in old_param.findall('.//keyframe'):
kv = (kf.get('value') or '').split()
if len(kv) == 2:
try:
xs.append(float(kv[0]))
ys.append(float(kv[1]))
except ValueError:
pass
# Kept for merging: a second zoom on the same clip
# (two emphatic beats a cut didn't separate) should
# stack alongside the first, not erase it — the
# earlier peak is still a real editorial decision.
old_keyframes.append((kf.get('time', '0s'), kf.get('value', '')))
if xs and ys:
base_x, base_y = min(xs), min(ys)
clip.remove(stale)
transform = ET.Element('adjust-transform')
for key, value in carried.items():
transform.set(key, value)
scale_param = ET.SubElement(transform, 'param')
scale_param.set('name', 'scale')
anim = ET.SubElement(scale_param, 'keyframeAnimation')
scale_value = f"{scale} {scale}"
for seconds, value in (
(start, "1 1"),
(start + ease, scale_value),
(end - ease, scale_value),
(end, "1 1"),
):
# Keyframe times live in the clip's SOURCE timebase — the same origin
# as its own ``start`` — not in clip-relative seconds. A clip whose
# media starts at, say, 3109.9s of timecode looks for the animation
# there; keyframes written at 0-5s land outside the clip entirely and
# Final Cut imports the zoom as nothing at all, silently. Matches what
# add_text_title already does, and only shows up on footage whose
# start isn't 0s — every synthetic fixture starts at 0s and hides it.
media_origin = self._parse_time(clip.get('start', '0s'))
rest_value = f"{_fmt_scale(base_x)} {_fmt_scale(base_y)}"
scale_value = f"{_fmt_scale(base_x * scale)} {_fmt_scale(base_y * scale)}"
# A return that lands right before a cut is wasted motion: the next
# clip begins on its own framing anyway, so all the viewer sees is a
# twitch on the way out. When the zoom runs to the end of the clip,
# hold the peak and let the cut do the resetting.
holds_to_cut = (
hold_at_end
if hold_at_end is not None
else (clip_duration - end) <= HOLD_AT_CUT_THRESHOLD
)
opens_at_peak = (
start_at_peak
if start_at_peak is not None
else start <= START_AT_CUT_THRESHOLD
)
# Only the ramps actually written have to fit in the window: a zoom
# that opens at the peak spends no time ramping in, and one held to
# the cut spends none ramping out.
needed = (0.0 if opens_at_peak else ease) + (0.0 if holds_to_cut else ramp_out)
if needed > (end - start):
raise ValueError(
f"the ramps ({needed}s) don't fit in the zoom window "
f"({end - start}s) — shorten them or widen start/end"
)
if opens_at_peak:
# The cut already delivered the change of framing; ramping up
# from it just looks like the shot settling.
keyframes = [(start, scale_value)]
else:
keyframes = [(start, rest_value), (start + ease, scale_value)]
if holds_to_cut:
keyframes.append((end, scale_value))
else:
# Hold the peak right up to the end, then drop back on the very
# next frame — the snap-back the edit wants, not a slow drift.
keyframes.append((end - ramp_out, scale_value))
keyframes.append((end, rest_value))
new_entries = [
((media_origin + self.snap_seconds_to_frame(seconds)), value)
for seconds, value in keyframes
]
new_start_time = new_entries[0][0]
new_end_time = new_entries[-1][0]
# Two calls on the same clip mean two different things depending on
# whether their windows overlap. Overlapping = redoing the *same*
# zoom with new numbers — the old keyframes are stale and all of
# them go. Disjoint = a second, separate beat that a cut didn't
# separate onto its own clip — that one stacks alongside the first
# instead of erasing it, since both are real editorial decisions.
old_times = [self._parse_time(t) for t, _ in old_keyframes]
old_span_overlaps_new = bool(old_times) and not (
max(old_times) < new_start_time or min(old_times) > new_end_time
)
if old_span_overlaps_new:
surviving_old: list = []
else:
surviving_old = [(self._parse_time(t), v) for t, v in old_keyframes]
all_entries = sorted(surviving_old + new_entries, key=lambda e: e[0])
for time_value, value in all_entries:
kf = ET.SubElement(anim, 'keyframe')
kf.set('time', self.snap_seconds_to_frame(seconds).to_fcpxml())
kf.set('time', time_value.to_fcpxml())
kf.set('value', value)
kf.set('interp', 'ease')
# Only 'time' and 'value' — no 'interp', no 'curve'. The DTD allows
# both, but Final Cut rejected 'interp' on this vector param
# ("does not support the interpolation attribute") and discarded
# the whole <param>. A hand-made zoom exported from FCP itself
# writes bare keyframes and relies on the DTD default
# (curve="smooth"), so we match that export exactly rather than
# guess which attributes survive its importer.
if position != "0 0":
pos_param = ET.SubElement(transform, 'param')
@@ -2707,7 +2912,18 @@ class FCPXMLModifier:
_TEXT_POSITION_KEY = '9999/10003/13260/3296672360/1/100/101'
# Layout params the "Text" template ships with. These keys are the
# template's own defaults and never vary between instances.
#
# "Build Out" is the one deliberate override: with "Apply Speed" set to
# "2 (Per Object)" below, the template's whole built-in animation (build
# in + build out) is always compressed to exactly fill the title's own
# on-screen duration — so on a short word-length clip, build out was
# eating time that build in needed to finish revealing the text before
# the cut. Disabling build out hands that entire compressed window to
# build in alone, which is what "sempre acelerado" turned out to mean:
# no separate speed knob needed. Value captured from a real FCP export
# with "Build Out" unchecked in the Inspector (see chat, 2026-08-18).
_TEXT_TITLE_PARAMS = (
('Build Out', '9999/10000/2/102', '0'),
('Layout Method', '9999/10003/13260/3296672360/2/314', '1 (Paragraph)'),
('Left Margin', '9999/10003/13260/3296672360/2/323', '-1210'),
('Right Margin', '9999/10003/13260/3296672360/2/324', '1210'),
@@ -2795,6 +3011,7 @@ class FCPXMLModifier:
bold: bool = True,
face: Optional[str] = None,
kerning: Optional[float] = None,
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
) -> ET.Element:
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
@@ -2849,13 +3066,31 @@ class FCPXMLModifier:
style_def.set('id', ts_id)
text_style = ET.SubElement(style_def, 'text-style')
text_style.set('font', font)
text_style.set('fontSize', str(font_size))
# Text.moti sizes type in frame pixels but positions in canvas points.
# See TEXT_TEMPLATE_FONT_SCALE: layout measures in points, so only the
# emitted size (and its kerning, to keep the same letter spacing) is
# converted here.
scale = float(font_scale) or 1.0
text_style.set('fontSize', f"{float(font_size) * scale:g}")
text_style.set('fontColor', font_color)
text_style.set('bold', '1' if bold else '0')
if face:
# FCP represents bold weight as the bold attribute — never as a
# fontFace. Writing ``bold="0" fontFace="Bold"`` (the previous
# behaviour) is contradictory and FCP refuses to render the text.
# Italic, by contrast, IS a face: FCP writes both ``fontFace`` and
# ``italic="1"``. See Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-19.
face_lower = (face or '').strip().lower()
if face_lower == 'bold':
text_style.set('bold', '1')
elif 'italic' in face_lower:
text_style.set('fontFace', face)
text_style.set('italic', '1')
else:
if bold:
text_style.set('bold', '1')
if face:
text_style.set('fontFace', face)
if kerning:
text_style.set('kerning', f"{float(kerning):g}")
text_style.set('kerning', f"{float(kerning) * scale:g}")
text_style.set('alignment', 'center')
text_style.set('lineSpacing', '-19')
@@ -3005,7 +3240,9 @@ class FCPXMLModifier:
def lay_out(pending: List[Dict]):
"""Place what fits; return (units, still-unplaced words)."""
if phrase_mode:
composition = compose_sentence(pending, config.style, box)
composition = compose_sentence(
pending, config.style, box, line_gap=config.line_gap,
)
return composition.blocks, composition.overflow
layout = layout_sentence(pending, config.style, box)
return layout.placed, layout.overflow
@@ -3083,19 +3320,98 @@ class FCPXMLModifier:
duration,
lane=lane,
name=f"caption_{uuid.uuid4().hex[:8]}",
position=unit.position_param(),
position=unit.position_param(config.text_scale),
font=unit.font or config.style.font,
font_size=int(round(unit.font_size)),
font_color=unit.color or config.style.active_color,
bold=config.style.bold,
face=unit.face,
kerning=unit.kerning,
font_scale=config.text_scale,
)
_dtd_insert(parent, title)
created.append(title)
if getattr(config, 'validate', False):
report = self.validate_subtitle_layout()
if blocking(report["severity"]):
raise ValueError(
"Subtitle layout validation failed: "
+ str(report["summary"])
)
return created
def validate_subtitle_layout(
self,
*,
safe_margin_x: float = 0.05,
safe_margin_y: float = 0.05,
min_font_size: Optional[float] = None,
min_distance: Optional[float] = None,
max_distance: Optional[float] = None,
) -> dict:
"""Re-measure every ``<title>`` in the document and report collisions.
Reconstructs each title's on-screen box from the values the writer
emitted (``fontSize``/``kerning``/``Position`` are already in template
space), then checks for temporal+spatial collisions, frame/safe-area
containment, and font fallbacks. This is the spec-16 validation pass the
layout engine does not do on its own — it only guarantees non-overlap
*by construction* while composing, and cannot see a hand-edited title.
Returns the ``collision.validate_titles`` report: ``severity`` (worst
bucket), ``issues`` (spec-16 occurrences) and ``summary`` (counts).
"""
titles = []
for elem in self.root.iter('title'):
text_el = elem.find('text/text-style')
text = (text_el.text or '').strip() if text_el is not None else ''
style = elem.find('text-style-def/text-style')
font = style.get('font') if style is not None else None
face = style.get('fontFace') if style is not None else None
font_size = (
float(style.get('fontSize', '0')) if style is not None else 0.0
)
kerning = (
float(style.get('kerning', '0') or 0)
if style is not None else 0.0
)
x = y = 0.0
for param in elem.findall('param'):
if param.get('name') == 'Position' and param.get('value'):
parts = param.get('value').split()
if len(parts) >= 2:
x, y = float(parts[0]), float(parts[1])
start = self._parse_time(elem.get('offset', '0s')).to_seconds()
duration = self._parse_time(elem.get('duration', '0s')).to_seconds()
titles.append({
'text': text,
'font': font,
'face': face,
'font_size': font_size,
'kerning': kerning,
'x': x,
'y': y,
'start': start,
'end': start + duration,
'group': start + duration,
})
return validate_titles(
titles,
self.frame_width(),
self.frame_height(),
safe_margin_x=safe_margin_x,
safe_margin_y=safe_margin_y,
min_font_size=min_font_size,
min_distance=min_distance,
max_distance=max_distance,
)
# ========================================================================
# AUDIO CLIP OPERATIONS (v0.6.0)
# ========================================================================