chore: atualização geral
This commit is contained in:
@@ -0,0 +1,472 @@
|
||||
"""Collision detection and layout validation for dynamic-subtitle titles.
|
||||
|
||||
Pure functions — no I/O, no FCPXML parsing — that answer one question over and
|
||||
over: given the boxes a set of titles occupy on screen, do any two titles that
|
||||
are on screen at the same time intersect? And are they inside the frame, inside
|
||||
the safe area, and using a font the layout actually measured?
|
||||
|
||||
This is the post-generation guarantee the layout engine only provides *by
|
||||
construction* (``text_layout.compose_sentence`` stacks lines so their ink boxes
|
||||
never touch). Re-running it over already-emitted titles catches the cases the
|
||||
layout cannot see: a hand-edited position, a template whose type scales
|
||||
differently than ``text_scale`` assumed, a font that fell back to an estimate,
|
||||
or a word pushed off frame by a long emphasis line.
|
||||
|
||||
Boxes are measured in the *emitted* template space (frame pixels) — the same
|
||||
numbers the writer wrote to the FCPXML (``fontSize``, ``kerning`` and
|
||||
``Position`` are all already scaled by ``text_scale``), so validation re-measures
|
||||
with ``measure_text``/``ink_extent`` against those same numbers and never
|
||||
re-applies the scale factor. See ``writer.validate_subtitle_layout``.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from math import hypot
|
||||
from typing import Dict, List, Optional, Sequence
|
||||
|
||||
from .text_layout import (
|
||||
ink_extent,
|
||||
measure_text,
|
||||
metrics_for,
|
||||
vertical_metrics_for,
|
||||
)
|
||||
|
||||
# Severity buckets for a spatial overlap, ordered from harmless to blocking.
|
||||
# ``render_tolerance`` is the 5px the renderer can round off; ``severe`` is a
|
||||
# real collision that must be fixed before export.
|
||||
OVERLAP_NONE = "none"
|
||||
OVERLAP_RENDER_TOLERANCE = "render_tolerance"
|
||||
OVERLAP_WARNING = "warning"
|
||||
OVERLAP_PROBABLE = "probable"
|
||||
OVERLAP_SEVERE = "severe"
|
||||
|
||||
# Max fraction of the smaller box a severe collision may cover (spec 7.2).
|
||||
SEVERE_OVERLAP_RATIO = 0.15
|
||||
|
||||
# issue types (spec 16)
|
||||
SPATIAL_COLLISION = "spatial_collision"
|
||||
OUTSIDE_FRAME = "outside_frame"
|
||||
OUTSIDE_SAFE_AREA = "outside_safe_area"
|
||||
INSUFFICIENT_SPACING = "insufficient_spacing"
|
||||
EXCESSIVE_SPACING = "excessive_spacing"
|
||||
FONT_MISSING = "font_missing"
|
||||
FONT_TOO_SMALL = "font_too_small"
|
||||
INVALID_ANCHOR = "invalid_anchor"
|
||||
UNRESOLVED_TRANSFORM = "unresolved_transform"
|
||||
|
||||
|
||||
@dataclass
|
||||
class Box:
|
||||
"""An axis-aligned rectangle in frame coordinates, y growing upward."""
|
||||
|
||||
left: float
|
||||
right: float
|
||||
bottom: float
|
||||
top: float
|
||||
|
||||
@property
|
||||
def width(self) -> float:
|
||||
return self.right - self.left
|
||||
|
||||
@property
|
||||
def height(self) -> float:
|
||||
return self.top - self.bottom
|
||||
|
||||
@property
|
||||
def area(self) -> float:
|
||||
return self.width * self.height
|
||||
|
||||
def overlaps(self, other: "Box") -> bool:
|
||||
"""True if the two boxes intersect (strict — touching edges do not)."""
|
||||
return (
|
||||
self.left < other.right
|
||||
and other.left < self.right
|
||||
and self.bottom < other.top
|
||||
and other.bottom < self.top
|
||||
)
|
||||
|
||||
|
||||
# A boundary the writer places deliberately exact — one block's title
|
||||
# duration set to literally equal the next block's start (see writer.py's
|
||||
# ``block_ends``) — can still land a few float-ULPs apart by the time it
|
||||
# gets here: an ``end`` re-derived as ``start + duration`` from two already-
|
||||
# rounded floats isn't bit-identical to a ``start`` read as one division of
|
||||
# the same exact fraction, even though both trace back to one FCPXML value.
|
||||
# Found on real footage: 7 of 8 "severe" collisions in one clip were exactly
|
||||
# this — same instant, off by ~1e-13s, nowhere near a real frame boundary
|
||||
# (~0.04s). A tolerance many orders below one frame absorbs the artifact
|
||||
# without hiding a genuine overlap.
|
||||
_BOUNDARY_EPSILON = 1e-6
|
||||
|
||||
|
||||
def temporal_overlap(
|
||||
start_a: float, end_a: float, start_b: float, end_b: float
|
||||
) -> bool:
|
||||
"""Whether the half-open intervals ``[start, end)`` intersect (spec 7.1).
|
||||
|
||||
Strict on both sides, so a title that ends exactly when the next begins is
|
||||
never treated as simultaneous — see ``_BOUNDARY_EPSILON`` for why "exactly"
|
||||
needs a tolerance rather than bare float comparison.
|
||||
"""
|
||||
return (
|
||||
start_a < end_b - _BOUNDARY_EPSILON
|
||||
and start_b < end_a - _BOUNDARY_EPSILON
|
||||
)
|
||||
|
||||
|
||||
def overlap_metrics(a: Box, b: Box) -> Dict[str, float]:
|
||||
"""Width, height, area and ratio of the intersection of ``a`` and ``b``.
|
||||
|
||||
``overlap_ratio`` is the shared area over the *smaller* box's area, so a
|
||||
small box swallowed by a big one reads as the severe case it is.
|
||||
"""
|
||||
overlap_width = min(a.right, b.right) - max(a.left, b.left)
|
||||
overlap_height = min(a.top, b.top) - max(a.bottom, b.bottom)
|
||||
overlap_area = max(0.0, overlap_width) * max(0.0, overlap_height)
|
||||
smaller = min(a.area, b.area)
|
||||
ratio = overlap_area / smaller if smaller > 0 else 0.0
|
||||
return {
|
||||
"overlap_width": overlap_width,
|
||||
"overlap_height": overlap_height,
|
||||
"overlap_area": overlap_area,
|
||||
"overlap_ratio": ratio,
|
||||
}
|
||||
|
||||
|
||||
def classify_overlap(metrics: Dict[str, float]) -> str:
|
||||
"""Severity bucket for an overlap, following spec 7.2.
|
||||
|
||||
Zero area is no conflict at all; a ratio above ``SEVERE_OVERLAP_RATIO`` is
|
||||
severe regardless of absolute size; otherwise the vertical penetration is
|
||||
bucketed into tolerance / warning / probable / severe.
|
||||
"""
|
||||
height = metrics["overlap_height"]
|
||||
if metrics["overlap_area"] <= 0:
|
||||
return OVERLAP_NONE
|
||||
if metrics["overlap_ratio"] > SEVERE_OVERLAP_RATIO:
|
||||
return OVERLAP_SEVERE
|
||||
if height <= 5:
|
||||
return OVERLAP_RENDER_TOLERANCE
|
||||
if height <= 20:
|
||||
return OVERLAP_WARNING
|
||||
if height <= 50:
|
||||
return OVERLAP_PROBABLE
|
||||
return OVERLAP_SEVERE
|
||||
|
||||
|
||||
def distance_between(a: Box, b: Box) -> Dict[str, float]:
|
||||
"""Gap between two non-overlapping boxes, per axis and euclidean (spec 8)."""
|
||||
if a.right < b.left:
|
||||
distance_x = b.left - a.right
|
||||
elif b.right < a.left:
|
||||
distance_x = a.left - b.right
|
||||
else:
|
||||
distance_x = 0.0
|
||||
|
||||
if a.top < b.bottom:
|
||||
distance_y = b.bottom - a.top
|
||||
elif b.top < a.bottom:
|
||||
distance_y = a.bottom - b.top
|
||||
else:
|
||||
distance_y = 0.0
|
||||
|
||||
return {
|
||||
"distance_x": distance_x,
|
||||
"distance_y": distance_y,
|
||||
"distance": hypot(distance_x, distance_y),
|
||||
}
|
||||
|
||||
|
||||
def separation_suggestion(
|
||||
a: Box, b: Box, min_gap: float = 0.0
|
||||
) -> Dict[str, float]:
|
||||
"""The minimum translation that separates two overlapping boxes (spec 10).
|
||||
|
||||
Picks the smallest of the four penetrations (move left/right/up/down) and
|
||||
reports that axis plus the required movement (penetration + ``min_gap``).
|
||||
"""
|
||||
move_left = a.right - b.left
|
||||
move_right = b.right - a.left
|
||||
move_down = a.top - b.bottom
|
||||
move_up = b.top - a.bottom
|
||||
candidates = [
|
||||
("horizontal", move_left),
|
||||
("horizontal", move_right),
|
||||
("vertical", move_down),
|
||||
("vertical", move_up),
|
||||
]
|
||||
axis, penetration = min(candidates, key=lambda kv: kv[1])
|
||||
return {
|
||||
"axis": axis,
|
||||
"minimum_movement": max(0.0, penetration + min_gap),
|
||||
}
|
||||
|
||||
|
||||
def measure_title_box(
|
||||
text: str,
|
||||
font_size: float,
|
||||
*,
|
||||
x: float,
|
||||
y: float,
|
||||
font: Optional[str] = None,
|
||||
face: Optional[str] = None,
|
||||
kerning: float = 0.0,
|
||||
) -> Box:
|
||||
"""The on-screen box of one title, measured in the emitted template space.
|
||||
|
||||
``font_size``/``kerning``/``x``/``y`` are the values the writer put into the
|
||||
FCPXML, so the box is comparable across every title in the document without
|
||||
any further scaling. Width comes from the real advance table, vertical
|
||||
extent from the real ink (accents and descenders included); the anchor is
|
||||
the title's centre.
|
||||
"""
|
||||
width = measure_text(
|
||||
text, font_size, kerning=kerning, font=font, face=face
|
||||
)
|
||||
top, bottom = ink_extent(text, font_size, font=font, face=face)
|
||||
return Box(
|
||||
left=x - width / 2,
|
||||
right=x + width / 2,
|
||||
bottom=y + bottom,
|
||||
top=y + top,
|
||||
)
|
||||
|
||||
|
||||
def _is_font_measured(font: Optional[str], face: Optional[str]) -> bool:
|
||||
"""Whether both advance and vertical metrics for ``font``/``face`` exist."""
|
||||
if metrics_for(font, face) is None:
|
||||
return False
|
||||
_, measured = vertical_metrics_for(font, face)
|
||||
return measured
|
||||
|
||||
|
||||
def _box_within(box: Box, limits: Box) -> bool:
|
||||
return (
|
||||
box.left >= limits.left
|
||||
and box.right <= limits.right
|
||||
and box.bottom >= limits.bottom
|
||||
and box.top <= limits.top
|
||||
)
|
||||
|
||||
|
||||
def _issue(severity: str, type_: str, **fields) -> Dict:
|
||||
return {"severity": severity, "type": type_, **fields}
|
||||
|
||||
|
||||
def validate_titles(
|
||||
titles: Sequence[Dict],
|
||||
frame_width: float,
|
||||
frame_height: float,
|
||||
*,
|
||||
safe_margin_x: float = 0.05,
|
||||
safe_margin_y: float = 0.05,
|
||||
min_font_size: Optional[float] = None,
|
||||
min_distance: Optional[float] = None,
|
||||
max_distance: Optional[float] = None,
|
||||
) -> Dict:
|
||||
"""Validate a set of already-positioned titles and return a report.
|
||||
|
||||
Each title dict must carry the emitted values:
|
||||
|
||||
- ``text`` (str)
|
||||
- ``font_size`` (float), ``kerning`` (float), ``x``/``y`` (floats)
|
||||
- ``font`` (str) and ``face`` (str|None)
|
||||
- ``start``/``end`` (seconds) for temporal overlap
|
||||
- ``group`` (hashable) for spacing checks: titles sharing a group are one
|
||||
block, expected to sit near each other (spec 14). Optional.
|
||||
|
||||
Returns ``{"severity", "issues", "summary"}`` where ``severity`` is the
|
||||
worst bucket seen and ``issues`` are the spec-16-shaped occurrences.
|
||||
"""
|
||||
frame_left = -frame_width / 2
|
||||
frame_right = frame_width / 2
|
||||
frame_bottom = -frame_height / 2
|
||||
frame_top = frame_height / 2
|
||||
frame_box = Box(frame_left, frame_right, frame_bottom, frame_top)
|
||||
|
||||
safe_box = Box(
|
||||
left=frame_left + safe_margin_x * frame_width,
|
||||
right=frame_right - safe_margin_x * frame_width,
|
||||
bottom=frame_bottom + safe_margin_y * frame_height,
|
||||
top=frame_top - safe_margin_y * frame_height,
|
||||
)
|
||||
|
||||
issues: List[Dict] = []
|
||||
boxes: List[Box] = []
|
||||
measured_flags: List[bool] = []
|
||||
for title in titles:
|
||||
text = str(title.get("text", "") or "")
|
||||
font = title.get("font") or None
|
||||
face = title.get("face") or None
|
||||
box = measure_title_box(
|
||||
text,
|
||||
float(title.get("font_size", 0.0)),
|
||||
x=float(title.get("x", 0.0)),
|
||||
y=float(title.get("y", 0.0)),
|
||||
font=font,
|
||||
face=face,
|
||||
kerning=float(title.get("kerning", 0.0)),
|
||||
)
|
||||
boxes.append(box)
|
||||
measured_flags.append(_is_font_measured(font, face))
|
||||
|
||||
if not _is_font_measured(font, face):
|
||||
issues.append(
|
||||
_issue(
|
||||
"warning",
|
||||
FONT_MISSING,
|
||||
title=text,
|
||||
font=font,
|
||||
face=face,
|
||||
message=(
|
||||
f"Font '{font or '?'}"
|
||||
+ (f" {face}" if face else "")
|
||||
+ "' has no embedded metrics; widths are estimated"
|
||||
),
|
||||
)
|
||||
)
|
||||
if min_font_size is not None and float(title.get("font_size", 0.0)) < min_font_size:
|
||||
issues.append(
|
||||
_issue(
|
||||
"warning",
|
||||
FONT_TOO_SMALL,
|
||||
title=text,
|
||||
font_size=float(title.get("font_size", 0.0)),
|
||||
minimum=min_font_size,
|
||||
)
|
||||
)
|
||||
if not _box_within(box, frame_box):
|
||||
issues.append(
|
||||
_issue(
|
||||
"error",
|
||||
OUTSIDE_FRAME,
|
||||
title=text,
|
||||
left=box.left,
|
||||
right=box.right,
|
||||
bottom=box.bottom,
|
||||
top=box.top,
|
||||
)
|
||||
)
|
||||
elif not _box_within(box, safe_box):
|
||||
issues.append(
|
||||
_issue(
|
||||
"warning",
|
||||
OUTSIDE_SAFE_AREA,
|
||||
title=text,
|
||||
left=box.left,
|
||||
right=box.right,
|
||||
bottom=box.bottom,
|
||||
top=box.top,
|
||||
)
|
||||
)
|
||||
|
||||
# Spatial collisions between temporally overlapping titles.
|
||||
for i in range(len(titles)):
|
||||
for j in range(i + 1, len(titles)):
|
||||
a, b = titles[i], titles[j]
|
||||
if not temporal_overlap(
|
||||
float(a.get("start", 0.0)), float(a.get("end", 0.0)),
|
||||
float(b.get("start", 0.0)), float(b.get("end", 0.0)),
|
||||
):
|
||||
continue
|
||||
box_a, box_b = boxes[i], boxes[j]
|
||||
if not box_a.overlaps(box_b):
|
||||
continue
|
||||
metrics = overlap_metrics(box_a, box_b)
|
||||
severity = classify_overlap(metrics)
|
||||
issues.append(
|
||||
_issue(
|
||||
severity,
|
||||
SPATIAL_COLLISION,
|
||||
first_title=str(a.get("text", "")),
|
||||
second_title=str(b.get("text", "")),
|
||||
time_start=float(a.get("start", 0.0)),
|
||||
time_end=float(b.get("end", 0.0)),
|
||||
overlap_width=metrics["overlap_width"],
|
||||
overlap_height=metrics["overlap_height"],
|
||||
overlap_area=metrics["overlap_area"],
|
||||
overlap_ratio=metrics["overlap_ratio"],
|
||||
suggested_correction=separation_suggestion(box_a, box_b),
|
||||
)
|
||||
)
|
||||
|
||||
# Spacing within a block (spec 8/14). Only when the caller asked for it —
|
||||
# a generic minimum can fire on the reference look's own tight stacking.
|
||||
if min_distance is not None or max_distance is not None:
|
||||
groups: Dict = {}
|
||||
for index, title in enumerate(titles):
|
||||
groups.setdefault(title.get("group", index), []).append(index)
|
||||
for members in groups.values():
|
||||
for m in range(len(members)):
|
||||
for n in range(m + 1, len(members)):
|
||||
i, j = members[m], members[n]
|
||||
box_a, box_b = boxes[i], boxes[j]
|
||||
if box_a.overlaps(box_b):
|
||||
continue
|
||||
gap = distance_between(box_a, box_b)["distance"]
|
||||
if min_distance is not None and gap < min_distance:
|
||||
issues.append(
|
||||
_issue(
|
||||
"warning",
|
||||
INSUFFICIENT_SPACING,
|
||||
first_title=str(titles[i].get("text", "")),
|
||||
second_title=str(titles[j].get("text", "")),
|
||||
distance=gap,
|
||||
minimum=min_distance,
|
||||
)
|
||||
)
|
||||
if max_distance is not None and gap > max_distance:
|
||||
issues.append(
|
||||
_issue(
|
||||
"warning",
|
||||
EXCESSIVE_SPACING,
|
||||
first_title=str(titles[i].get("text", "")),
|
||||
second_title=str(titles[j].get("text", "")),
|
||||
distance=gap,
|
||||
maximum=max_distance,
|
||||
)
|
||||
)
|
||||
|
||||
_rank = {
|
||||
OVERLAP_NONE: 0,
|
||||
OVERLAP_RENDER_TOLERANCE: 1,
|
||||
OVERLAP_WARNING: 2,
|
||||
OVERLAP_PROBABLE: 3,
|
||||
OVERLAP_SEVERE: 4,
|
||||
}
|
||||
severities = [issue["severity"] for issue in issues]
|
||||
worst = max(severities, key=lambda s: _rank.get(s, 0), default=OVERLAP_NONE)
|
||||
|
||||
return {
|
||||
"severity": worst,
|
||||
"issues": issues,
|
||||
"summary": {
|
||||
"title_count": len(titles),
|
||||
"issue_count": len(issues),
|
||||
"spatial_collision": sum(
|
||||
1 for i in issues if i["type"] == SPATIAL_COLLISION
|
||||
),
|
||||
"outside_frame": sum(
|
||||
1 for i in issues if i["type"] == OUTSIDE_FRAME
|
||||
),
|
||||
"outside_safe_area": sum(
|
||||
1 for i in issues if i["type"] == OUTSIDE_SAFE_AREA
|
||||
),
|
||||
"font_missing": sum(
|
||||
1 for i in issues if i["type"] == FONT_MISSING
|
||||
),
|
||||
"font_too_small": sum(
|
||||
1 for i in issues if i["type"] == FONT_TOO_SMALL
|
||||
),
|
||||
"insufficient_spacing": sum(
|
||||
1 for i in issues if i["type"] == INSUFFICIENT_SPACING
|
||||
),
|
||||
"excessive_spacing": sum(
|
||||
1 for i in issues if i["type"] == EXCESSIVE_SPACING
|
||||
),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def blocking(severity: str) -> bool:
|
||||
"""Whether a validation severity should block export (spec 16)."""
|
||||
return severity in (OVERLAP_SEVERE, OVERLAP_PROBABLE)
|
||||
+31
-1
@@ -37,6 +37,36 @@ def diarization_capability(token: Optional[str]) -> Tuple[bool, str]:
|
||||
return True, "Identificação de participantes disponível."
|
||||
|
||||
|
||||
def _load_waveform(path: str) -> Optional[dict]:
|
||||
"""Decode ``path`` ourselves into the waveform dict pyannote accepts.
|
||||
|
||||
pyannote 4.x decodes audio through torchcodec, which links against a
|
||||
specific FFmpeg major version and fails outright when the installed one
|
||||
differs (``libavutil.56.dylib`` not found) — taking diarization down on
|
||||
an otherwise working machine. Handing it an already-decoded waveform
|
||||
skips that path entirely and reuses the ffmpeg extraction the acoustic
|
||||
analysis already relies on, so video containers work too.
|
||||
|
||||
Returns ``None`` when decoding is not possible, letting the caller fall
|
||||
back to passing the path and whatever pyannote can do with it.
|
||||
"""
|
||||
try:
|
||||
import soundfile
|
||||
import torch
|
||||
|
||||
from .voice_features import decodable_audio
|
||||
|
||||
with decodable_audio(path) as audio_path:
|
||||
if audio_path is None:
|
||||
return None
|
||||
data, sample_rate = soundfile.read(audio_path, dtype="float32", always_2d=True)
|
||||
# soundfile gives (samples, channels); pyannote wants (channels, samples)
|
||||
return {"waveform": torch.from_numpy(data.T), "sample_rate": int(sample_rate)}
|
||||
except Exception:
|
||||
logger.info("could not pre-decode %s for diarization", path)
|
||||
return None
|
||||
|
||||
|
||||
def diarize(
|
||||
path: str,
|
||||
token: Optional[str],
|
||||
@@ -67,7 +97,7 @@ def diarize(
|
||||
n = str(num_speakers or "").strip()
|
||||
if n.isdigit() and int(n) > 0:
|
||||
kwargs["num_speakers"] = int(n)
|
||||
result = pipe(path, **kwargs)
|
||||
result = pipe(_load_waveform(path) or path, **kwargs)
|
||||
# pyannote.audio >= 4.0 wraps the annotation; normalize to the raw one.
|
||||
if hasattr(result, "exclusive_speaker_diarization"):
|
||||
result = result.exclusive_speaker_diarization
|
||||
|
||||
@@ -0,0 +1,133 @@
|
||||
"""Emphasis index — how much a spoken word "pops" acoustically.
|
||||
|
||||
Pure functions over already-extracted per-word features (energy, pitch
|
||||
delta, rate delta, pause before, duration); no I/O, no external dependency.
|
||||
Combines them into a single ``[0, 1]`` score, configurable via
|
||||
:class:`EmphasisWeights` so the weighting can be tuned (and persisted,
|
||||
see ``model_manager.load_voice_analysis_config``) without touching code.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import List, Sequence
|
||||
|
||||
_FIELDS = ("energy", "pitch_variation", "rate_variation", "pause_before", "duration")
|
||||
|
||||
|
||||
@dataclass
|
||||
class EmphasisWeights:
|
||||
energy: float = 0.30
|
||||
pitch_variation: float = 0.25
|
||||
rate_variation: float = 0.20
|
||||
pause_before: float = 0.15
|
||||
duration: float = 0.10
|
||||
|
||||
def as_dict(self) -> dict:
|
||||
return {field: getattr(self, field) for field in _FIELDS}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, data: dict) -> "EmphasisWeights":
|
||||
defaults = cls()
|
||||
return cls(**{field: float(data.get(field, getattr(defaults, field))) for field in _FIELDS})
|
||||
|
||||
|
||||
def _clamp01(x: float) -> float:
|
||||
return max(0.0, min(1.0, x))
|
||||
|
||||
|
||||
def pause_weight(
|
||||
pause_before: float, max_pause: float = 1.5, ignore_above: float = 3.0
|
||||
) -> float:
|
||||
"""How much a preceding silence counts as emphasis, in ``[0, 1]``.
|
||||
|
||||
A short beat before a word is real emphasis: the speaker is setting it
|
||||
up. A *long* gap is not — it is an edit point, a B-roll insert, or the
|
||||
other person in the room talking. Measured on real footage, gaps of
|
||||
6-9s were scoring as the most emphatic moments in the recording purely
|
||||
because the scale saturated, ranking a scene change above a word the
|
||||
speaker actually hit hard.
|
||||
|
||||
So the contribution rises up to ``max_pause`` and then drops to zero
|
||||
past ``ignore_above``, instead of saturating. Set ``ignore_above`` to
|
||||
``0`` to disable the cutoff and keep the old saturating behaviour.
|
||||
"""
|
||||
if pause_before <= 0 or max_pause <= 0:
|
||||
return 0.0
|
||||
if ignore_above > 0 and pause_before > ignore_above:
|
||||
return 0.0
|
||||
return _clamp01(pause_before / max_pause)
|
||||
|
||||
|
||||
def compute_emphasis(
|
||||
energy: float,
|
||||
pitch_delta: float,
|
||||
rate_delta: float,
|
||||
pause_before: float,
|
||||
word_duration: float,
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
*,
|
||||
max_pause: float = 1.5,
|
||||
max_duration: float = 1.0,
|
||||
pause_ignore_above: float = 3.0,
|
||||
) -> float:
|
||||
"""Emphasis score in ``[0, 1]`` for one word.
|
||||
|
||||
``energy``/``pitch_delta``/``rate_delta`` are expected already
|
||||
normalized to roughly ``[0, 1]`` (deltas may be negative — only their
|
||||
magnitude counts as emphasis). ``pause_before``/``word_duration`` are
|
||||
raw seconds; duration saturates at ``max_duration``, while the pause
|
||||
contribution is shaped by :func:`pause_weight`.
|
||||
"""
|
||||
energy_n = _clamp01(energy)
|
||||
pitch_n = _clamp01(abs(pitch_delta))
|
||||
rate_n = _clamp01(abs(rate_delta))
|
||||
pause_n = pause_weight(pause_before, max_pause, pause_ignore_above)
|
||||
duration_n = _clamp01(word_duration / max_duration) if max_duration > 0 else 0.0
|
||||
|
||||
total_weight = sum(getattr(weights, field) for field in _FIELDS)
|
||||
if total_weight <= 0:
|
||||
return 0.0
|
||||
|
||||
score = (
|
||||
weights.energy * energy_n
|
||||
+ weights.pitch_variation * pitch_n
|
||||
+ weights.rate_variation * rate_n
|
||||
+ weights.pause_before * pause_n
|
||||
+ weights.duration * duration_n
|
||||
)
|
||||
return _clamp01(score / total_weight)
|
||||
|
||||
|
||||
def annotate_emphasis(
|
||||
words: Sequence[dict],
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
*,
|
||||
max_pause: float = 1.5,
|
||||
max_duration: float = 1.0,
|
||||
pause_ignore_above: float = 3.0,
|
||||
) -> List[dict]:
|
||||
"""Return copies of ``words`` with an ``"emphasis"`` key added.
|
||||
|
||||
Each word dict is expected to carry ``energy``, ``pitch_delta``,
|
||||
``rate_delta``, ``pause_before`` (all pre-computed, e.g. by
|
||||
``voice_features.py``), plus ``start``/``end`` — or an explicit
|
||||
``duration`` — to derive word length.
|
||||
"""
|
||||
out: List[dict] = []
|
||||
for w in words:
|
||||
ww = dict(w)
|
||||
duration = ww.get("duration")
|
||||
if duration is None:
|
||||
duration = max(0.0, float(ww.get("end", 0.0)) - float(ww.get("start", 0.0)))
|
||||
ww["emphasis"] = compute_emphasis(
|
||||
energy=float(ww.get("energy") or 0.0),
|
||||
pitch_delta=float(ww.get("pitch_delta") or 0.0),
|
||||
rate_delta=float(ww.get("rate_delta") or 0.0),
|
||||
pause_before=float(ww.get("pause_before") or 0.0),
|
||||
word_duration=float(duration),
|
||||
weights=weights,
|
||||
max_pause=max_pause,
|
||||
max_duration=max_duration,
|
||||
pause_ignore_above=pause_ignore_above,
|
||||
)
|
||||
out.append(ww)
|
||||
return out
|
||||
@@ -359,3 +359,291 @@ def save_num_speakers(num: str) -> str:
|
||||
data["num_speakers"] = val
|
||||
_write_config(data)
|
||||
return val
|
||||
|
||||
|
||||
DEFAULT_VOICE_ANALYSIS_CONFIG: dict = {
|
||||
"energy_threshold": 0.5,
|
||||
"emphasis_weights": {
|
||||
"energy": 0.30,
|
||||
"pitch_variation": 0.25,
|
||||
"rate_variation": 0.20,
|
||||
"pause_before": 0.15,
|
||||
"duration": 0.10,
|
||||
},
|
||||
# Peaks are selected RELATIVELY — the top slice of the distribution —
|
||||
# because the emphasis index is a weighted average whose real range
|
||||
# depends on the material. Measured on a 17-minute interview the index
|
||||
# never passed 0.55, so any absolute cutoff near the spec's 0.85 selects
|
||||
# nothing; on punchier material the same cutoff would flood the edit.
|
||||
# 2% of words is roughly one highlight every 50 words.
|
||||
"peak_percentile": 0.02,
|
||||
# Guard for genuinely flat audio, where even the top of the distribution
|
||||
# carries no emphasis worth cutting on.
|
||||
"emphasis_floor": 0.25,
|
||||
"emotion_enabled": False,
|
||||
"emotion_sensitivity": 0.5,
|
||||
}
|
||||
|
||||
|
||||
def load_voice_analysis_config() -> dict:
|
||||
"""The persisted voice-analysis thresholds/weights, merged over defaults.
|
||||
|
||||
Backs the "Análise de Voz" settings screen: energy threshold (how loud
|
||||
counts as "high energy"), the emphasis-index weights (see
|
||||
``emphasis.EmphasisWeights``), the punch-in emphasis cutoff, and the
|
||||
emotion-detection toggle/sensitivity. Unknown/malformed stored values
|
||||
fall back to the default rather than raising, so a hand-edited or
|
||||
partially-written config.json never breaks the settings screen.
|
||||
"""
|
||||
cfg = {
|
||||
**DEFAULT_VOICE_ANALYSIS_CONFIG,
|
||||
"emphasis_weights": dict(DEFAULT_VOICE_ANALYSIS_CONFIG["emphasis_weights"]),
|
||||
}
|
||||
stored = _load_config().get("voice_analysis")
|
||||
if not isinstance(stored, dict):
|
||||
return cfg
|
||||
for key in ("energy_threshold", "peak_percentile", "emphasis_floor", "emotion_sensitivity"):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = max(0.0, min(1.0, float(stored[key])))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if "emotion_enabled" in stored:
|
||||
cfg["emotion_enabled"] = bool(stored["emotion_enabled"])
|
||||
weights = stored.get("emphasis_weights")
|
||||
if isinstance(weights, dict):
|
||||
for key in cfg["emphasis_weights"]:
|
||||
if key in weights:
|
||||
try:
|
||||
cfg["emphasis_weights"][key] = max(0.0, float(weights[key]))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
return cfg
|
||||
|
||||
|
||||
def save_voice_analysis_config(
|
||||
energy_threshold: float | None = None,
|
||||
emphasis_weights: dict | None = None,
|
||||
peak_percentile: float | None = None,
|
||||
emphasis_floor: float | None = None,
|
||||
emotion_enabled: bool | None = None,
|
||||
emotion_sensitivity: float | None = None,
|
||||
) -> dict:
|
||||
"""Persist voice-analysis thresholds/weights. Only given fields change.
|
||||
|
||||
Returns the full merged config (same shape as
|
||||
:func:`load_voice_analysis_config`) so callers can render it back
|
||||
immediately without a second round-trip.
|
||||
"""
|
||||
cfg = load_voice_analysis_config()
|
||||
if energy_threshold is not None:
|
||||
cfg["energy_threshold"] = max(0.0, min(1.0, float(energy_threshold)))
|
||||
if peak_percentile is not None:
|
||||
cfg["peak_percentile"] = max(0.0, min(1.0, float(peak_percentile)))
|
||||
if emphasis_floor is not None:
|
||||
cfg["emphasis_floor"] = max(0.0, min(1.0, float(emphasis_floor)))
|
||||
if emotion_enabled is not None:
|
||||
cfg["emotion_enabled"] = bool(emotion_enabled)
|
||||
if emotion_sensitivity is not None:
|
||||
cfg["emotion_sensitivity"] = max(0.0, min(1.0, float(emotion_sensitivity)))
|
||||
if emphasis_weights is not None:
|
||||
for key, value in emphasis_weights.items():
|
||||
if key in cfg["emphasis_weights"] and value is not None:
|
||||
cfg["emphasis_weights"][key] = max(0.0, float(value))
|
||||
data = _load_config()
|
||||
data["voice_analysis"] = cfg
|
||||
_write_config(data)
|
||||
return cfg
|
||||
|
||||
|
||||
# Mirrors the "Legendas Dinâmicas" tab's own defaults (MacApp/Sources/
|
||||
# CaptionsView.swift), so a fresh install shows the same look in the UI and
|
||||
# in what generate_dynamic_subtitles renders when no override is passed.
|
||||
DEFAULT_DYNAMIC_SUBTITLE_CONFIG: dict = {
|
||||
"band_height": 0.22,
|
||||
"block_center_y": -167.0,
|
||||
"line_gap": 8.0,
|
||||
"font": "Helvetica Neue",
|
||||
"font_size": 104,
|
||||
"emphasis_font": "Playfair Display",
|
||||
"emphasis_face": "Medium Italic",
|
||||
"emphasis_size": 265,
|
||||
"active_color": "1 1 1 1",
|
||||
"emphasis_color": "1 1 1 1",
|
||||
"text_scale": 2.0,
|
||||
}
|
||||
|
||||
|
||||
def load_dynamic_subtitle_config() -> dict:
|
||||
"""The persisted dynamic-subtitle style, merged over defaults.
|
||||
|
||||
Backs the "Legendas Dinâmicas" settings screen AND is the fallback
|
||||
``generate_dynamic_subtitles`` reads for any field the caller doesn't
|
||||
explicitly override — so the style configured in the UI is what actually
|
||||
renders, without the app having to thread every field through each call.
|
||||
Unknown/malformed stored values fall back to the default, same as
|
||||
:func:`load_voice_analysis_config`.
|
||||
"""
|
||||
cfg = dict(DEFAULT_DYNAMIC_SUBTITLE_CONFIG)
|
||||
stored = _load_config().get("dynamic_subtitles")
|
||||
if not isinstance(stored, dict):
|
||||
return cfg
|
||||
for key in ("band_height", "block_center_y", "line_gap", "text_scale"):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = float(stored[key])
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
for key in ("font_size", "emphasis_size"):
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = int(stored[key])
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
for key in ("font", "emphasis_font", "emphasis_face", "active_color", "emphasis_color"):
|
||||
if key in stored and isinstance(stored[key], str) and stored[key]:
|
||||
cfg[key] = stored[key]
|
||||
return cfg
|
||||
|
||||
|
||||
def save_dynamic_subtitle_config(**fields) -> dict:
|
||||
"""Persist dynamic-subtitle style fields. Only given fields change.
|
||||
|
||||
Accepts the same keys as :data:`DEFAULT_DYNAMIC_SUBTITLE_CONFIG`; unknown
|
||||
keys are ignored so a newer app talking to an older config shape degrades
|
||||
quietly. Returns the full merged config, mirroring
|
||||
:func:`save_voice_analysis_config`.
|
||||
"""
|
||||
cfg = load_dynamic_subtitle_config()
|
||||
for key, value in fields.items():
|
||||
if key not in DEFAULT_DYNAMIC_SUBTITLE_CONFIG or value is None:
|
||||
continue
|
||||
if isinstance(DEFAULT_DYNAMIC_SUBTITLE_CONFIG[key], float):
|
||||
try:
|
||||
cfg[key] = float(value)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
elif isinstance(DEFAULT_DYNAMIC_SUBTITLE_CONFIG[key], int):
|
||||
try:
|
||||
cfg[key] = int(value)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
else:
|
||||
cfg[key] = str(value)
|
||||
data = _load_config()
|
||||
data["dynamic_subtitles"] = cfg
|
||||
_write_config(data)
|
||||
return cfg
|
||||
|
||||
|
||||
# Mirrors the silence thresholds the detection/removal handlers use when no
|
||||
# argument is passed (server_tools/qc.py). Persisted so the app's slider and
|
||||
# any later run agree without threading three fields through every call.
|
||||
DEFAULT_SILENCE_CONFIG: dict = {
|
||||
# dBFS below which audio counts as silence.
|
||||
"noise_db": -30.0,
|
||||
# Seconds a quiet stretch must last before it's a cut candidate.
|
||||
"min_silence": 0.5,
|
||||
# Seconds left inside each cut so speech never gets clipped at the edges.
|
||||
"padding": 0.05,
|
||||
}
|
||||
|
||||
|
||||
def load_silence_config() -> dict:
|
||||
"""The persisted silence-detection thresholds, merged over defaults.
|
||||
|
||||
Read by ``detect_media_silence``/``remove_media_silence`` as their
|
||||
fallback, so the tolerance chosen in the app is what actually runs.
|
||||
Malformed stored values fall back to the default rather than raising,
|
||||
matching :func:`load_voice_analysis_config`.
|
||||
"""
|
||||
cfg = dict(DEFAULT_SILENCE_CONFIG)
|
||||
stored = _load_config().get("silence")
|
||||
if not isinstance(stored, dict):
|
||||
return cfg
|
||||
for key in cfg:
|
||||
if key in stored:
|
||||
try:
|
||||
cfg[key] = float(stored[key])
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
return cfg
|
||||
|
||||
|
||||
def save_silence_config(
|
||||
noise_db: float | None = None,
|
||||
min_silence: float | None = None,
|
||||
padding: float | None = None,
|
||||
) -> dict:
|
||||
"""Persist silence thresholds. Only the given fields change.
|
||||
|
||||
Values are clamped to the same ranges the handlers validate against, so
|
||||
a bad write here can't produce a config the tools would later reject.
|
||||
"""
|
||||
cfg = load_silence_config()
|
||||
if noise_db is not None:
|
||||
try:
|
||||
cfg["noise_db"] = max(-120.0, min(0.0, float(noise_db)))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if min_silence is not None:
|
||||
try:
|
||||
cfg["min_silence"] = max(0.01, min(3600.0, float(min_silence)))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if padding is not None:
|
||||
try:
|
||||
cfg["padding"] = max(0.0, min(5.0, float(padding)))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
data = _load_config()
|
||||
data["silence"] = cfg
|
||||
_write_config(data)
|
||||
return cfg
|
||||
|
||||
|
||||
# Last project worked on, so the app reopens where the user left off instead of
|
||||
# making them pick the folder again every launch. Only paths that still exist
|
||||
# are handed back — a project on an unmounted volume degrades to "none selected"
|
||||
# rather than to a dead path the tools would later fail on.
|
||||
DEFAULT_PROJECT_CONFIG: dict = {
|
||||
# Folder every generated file (transcript .json, XML, SRT) is written to.
|
||||
"folder": "",
|
||||
# The .fcpxml/.fcpxmld that was loaded from it.
|
||||
"file": "",
|
||||
}
|
||||
|
||||
|
||||
def load_project_config() -> dict:
|
||||
"""The persisted last project (folder + file), merged over defaults.
|
||||
|
||||
Paths that no longer exist on disk come back empty, matching what the app
|
||||
shows for "nothing selected". Malformed stored values fall back to the
|
||||
default rather than raising, same as :func:`load_voice_analysis_config`.
|
||||
"""
|
||||
cfg = dict(DEFAULT_PROJECT_CONFIG)
|
||||
stored = _load_config().get("project")
|
||||
if not isinstance(stored, dict):
|
||||
return cfg
|
||||
for key in cfg:
|
||||
value = stored.get(key)
|
||||
if isinstance(value, str) and value and Path(value).exists():
|
||||
cfg[key] = value
|
||||
return cfg
|
||||
|
||||
|
||||
def save_project_config(folder: str | None = None, file: str | None = None) -> dict:
|
||||
"""Persist the last project folder/file. Only the given fields change.
|
||||
|
||||
Passing an empty string clears a field (the app does this when the user
|
||||
deselects), while ``None`` leaves it untouched.
|
||||
"""
|
||||
cfg = load_project_config()
|
||||
for key, value in (("folder", folder), ("file", file)):
|
||||
if value is None:
|
||||
continue
|
||||
cfg[key] = str(Path(value).expanduser()) if str(value).strip() else ""
|
||||
data = _load_config()
|
||||
data["project"] = cfg
|
||||
_write_config(data)
|
||||
return cfg
|
||||
|
||||
@@ -13,6 +13,8 @@ from functools import total_ordering
|
||||
from math import gcd
|
||||
from typing import Any, Callable, Dict, List, Optional, Tuple
|
||||
|
||||
from .text_layout import REFERENCE_BLOCK_LINE_GAP, TEXT_TEMPLATE_FONT_SCALE
|
||||
|
||||
# ============================================================================
|
||||
# ENUMS
|
||||
# ============================================================================
|
||||
@@ -1071,3 +1073,19 @@ class DynamicSubtitleConfig:
|
||||
# grouped, the key word alone and large (the reference look). "word": one
|
||||
# title per word, the earlier rhythm.
|
||||
granularity: str = "phrase"
|
||||
# Ratio between the template's fontSize space and the canvas-point space
|
||||
# its Position uses. See text_layout.TEXT_TEMPLATE_FONT_SCALE: the "Text"
|
||||
# (Text.moti) template sizes type in frame pixels, so a size chosen in
|
||||
# points renders half as large unless it is converted on the way out.
|
||||
text_scale: float = TEXT_TEMPLATE_FONT_SCALE
|
||||
# Vertical air between stacked lines, in canvas points. Negative values
|
||||
# deliberately overlap the lines — the display italic tucking under the
|
||||
# line above is a real editorial look, and the stacking arithmetic places
|
||||
# ink boxes edge to edge, so a negative gap moves them by exactly that
|
||||
# much rather than colliding unpredictably.
|
||||
line_gap: float = REFERENCE_BLOCK_LINE_GAP
|
||||
# Run the post-generation collision validation (collision.validate_titles)
|
||||
# and refuse to emit when it reports a blocking overlap. Off by default so
|
||||
# generation stays byte-identical to before this flag existed; flip it on
|
||||
# for a guaranteed no-collision export.
|
||||
validate: bool = False
|
||||
|
||||
+91
-16
@@ -122,6 +122,33 @@ REFERENCE_BLOCK_LINE_GAP = 8.0
|
||||
# emphasis line's edge; the reference leaves a little air.
|
||||
REFERENCE_STAGGER_RATIO = 0.8
|
||||
|
||||
# Extra gap, as a fraction of the emphasis line's font size, added only to
|
||||
# the boundary right below it. The display italic's slant leans its stems
|
||||
# past the vertical ink box the metrics measure, so a body line directly
|
||||
# under the emphasis line reads tighter than the same nominal gap anywhere
|
||||
# else in the stack — this cushion (~14pt at the 230pt reference size)
|
||||
# closes that optical gap without touching the user's `line_gap` elsewhere.
|
||||
_EMPHASIS_ITALIC_CUSHION_RATIO = 0.06
|
||||
|
||||
# The numbers above were read off a hand export that used the "Essencial -
|
||||
# Título" template. That template never rendered when we generated it (see
|
||||
# Engine/docs/05_EXPERIENCIAS.md, 2026-08-17), so the writer switched to FCP's
|
||||
# own "Basic Text > Text" (Text.moti) — whose coordinate space is the FRAME
|
||||
# ITSELF (2160x3840), not the half-scale point canvas the numbers above were
|
||||
# measured in. Everything the template reads is in that space: fontSize,
|
||||
# kerning AND Position alike.
|
||||
#
|
||||
# Getting this half-right is worse than getting it wrong. Scaling only the type
|
||||
# left the block at the old spread with twice the type in it, so the lines
|
||||
# collided; scaling only the positions would spread a block of half-size type
|
||||
# across the frame. The layout keeps measuring in canvas points — every
|
||||
# constant above depends on that — and this single factor converts the whole
|
||||
# result on the way out, which is the only way the two stay in step.
|
||||
#
|
||||
# Exposed as `text_scale` on DynamicSubtitleConfig for a template authored
|
||||
# against a different space.
|
||||
TEXT_TEMPLATE_FONT_SCALE = 2.0
|
||||
|
||||
|
||||
def metrics_for(font: Optional[str], face: Optional[str] = None) -> Optional[Dict]:
|
||||
"""The embedded advance table for *font*/*face*, or None if uncovered.
|
||||
@@ -296,9 +323,14 @@ class PlacedWord:
|
||||
and other.bottom < self.top
|
||||
)
|
||||
|
||||
def position_param(self) -> str:
|
||||
"""The value for the title's "Posição" param, as FCP writes it."""
|
||||
return f"{self.x:g} {self.y:g}"
|
||||
def position_param(self, scale: float = 1.0) -> str:
|
||||
"""The value for the title's "Posição" param, as FCP writes it.
|
||||
|
||||
*scale* converts from canvas points to the template's own space; see
|
||||
TEXT_TEMPLATE_FONT_SCALE. It must be the same factor the emitted
|
||||
fontSize uses, or the type and the spacing drift apart.
|
||||
"""
|
||||
return f"{self.x * scale:g} {self.y * scale:g}"
|
||||
|
||||
# The rest of this block is the interface a placed unit shares with
|
||||
# PlacedBlock, so the writer emits titles from either without caring
|
||||
@@ -649,9 +681,14 @@ class PlacedBlock:
|
||||
"""When this block finishes being spoken, in seconds."""
|
||||
return max(float(w.get('end', 0.0)) for w in self.words)
|
||||
|
||||
def position_param(self) -> str:
|
||||
"""The value for the title's "Posição" param, as FCP writes it."""
|
||||
return f"{self.x:g} {self.y:g}"
|
||||
def position_param(self, scale: float = 1.0) -> str:
|
||||
"""The value for the title's "Posição" param, as FCP writes it.
|
||||
|
||||
*scale* converts from canvas points to the template's own space; see
|
||||
TEXT_TEMPLATE_FONT_SCALE. It must be the same factor the emitted
|
||||
fontSize uses, or the type and the spacing drift apart.
|
||||
"""
|
||||
return f"{self.x * scale:g} {self.y * scale:g}"
|
||||
|
||||
def overlaps(self, other: 'PlacedBlock') -> bool:
|
||||
return (
|
||||
@@ -742,10 +779,33 @@ def compose_sentence(
|
||||
for run in body_lines(entries[emphasis_index + 1:]):
|
||||
lines.append((run, body_look, False))
|
||||
|
||||
# A body run wraps onto a new line when it doesn't fit — but the
|
||||
# emphasis line is always exactly one word, so it can't wrap, and
|
||||
# nothing capped its size against the box. A long or all-caps word (an
|
||||
# emphasis pass sometimes upper-cases its pick) could run past both
|
||||
# edges of the frame — found on real footage, wide enough to spill off
|
||||
# BOTH sides while centred. Shrinking it back to the box scales its
|
||||
# font_size and kerning by the same factor, so the ink height used for
|
||||
# stacking below shrinks with it too — restoring the vertical
|
||||
# non-overlap the rest of this function already guarantees by
|
||||
# construction. Never shrunk below the body size: emphasis smaller
|
||||
# than body text isn't emphasis anymore, it's just a different font.
|
||||
def fit_emphasis(text: str, look) -> tuple:
|
||||
size, kerning, width = measure(text, look)
|
||||
if width <= box.width:
|
||||
return size, kerning, width
|
||||
floor = float(body_look.font_size) * scale
|
||||
fit = max(box.width / width, floor / size) if size > 0 else 1.0
|
||||
fit = min(fit, 1.0)
|
||||
return size * fit, kerning * fit, width * fit
|
||||
|
||||
measured = []
|
||||
for run, look, is_emphasis in lines:
|
||||
text = ' '.join(t for _, t in run)
|
||||
size, kerning, width = measure(text, look)
|
||||
if is_emphasis:
|
||||
size, kerning, width = fit_emphasis(text, look)
|
||||
else:
|
||||
size, kerning, width = measure(text, look)
|
||||
# Stack on the real ink each line contains, not on a nominal
|
||||
# cap-height: the display italic's accents and descenders run well
|
||||
# past it, and a nominal box lets them collide with the neighbour.
|
||||
@@ -764,14 +824,29 @@ def compose_sentence(
|
||||
# loses the whole point of the look — so if it does not fit, everything
|
||||
# from the emphasis on overflows together.
|
||||
gap = line_gap * scale
|
||||
|
||||
# The emphasis line's italic slant carries visual weight below its own
|
||||
# ink box — Playfair's stems lean past what the vertical metrics measure
|
||||
# — so a body line sitting right under it reads tighter than the same
|
||||
# nominal gap elsewhere, even though the ink boxes themselves never
|
||||
# touch. Add a size-proportional cushion only to the boundary right
|
||||
# after the emphasis line; every other pair keeps exactly the caller's
|
||||
# ``line_gap``.
|
||||
def pair_gap(prev_line: dict) -> float:
|
||||
if prev_line['emphasis']:
|
||||
return gap + _EMPHASIS_ITALIC_CUSHION_RATIO * prev_line['font_size']
|
||||
return gap
|
||||
|
||||
kept = 0
|
||||
total = 0.0
|
||||
prev = None
|
||||
for line in measured:
|
||||
advance = line['height'] if not kept else line['height'] + gap
|
||||
if kept and total + advance > box.height:
|
||||
advance = line['height'] if prev is None else line['height'] + pair_gap(prev)
|
||||
if prev is not None and total + advance > box.height:
|
||||
break
|
||||
total += advance
|
||||
kept += 1
|
||||
prev = line
|
||||
kept = max(kept, 1)
|
||||
if not any(line['emphasis'] for line in measured[:kept]):
|
||||
kept = min(kept, next(
|
||||
@@ -783,12 +858,12 @@ def compose_sentence(
|
||||
result.overflow.extend(w for w, _ in line['run'])
|
||||
|
||||
visible = measured[:kept]
|
||||
# Stack the ink boxes edge to edge with exactly *gap* between them, then
|
||||
# centre the whole stack on the band. Because the boxes are the real ink,
|
||||
# "no overlap" is a property of the arithmetic, not of a safety factor.
|
||||
stack_height = (
|
||||
sum(line['height'] for line in visible) + gap * (len(visible) - 1)
|
||||
)
|
||||
# Stack the ink boxes edge to edge with exactly *gap* between them (plus
|
||||
# the emphasis cushion where it applies), then centre the whole stack on
|
||||
# the band. Because the boxes are the real ink, "no overlap" is a
|
||||
# property of the arithmetic, not of a safety factor.
|
||||
gaps = [pair_gap(visible[i - 1]) for i in range(1, len(visible))]
|
||||
stack_height = sum(line['height'] for line in visible) + sum(gaps)
|
||||
edge = box.center_y + stack_height / 2
|
||||
|
||||
# Body lines hang off the emphasis line's edges, alternating sides in
|
||||
@@ -797,7 +872,7 @@ def compose_sentence(
|
||||
side = -1
|
||||
for index, line in enumerate(visible):
|
||||
if index:
|
||||
edge -= gap
|
||||
edge -= gaps[index - 1]
|
||||
cursor_y = edge - line['ink_top']
|
||||
edge = cursor_y + line['ink_bottom']
|
||||
if line['emphasis']:
|
||||
|
||||
@@ -16,7 +16,7 @@ import logging
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import List, Optional, Sequence, Tuple
|
||||
from typing import Callable, List, Optional, Sequence, Tuple
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -118,7 +118,10 @@ def invert_ranges(
|
||||
|
||||
|
||||
def transcribe(
|
||||
path: str, model_size: str = "base", language: Optional[str] = None
|
||||
path: str,
|
||||
model_size: str = "base",
|
||||
language: Optional[str] = None,
|
||||
progress_cb: Optional[Callable[[float], None]] = None,
|
||||
) -> Optional[dict]:
|
||||
"""Transcribe an audio/video file locally with word-level timestamps.
|
||||
|
||||
@@ -170,6 +173,10 @@ def transcribe(
|
||||
)
|
||||
segments: List[dict] = []
|
||||
words: List[dict] = []
|
||||
# `info.duration` is known upfront (from the container), so each
|
||||
# segment's end time — yielded lazily as faster-whisper decodes —
|
||||
# gives real, granular progress instead of a single before/after step.
|
||||
total_duration = float(info.duration) if info.duration else 0.0
|
||||
for seg in segments_iter:
|
||||
start = float(seg.start)
|
||||
end = float(seg.end)
|
||||
@@ -182,6 +189,8 @@ def transcribe(
|
||||
"end_fmt": format_timestamp(end),
|
||||
}
|
||||
)
|
||||
if progress_cb is not None and total_duration > 0:
|
||||
progress_cb(min(end / total_duration, 1.0))
|
||||
for w in seg.words or []:
|
||||
ws = float(w.start)
|
||||
we = float(w.end)
|
||||
|
||||
@@ -0,0 +1,248 @@
|
||||
"""Voice actions — the editing decisions produced from a voice timeline.
|
||||
|
||||
This is the contract between *deciding* and *applying*. Whoever makes the
|
||||
editorial call — the deterministic rules engine, or a model reading the
|
||||
voice timeline JSON — emits the same list of actions, and one applier turns
|
||||
it into FCPXML. Nothing that produces actions ever touches XML.
|
||||
|
||||
Every action's ``start``/``end`` is in **original source seconds**, matching
|
||||
the voice timeline. That matters: cuts shift everything after them, so if
|
||||
decisions were expressed in post-cut time they would silently land in the
|
||||
wrong place the moment a cut was added. Keeping one origin and resolving the
|
||||
shift at apply time (:func:`shift_after_cuts`) removes that whole class of bug.
|
||||
|
||||
Actions arriving from a model are untrusted input: :func:`parse_actions`
|
||||
validates and reports what it rejected rather than raising, so one malformed
|
||||
row never discards a whole edit.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, List, Optional, Sequence, Tuple
|
||||
|
||||
# What an action can ask for. Deliberately small — each maps onto one
|
||||
# existing writer capability, so no new XML knowledge lives here.
|
||||
ACTION_KINDS = ("cut", "zoom", "text", "marker")
|
||||
|
||||
# Bounds for a zoom's scale factor. Below 1.0 is a pull-back, not a punch-in;
|
||||
# above 3x the image falls apart on any normal footage.
|
||||
MIN_ZOOM_SCALE = 1.0
|
||||
MAX_ZOOM_SCALE = 3.0
|
||||
|
||||
MAX_TEXT_LENGTH = 120
|
||||
|
||||
|
||||
@dataclass
|
||||
class VoiceAction:
|
||||
"""One editing decision, in original source time."""
|
||||
|
||||
kind: str
|
||||
start: float
|
||||
end: float
|
||||
params: dict = field(default_factory=dict)
|
||||
reason: str = ""
|
||||
speaker: str = ""
|
||||
|
||||
@property
|
||||
def duration(self) -> float:
|
||||
return max(0.0, self.end - self.start)
|
||||
|
||||
def as_dict(self) -> dict:
|
||||
return {
|
||||
"kind": self.kind,
|
||||
"start": round(self.start, 3),
|
||||
"end": round(self.end, 3),
|
||||
"params": self.params,
|
||||
"reason": self.reason,
|
||||
"speaker": self.speaker,
|
||||
}
|
||||
|
||||
|
||||
def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]:
|
||||
"""Turn one raw row into a VoiceAction, or explain why it can't be."""
|
||||
where = f"action[{index}]"
|
||||
if not isinstance(raw, dict):
|
||||
return None, f"{where}: expected an object, got {type(raw).__name__}"
|
||||
|
||||
kind = str(raw.get("kind", "")).strip().lower()
|
||||
if kind not in ACTION_KINDS:
|
||||
return None, f"{where}: unknown kind {raw.get('kind')!r} (expected one of {', '.join(ACTION_KINDS)})"
|
||||
|
||||
try:
|
||||
start = float(raw.get("start"))
|
||||
end = float(raw.get("end"))
|
||||
except (TypeError, ValueError):
|
||||
return None, f"{where}: start/end must be numbers (seconds)"
|
||||
|
||||
if start < 0:
|
||||
return None, f"{where}: start is negative ({start})"
|
||||
if end <= start:
|
||||
return None, f"{where}: end ({end}) must be after start ({start})"
|
||||
|
||||
params = raw.get("params")
|
||||
params = dict(params) if isinstance(params, dict) else {}
|
||||
|
||||
if kind == "zoom":
|
||||
try:
|
||||
scale = float(params.get("scale", 1.3))
|
||||
except (TypeError, ValueError):
|
||||
return None, f"{where}: zoom scale must be a number"
|
||||
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
|
||||
return None, (
|
||||
f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}"
|
||||
)
|
||||
params["scale"] = scale
|
||||
|
||||
if kind == "text":
|
||||
content = str(params.get("content", "")).strip()
|
||||
if not content:
|
||||
return None, f"{where}: text action needs params.content"
|
||||
params["content"] = content[:MAX_TEXT_LENGTH]
|
||||
|
||||
return (
|
||||
VoiceAction(
|
||||
kind=kind,
|
||||
start=start,
|
||||
end=end,
|
||||
params=params,
|
||||
reason=str(raw.get("reason", "")),
|
||||
speaker=str(raw.get("speaker", "")),
|
||||
),
|
||||
"",
|
||||
)
|
||||
|
||||
|
||||
def parse_actions(data: Any) -> Tuple[List[VoiceAction], List[str]]:
|
||||
"""Validate a decision list into actions, collecting rejections.
|
||||
|
||||
Accepts either a bare list of actions or ``{"actions": [...]}`` — the
|
||||
shape a model is most likely to return. Returns ``(actions, errors)``;
|
||||
a row that fails validation is reported and skipped, never fatal.
|
||||
"""
|
||||
if isinstance(data, dict):
|
||||
data = data.get("actions", [])
|
||||
if not isinstance(data, Sequence) or isinstance(data, (str, bytes)):
|
||||
return [], ["expected a list of actions, or an object with an 'actions' list"]
|
||||
|
||||
actions: List[VoiceAction] = []
|
||||
errors: List[str] = []
|
||||
for i, raw in enumerate(data):
|
||||
action, error = _validate_one(raw, i)
|
||||
if action is not None:
|
||||
actions.append(action)
|
||||
else:
|
||||
errors.append(error)
|
||||
return actions, errors
|
||||
|
||||
|
||||
def speaker_cut_actions(
|
||||
timeline: dict,
|
||||
speaker_ids: Sequence[str],
|
||||
padding: float = 0.15,
|
||||
) -> List[VoiceAction]:
|
||||
"""Cut actions removing everything the given speakers say.
|
||||
|
||||
The everyday case on a testimonial shoot: an interviewer or a crew
|
||||
member talks over the take, and only the subject should survive the
|
||||
edit. ``padding`` trims slightly *inside* each segment rather than
|
||||
around it — speech boundaries from a transcript are approximate, and
|
||||
eating into the neighbouring silence is far safer than clipping the
|
||||
first syllable of the person being kept.
|
||||
"""
|
||||
wanted = {str(s) for s in speaker_ids}
|
||||
actions: List[VoiceAction] = []
|
||||
for segment in timeline.get("segments", []):
|
||||
if str(segment.get("speaker", "")) not in wanted:
|
||||
continue
|
||||
start = float(segment.get("start", 0.0)) + padding
|
||||
end = float(segment.get("end", 0.0)) - padding
|
||||
if end <= start:
|
||||
continue
|
||||
actions.append(
|
||||
VoiceAction(
|
||||
kind="cut",
|
||||
start=start,
|
||||
end=end,
|
||||
reason=f"fala de {segment.get('speaker')}",
|
||||
speaker=str(segment.get("speaker", "")),
|
||||
)
|
||||
)
|
||||
return actions
|
||||
|
||||
|
||||
def merge_cut_ranges(actions: Sequence[VoiceAction]) -> List[Tuple[float, float]]:
|
||||
"""The cut actions as merged, sorted, non-overlapping source ranges."""
|
||||
cuts = sorted((a.start, a.end) for a in actions if a.kind == "cut")
|
||||
merged: List[Tuple[float, float]] = []
|
||||
for start, end in cuts:
|
||||
if merged and start <= merged[-1][1]:
|
||||
merged[-1] = (merged[-1][0], max(merged[-1][1], end))
|
||||
else:
|
||||
merged.append((start, end))
|
||||
return merged
|
||||
|
||||
|
||||
def shift_after_cuts(
|
||||
time: float, cuts: Sequence[Tuple[float, float]]
|
||||
) -> Optional[float]:
|
||||
"""Where source ``time`` lands once ``cuts`` are removed.
|
||||
|
||||
Returns ``None`` when the time falls *inside* a cut — the material it
|
||||
referred to no longer exists, so the action that pointed at it must be
|
||||
dropped rather than silently slid onto neighbouring content.
|
||||
``cuts`` must be merged and sorted (see :func:`merge_cut_ranges`).
|
||||
"""
|
||||
shift = 0.0
|
||||
for start, end in cuts:
|
||||
if time < start:
|
||||
break
|
||||
if time < end:
|
||||
return None
|
||||
shift += end - start
|
||||
return time - shift
|
||||
|
||||
|
||||
def resolve_actions(
|
||||
actions: Sequence[VoiceAction],
|
||||
) -> Tuple[List[Tuple[float, float]], List[VoiceAction], List[VoiceAction]]:
|
||||
"""Split a decision list into what to cut and what to place afterwards.
|
||||
|
||||
Returns ``(cut_ranges, placed, dropped)``. Non-cut actions are moved onto
|
||||
their post-cut times; any that pointed into removed material land in
|
||||
``dropped`` so the caller can report them instead of losing them quietly.
|
||||
"""
|
||||
cut_ranges = merge_cut_ranges(actions)
|
||||
placed: List[VoiceAction] = []
|
||||
dropped: List[VoiceAction] = []
|
||||
|
||||
for action in actions:
|
||||
if action.kind == "cut":
|
||||
continue
|
||||
new_start = shift_after_cuts(action.start, cut_ranges)
|
||||
if new_start is None:
|
||||
dropped.append(action)
|
||||
continue
|
||||
if action.kind == "marker":
|
||||
# A marker is a point, not a span: it survives as long as its own
|
||||
# instant does. Requiring its nominal end to survive too would
|
||||
# drop exactly the markers worth keeping — the ones flagging a
|
||||
# join, which sit right against a cut edge by definition.
|
||||
new_end = new_start + action.duration
|
||||
else:
|
||||
new_end = shift_after_cuts(action.end, cut_ranges)
|
||||
if new_end is None:
|
||||
dropped.append(action)
|
||||
continue
|
||||
if new_end <= new_start:
|
||||
dropped.append(action)
|
||||
continue
|
||||
placed.append(
|
||||
VoiceAction(
|
||||
kind=action.kind,
|
||||
start=new_start,
|
||||
end=new_end,
|
||||
params=action.params,
|
||||
reason=action.reason,
|
||||
speaker=action.speaker,
|
||||
)
|
||||
)
|
||||
return cut_ranges, placed, dropped
|
||||
@@ -0,0 +1,220 @@
|
||||
"""Acoustic features for voice analysis — pitch, energy, rate, pauses.
|
||||
|
||||
Mirrors the ``media_intel.py`` contract: librosa is an optional dependency
|
||||
(``pip install 'fcp-mcp-server[intelligence]'``, already required by beat
|
||||
detection), imported lazily, and every extractor degrades to ``None`` when
|
||||
the library is missing or the file cannot be analyzed — never crashes.
|
||||
|
||||
``compute_speech_rate``/``compute_pauses`` are pure functions over
|
||||
word-timestamp dicts (the shape ``transcribe.py`` already produces) and need
|
||||
no audio file at all.
|
||||
"""
|
||||
|
||||
import contextlib
|
||||
import logging
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Iterator, List, Optional, Sequence, Tuple
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Human voice fundamental frequency range (covers low male to high female/child).
|
||||
PITCH_FMIN_HZ = 65.0
|
||||
PITCH_FMAX_HZ = 1000.0
|
||||
|
||||
# Formats librosa reads directly through soundfile. Anything else — notably
|
||||
# the .mov/.mp4 that source footage actually arrives in — must be decoded by
|
||||
# ffmpeg first, or analysis fails outright.
|
||||
NATIVE_AUDIO_SUFFIXES = {".wav", ".aif", ".aiff", ".flac"}
|
||||
|
||||
# Voice analysis only needs the speech band: 16 kHz mono is well above the
|
||||
# Nyquist limit for our 1 kHz pitch ceiling, and keeps the extracted file
|
||||
# small and fast to decode even for hour-long footage.
|
||||
EXTRACT_SAMPLE_RATE = 16000
|
||||
EXTRACT_TIMEOUT_SECONDS = 600
|
||||
|
||||
|
||||
@contextlib.contextmanager
|
||||
def decodable_audio(path: str) -> Iterator[Optional[str]]:
|
||||
"""Yield a path librosa can read, extracting the audio track if needed.
|
||||
|
||||
Audio files pass straight through. Video containers are decoded to a
|
||||
temporary mono WAV with ffmpeg and cleaned up on exit. Yields ``None``
|
||||
when the audio cannot be obtained (no ffmpeg, no audio track, failure),
|
||||
keeping the graceful-degradation contract of this module.
|
||||
"""
|
||||
file_path = Path(path)
|
||||
if file_path.suffix.lower() in NATIVE_AUDIO_SUFFIXES:
|
||||
yield str(file_path)
|
||||
return
|
||||
|
||||
if shutil.which("ffmpeg") is None:
|
||||
logger.info("ffmpeg not found on PATH; cannot extract audio from %s", file_path)
|
||||
yield None
|
||||
return
|
||||
|
||||
tmp_dir = tempfile.mkdtemp(prefix="fcp_voice_")
|
||||
wav_path = Path(tmp_dir) / "audio.wav"
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[
|
||||
"ffmpeg", "-hide_banner", "-nostdin", "-y",
|
||||
"-i", str(file_path),
|
||||
"-vn", # audio only: decoding video would dominate the runtime
|
||||
"-ac", "1",
|
||||
"-ar", str(EXTRACT_SAMPLE_RATE),
|
||||
str(wav_path),
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=EXTRACT_TIMEOUT_SECONDS,
|
||||
)
|
||||
if result.returncode != 0 or not wav_path.is_file():
|
||||
logger.warning("ffmpeg could not extract audio from %s", file_path)
|
||||
yield None
|
||||
else:
|
||||
yield str(wav_path)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
logger.warning("audio extraction failed for %s", file_path)
|
||||
yield None
|
||||
finally:
|
||||
shutil.rmtree(tmp_dir, ignore_errors=True)
|
||||
|
||||
|
||||
def features_capability() -> Tuple[bool, str]:
|
||||
"""Whether pitch/energy extraction is available (librosa installed)."""
|
||||
try:
|
||||
import librosa # noqa: F401
|
||||
except Exception:
|
||||
return False, "Análise acústica indisponível: componente librosa ausente."
|
||||
return True, "Análise acústica disponível."
|
||||
|
||||
|
||||
def extract_pitch(
|
||||
path: str, hop_length: int = 512, max_analysis_seconds: float = 1200.0
|
||||
) -> Optional[List[Tuple[float, float]]]:
|
||||
"""Frame-level pitch (F0) track via librosa's ``pyin``.
|
||||
|
||||
Returns ``[(time_seconds, hz), ...]`` for voiced frames only (unvoiced
|
||||
frames, where ``pyin`` reports no pitch, are dropped), or ``None`` when
|
||||
librosa is unavailable or the file cannot be analyzed.
|
||||
"""
|
||||
file_path = Path(path)
|
||||
if not file_path.is_file():
|
||||
return None
|
||||
try:
|
||||
import librosa
|
||||
except ImportError:
|
||||
logger.info("librosa not installed; pitch extraction unavailable")
|
||||
return None
|
||||
try:
|
||||
with decodable_audio(str(file_path)) as audio_path:
|
||||
if audio_path is None:
|
||||
return None
|
||||
y, sr = librosa.load(audio_path, sr=None, mono=True, duration=max_analysis_seconds)
|
||||
f0, voiced_flag, _voiced_prob = librosa.pyin(
|
||||
y, fmin=PITCH_FMIN_HZ, fmax=PITCH_FMAX_HZ, sr=sr, hop_length=hop_length
|
||||
)
|
||||
times = librosa.times_like(f0, sr=sr, hop_length=hop_length)
|
||||
except Exception:
|
||||
logger.warning("librosa pitch analysis failed for %s", file_path)
|
||||
return None
|
||||
return [
|
||||
(float(t), float(hz))
|
||||
for t, hz, voiced in zip(times, f0, voiced_flag)
|
||||
if voiced and hz == hz # ``hz == hz`` filters NaN without importing math/numpy here
|
||||
]
|
||||
|
||||
|
||||
def extract_energy(
|
||||
path: str, hop_length: int = 512, max_analysis_seconds: float = 1200.0
|
||||
) -> Optional[List[Tuple[float, float]]]:
|
||||
"""Frame-level RMS energy track via librosa.
|
||||
|
||||
Returns ``[(time_seconds, rms), ...]``, or ``None`` when librosa is
|
||||
unavailable or the file cannot be analyzed.
|
||||
"""
|
||||
file_path = Path(path)
|
||||
if not file_path.is_file():
|
||||
return None
|
||||
try:
|
||||
import librosa
|
||||
except ImportError:
|
||||
logger.info("librosa not installed; energy extraction unavailable")
|
||||
return None
|
||||
try:
|
||||
with decodable_audio(str(file_path)) as audio_path:
|
||||
if audio_path is None:
|
||||
return None
|
||||
y, sr = librosa.load(audio_path, sr=None, mono=True, duration=max_analysis_seconds)
|
||||
rms = librosa.feature.rms(y=y, hop_length=hop_length)[0]
|
||||
times = librosa.times_like(rms, sr=sr, hop_length=hop_length)
|
||||
except Exception:
|
||||
logger.warning("librosa energy analysis failed for %s", file_path)
|
||||
return None
|
||||
return [(float(t), float(r)) for t, r in zip(times, rms)]
|
||||
|
||||
|
||||
def _window_average(track: Sequence[Tuple[float, float]], start: float, end: float) -> Optional[float]:
|
||||
"""Average of ``track`` values whose timestamp falls in ``[start, end]``."""
|
||||
values = [v for t, v in track if start <= t <= end]
|
||||
if not values:
|
||||
return None
|
||||
return sum(values) / len(values)
|
||||
|
||||
|
||||
def word_pitch_energy(
|
||||
words: Sequence[dict],
|
||||
pitch_track: Optional[Sequence[Tuple[float, float]]],
|
||||
energy_track: Optional[Sequence[Tuple[float, float]]],
|
||||
) -> List[dict]:
|
||||
"""Attach average pitch/energy over each word's ``[start, end]`` span.
|
||||
|
||||
Words carry ``pitch_hz``/``energy`` (``None`` when the span has no
|
||||
voiced frames or a track is unavailable). Both tracks are the output of
|
||||
:func:`extract_pitch`/:func:`extract_energy`.
|
||||
"""
|
||||
out: List[dict] = []
|
||||
for w in words:
|
||||
ww = dict(w)
|
||||
start = float(w.get("start", 0.0))
|
||||
end = float(w.get("end", start))
|
||||
ww["pitch_hz"] = _window_average(pitch_track, start, end) if pitch_track else None
|
||||
ww["energy"] = _window_average(energy_track, start, end) if energy_track else None
|
||||
out.append(ww)
|
||||
return out
|
||||
|
||||
|
||||
def compute_speech_rate(words: Sequence[dict], window_seconds: float = 3.0) -> List[float]:
|
||||
"""Local speech rate (words/second) around each word.
|
||||
|
||||
For word *i*, counts every word whose start falls within
|
||||
``[start_i - window_seconds, start_i]`` and divides by
|
||||
``window_seconds`` — a trailing local rate, cheap to compute and stable
|
||||
against a single long/short word skewing the whole utterance's average.
|
||||
"""
|
||||
starts = [float(w.get("start", 0.0)) for w in words]
|
||||
rates: List[float] = []
|
||||
for i, s in enumerate(starts):
|
||||
lo = s - window_seconds
|
||||
count = sum(1 for t in starts[: i + 1] if t >= lo)
|
||||
rates.append(count / window_seconds if window_seconds > 0 else 0.0)
|
||||
return rates
|
||||
|
||||
|
||||
def compute_pauses(words: Sequence[dict]) -> List[float]:
|
||||
"""Silence (seconds) immediately before each word.
|
||||
|
||||
The first word's "pause before" is the time from the start of the audio
|
||||
to its own start; every other word measures the gap since the previous
|
||||
word's end (clamped to ``0`` for overlapping/adjacent words).
|
||||
"""
|
||||
pauses: List[float] = []
|
||||
prev_end = 0.0
|
||||
for w in words:
|
||||
start = float(w.get("start", 0.0))
|
||||
pauses.append(max(0.0, start - prev_end))
|
||||
prev_end = float(w.get("end", start))
|
||||
return pauses
|
||||
@@ -0,0 +1,505 @@
|
||||
"""Voice timeline — the consolidated, AI-readable view of how a video is spoken.
|
||||
|
||||
This is the *source of truth* between analysis and editing: it merges what
|
||||
was said (transcript), who said it (diarization), and how it was said
|
||||
(pitch/energy/rate/pauses → emphasis) into one JSON document, decoupling the
|
||||
audio analysis from FCPXML generation entirely.
|
||||
|
||||
The shape is designed to be handed to a language model so it can reason about
|
||||
the narrative — which beats carry weight, where a speaker changes, where the
|
||||
delivery peaks — and decide how to direct the edit. Two design choices serve
|
||||
that goal:
|
||||
|
||||
* **Layered, not flat.** A ``summary`` gives the whole picture in a few
|
||||
numbers, ``segments`` group words into utterances with their own
|
||||
aggregates, and ``words`` hold the fine detail. A model can reason from
|
||||
the top layer and only descend where it matters, instead of parsing
|
||||
thousands of word rows to find the shape of the piece.
|
||||
* **Normalized, self-describing values.** Every acoustic value is 0–1 and
|
||||
relative to *this* recording (a quiet podcast and a shouted ad both use
|
||||
the full range), and ``scales`` documents that contract inline, so the
|
||||
numbers are interpretable without external context.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Callable, List, Optional, Sequence, Tuple
|
||||
|
||||
from .diarize import DEFAULT_SPEAKER, assign_speakers, build_speakers, diarize
|
||||
from .emphasis import EmphasisWeights, annotate_emphasis
|
||||
from .voice_features import (
|
||||
compute_pauses,
|
||||
compute_speech_rate,
|
||||
extract_energy,
|
||||
extract_pitch,
|
||||
word_pitch_energy,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
VOICE_TIMELINE_VERSION = "1.0"
|
||||
|
||||
# Silence long enough to mean the take stopped rather than the speaker paused.
|
||||
# On real footage, boundaries between retakes showed gaps of 3.6-19.8s while
|
||||
# dramatic beats inside a delivered line stayed under ~2s.
|
||||
TAKE_BOUNDARY_GAP = 3.0
|
||||
|
||||
# How to read the values in this document, split by the level they live on.
|
||||
# Embedded in the output so a model consuming the JSON needs no external
|
||||
# documentation — and kept honest: a metric listed under "word" must exist on
|
||||
# every word row, and one under "segment" on every segment row.
|
||||
VALUE_SCALES = {
|
||||
"word": {
|
||||
"energy": "0-1, loudness relative to the loudest moment of this recording",
|
||||
"pitch_delta": "0-1, how far this word's pitch sits from the speaker's average",
|
||||
"rate_delta": "0-1, how much the local speaking rate departs from the average",
|
||||
"pause_before": "seconds of silence immediately before the word",
|
||||
"emphasis": "0-1 combined index; high values are punch-in/highlight candidates",
|
||||
},
|
||||
"segment": {
|
||||
"gap_before": "seconds of silence before this line",
|
||||
"take_boundary": "true when the gap is long enough that the take likely restarted here",
|
||||
"avg_energy": "0-1 mean loudness across the line",
|
||||
"peak_emphasis": "0-1 highest emphasis of any word in the line",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _normalize(value: Optional[float], maximum: float) -> float:
|
||||
"""Scale ``value`` into 0-1 against ``maximum`` (0.0 when unavailable)."""
|
||||
if value is None or maximum <= 0:
|
||||
return 0.0
|
||||
return max(0.0, min(1.0, value / maximum))
|
||||
|
||||
|
||||
def _round_word(word: dict) -> dict:
|
||||
"""One word row, rounded to a size a model can read without noise.
|
||||
|
||||
The raw ``energy_raw``/``pitch_hz`` ride along beside the normalized
|
||||
values so the document can be re-analyzed over a subset later. That
|
||||
matters after cutting: every normalized value is relative to the
|
||||
loudest moment of the *whole* recording, and if that moment gets cut
|
||||
the survivors are scored against something that no longer exists.
|
||||
"""
|
||||
return {
|
||||
"text": word.get("word", ""),
|
||||
"start": round(float(word.get("start", 0.0)), 3),
|
||||
"end": round(float(word.get("end", 0.0)), 3),
|
||||
"speaker": word.get("speaker_id", DEFAULT_SPEAKER),
|
||||
"energy": round(word.get("energy_norm", 0.0), 3),
|
||||
"pitch_delta": round(word.get("pitch_delta", 0.0), 3),
|
||||
"rate_delta": round(word.get("rate_delta", 0.0), 3),
|
||||
"pause_before": round(word.get("pause_before", 0.0), 3),
|
||||
"emphasis": round(word.get("emphasis", 0.0), 3),
|
||||
"energy_raw": word.get("energy"),
|
||||
"pitch_hz": word.get("pitch_hz"),
|
||||
}
|
||||
|
||||
|
||||
def enrich_words(
|
||||
words: Sequence[dict],
|
||||
pitch_track: Optional[Sequence] = None,
|
||||
energy_track: Optional[Sequence] = None,
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
already_measured: bool = False,
|
||||
) -> List[dict]:
|
||||
"""Attach normalized acoustic features + the emphasis index to each word.
|
||||
|
||||
Normalization is per-recording: energy against the loudest word, pitch
|
||||
against the spread around this recording's average, rate against the
|
||||
largest local departure. That makes the numbers comparable within a
|
||||
piece regardless of how it was recorded.
|
||||
|
||||
Set ``already_measured`` when the words already carry ``energy`` and
|
||||
``pitch_hz`` from a previous pass — re-analyzing a subset, say. The
|
||||
frame tracks are then unnecessary, and sampling them again would
|
||||
overwrite good values with ``None``.
|
||||
"""
|
||||
if not words:
|
||||
return []
|
||||
|
||||
if not already_measured:
|
||||
words = word_pitch_energy(words, pitch_track, energy_track)
|
||||
rates = compute_speech_rate(words)
|
||||
pauses = compute_pauses(words)
|
||||
|
||||
energies = [w["energy"] for w in words if w.get("energy") is not None]
|
||||
max_energy = max(energies) if energies else 0.0
|
||||
pitches = [w["pitch_hz"] for w in words if w.get("pitch_hz") is not None]
|
||||
avg_pitch = sum(pitches) / len(pitches) if pitches else 0.0
|
||||
pitch_span = (max(pitches) - min(pitches)) if len(pitches) > 1 else 0.0
|
||||
avg_rate = sum(rates) / len(rates) if rates else 0.0
|
||||
max_rate = max(rates) if rates else 0.0
|
||||
|
||||
enriched: List[dict] = []
|
||||
for i, w in enumerate(words):
|
||||
ww = dict(w)
|
||||
ww["energy_norm"] = _normalize(w.get("energy"), max_energy)
|
||||
pitch = w.get("pitch_hz")
|
||||
ww["pitch_delta"] = (
|
||||
_normalize(abs(pitch - avg_pitch), pitch_span) if pitch is not None else 0.0
|
||||
)
|
||||
ww["rate_delta"] = _normalize(abs(rates[i] - avg_rate), max_rate)
|
||||
ww["pause_before"] = pauses[i]
|
||||
enriched.append(ww)
|
||||
|
||||
annotated = annotate_emphasis(
|
||||
[{**w, "energy": w["energy_norm"]} for w in enriched], weights=weights
|
||||
)
|
||||
for word, scored in zip(enriched, annotated):
|
||||
word["emphasis"] = scored["emphasis"]
|
||||
return enriched
|
||||
|
||||
|
||||
def _segment_rows(segments: Sequence[dict], words: Sequence[dict]) -> List[dict]:
|
||||
"""Group enriched words under their segment, with per-segment aggregates.
|
||||
|
||||
The aggregates are what let a model judge a whole utterance ("this line
|
||||
is delivered hot, that one trails off") without reading every word.
|
||||
"""
|
||||
rows: List[dict] = []
|
||||
previous_end = 0.0
|
||||
for seg in segments:
|
||||
start = float(seg.get("start", 0.0))
|
||||
end = float(seg.get("end", 0.0))
|
||||
in_seg = [w for w in words if start <= float(w.get("start", 0.0)) < end]
|
||||
energies = [w["energy_norm"] for w in in_seg]
|
||||
emphases = [w["emphasis"] for w in in_seg]
|
||||
gap = max(0.0, start - previous_end)
|
||||
rows.append(
|
||||
{
|
||||
"start": round(start, 3),
|
||||
"end": round(end, 3),
|
||||
"speaker": seg.get("speaker_id", DEFAULT_SPEAKER),
|
||||
"text": (seg.get("text") or "").strip(),
|
||||
# Silence before this line. Long gaps are where the camera
|
||||
# stopped or the take restarted, so this is the structural
|
||||
# hint for splitting a recording into takes — the same signal
|
||||
# that is *noise* for emphasis (see emphasis.pause_weight).
|
||||
"gap_before": round(gap, 3),
|
||||
"take_boundary": gap >= TAKE_BOUNDARY_GAP,
|
||||
"avg_energy": round(sum(energies) / len(energies), 3) if energies else 0.0,
|
||||
"peak_emphasis": round(max(emphases), 3) if emphases else 0.0,
|
||||
"words": [_round_word(w) for w in in_seg],
|
||||
}
|
||||
)
|
||||
previous_end = end
|
||||
return rows
|
||||
|
||||
|
||||
# Words too common to ever be the point of a punch-in. A zoom lands on what a
|
||||
# sentence is *about*, and an article spoken loudly is still an article.
|
||||
_FUNCTION_WORDS = {
|
||||
"a", "o", "e", "de", "da", "do", "que", "é", "em", "um", "uma", "as", "os",
|
||||
"no", "na", "com", "pra", "para", "por", "se", "mais", "isso", "aí", "tudo",
|
||||
"ao", "à", "dos", "das", "nos", "nas", "ou", "mas", "já", "ele", "ela",
|
||||
"eu", "você", "seu", "sua", "meu", "minha", "esse", "essa", "aquele",
|
||||
}
|
||||
|
||||
|
||||
def _survives(start: float, end: float, cuts: Sequence[Tuple[float, float]]) -> bool:
|
||||
"""Whether a span lies entirely outside every removed range."""
|
||||
return all(end <= cut_start or start >= cut_end for cut_start, cut_end in cuts)
|
||||
|
||||
|
||||
def restrict_to_kept(
|
||||
timeline: dict,
|
||||
cut_ranges: Sequence[Tuple[float, float]],
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
peak_percentile: float = 0.02,
|
||||
emphasis_floor: float = 0.25,
|
||||
) -> dict:
|
||||
"""Re-analyze a timeline over only the material that survives ``cut_ranges``.
|
||||
|
||||
Emphasis is *relative*: energy is scored against the loudest word,
|
||||
pitch against the spread of the recording. Cut the loudest moment out —
|
||||
a laugh, an aside to the crew — and every remaining score is measured
|
||||
against something the viewer will never see. Re-running the
|
||||
normalization over just the survivors is what makes "the most emphatic
|
||||
line of the final video" a meaningful question.
|
||||
|
||||
Returns a timeline of the same shape, with times still in original
|
||||
source seconds so the result can be fed straight back as actions.
|
||||
"""
|
||||
kept_words = [
|
||||
w
|
||||
for segment in timeline.get("segments", [])
|
||||
for w in segment.get("words", [])
|
||||
if _survives(w["start"], w["end"], cut_ranges)
|
||||
]
|
||||
# enrich_words expects the raw analysis keys, not the normalized ones.
|
||||
raw = [
|
||||
{
|
||||
"word": w["text"],
|
||||
"start": w["start"],
|
||||
"end": w["end"],
|
||||
"speaker_id": w.get("speaker", DEFAULT_SPEAKER),
|
||||
"energy": w.get("energy_raw"),
|
||||
"pitch_hz": w.get("pitch_hz"),
|
||||
}
|
||||
for w in kept_words
|
||||
]
|
||||
enriched = enrich_words(raw, weights=weights, already_measured=True)
|
||||
|
||||
kept_segments = [
|
||||
{**s, "words": [w for w in s.get("words", []) if _survives(w["start"], w["end"], cut_ranges)]}
|
||||
for s in timeline.get("segments", [])
|
||||
]
|
||||
kept_segments = [s for s in kept_segments if s["words"]]
|
||||
rows = _segment_rows(
|
||||
[{"text": s["text"], "start": s["start"], "end": s["end"],
|
||||
"speaker_id": s.get("speaker", DEFAULT_SPEAKER)} for s in kept_segments],
|
||||
enriched,
|
||||
)
|
||||
duration = sum(s["end"] - s["start"] for s in rows)
|
||||
return {
|
||||
**timeline,
|
||||
"summary": _summary(enriched, rows, timeline.get("speakers", []),
|
||||
duration, peak_percentile, emphasis_floor),
|
||||
"segments": rows,
|
||||
}
|
||||
|
||||
|
||||
def sentence_end(segments: Sequence[dict], index: int) -> float:
|
||||
"""Where the sentence starting at ``segments[index]`` actually finishes.
|
||||
|
||||
Transcription segments break on breath and timing, not on grammar — a
|
||||
sentence routinely spans two or three of them ("…que dá aquele ar" /
|
||||
"de elegância, isso é desejo de muitas mulheres, né?"). A zoom that
|
||||
ends on a segment boundary would therefore release mid-thought, so the
|
||||
window is extended until a segment closes with terminal punctuation.
|
||||
"""
|
||||
last = float(segments[index]["end"])
|
||||
for offset, segment in enumerate(segments[index:]):
|
||||
# A long gap means the take stopped; never run a zoom across that.
|
||||
# Checked before adopting the end, or the boundary segment's own
|
||||
# end would already have been taken.
|
||||
if offset > 0 and segment.get("take_boundary"):
|
||||
break
|
||||
last = float(segment["end"])
|
||||
if (segment.get("text") or "").strip().endswith((".", "!", "?", "…")):
|
||||
break
|
||||
return last
|
||||
|
||||
|
||||
def suggest_zoom_windows(
|
||||
timeline: dict,
|
||||
min_gap: float = 8.0,
|
||||
max_zooms: Optional[int] = None,
|
||||
) -> List[dict]:
|
||||
"""Propose punch-in windows over a timeline's strongest lines.
|
||||
|
||||
One zoom per line at most, taken from the line's most emphatic
|
||||
*content* word — a loudly spoken "a" is still an article, so function
|
||||
words are skipped. The window runs from that word to the end of its
|
||||
line, which is the shape the edit wants: the move lands with the word
|
||||
and holds through the rest of the phrase.
|
||||
|
||||
``min_gap`` keeps successive zooms apart; effects stacked close
|
||||
together read as nervous editing rather than emphasis.
|
||||
"""
|
||||
segments = timeline.get("segments", [])
|
||||
candidates: List[dict] = []
|
||||
for i, segment in enumerate(segments):
|
||||
content = [
|
||||
w for w in segment.get("words", [])
|
||||
if w["text"].strip(",.!?;:").lower() not in _FUNCTION_WORDS
|
||||
]
|
||||
if not content:
|
||||
continue
|
||||
best = max(content, key=lambda w: w["emphasis"])
|
||||
candidates.append({
|
||||
"start": best["start"],
|
||||
# Hold through to the end of the sentence, not of the segment —
|
||||
# releasing mid-thought is what makes a punch-in feel arbitrary.
|
||||
"end": sentence_end(segments, i),
|
||||
"word": best["text"],
|
||||
"emphasis": best["emphasis"],
|
||||
"line": segment["text"],
|
||||
})
|
||||
|
||||
chosen: List[dict] = []
|
||||
for candidate in sorted(candidates, key=lambda c: c["emphasis"], reverse=True):
|
||||
if max_zooms is not None and len(chosen) >= max_zooms:
|
||||
break
|
||||
if any(abs(candidate["start"] - c["start"]) < min_gap for c in chosen):
|
||||
continue
|
||||
chosen.append(candidate)
|
||||
return sorted(chosen, key=lambda c: c["start"])
|
||||
|
||||
|
||||
def speaker_profiles(segments: Sequence[dict], duration: float) -> List[dict]:
|
||||
"""Per-speaker statistics and sample lines, so a person can tell who is who.
|
||||
|
||||
A bare ``SPEAKER_00`` label is useless for deciding whose audio to cut.
|
||||
What identifies a role is *how* someone participates: an interviewer or
|
||||
a crew member asks short questions and holds little of the runtime,
|
||||
while the subject speaks in long stretches. ``avg_segment`` and
|
||||
``share`` capture exactly that contrast, and the sample lines confirm
|
||||
it in the person's own words.
|
||||
"""
|
||||
by_speaker: dict = {}
|
||||
for seg in segments:
|
||||
sid = seg.get("speaker", seg.get("speaker_id", DEFAULT_SPEAKER))
|
||||
length = max(0.0, float(seg.get("end", 0.0)) - float(seg.get("start", 0.0)))
|
||||
entry = by_speaker.setdefault(sid, {"seconds": 0.0, "segments": [], "words": 0})
|
||||
entry["seconds"] += length
|
||||
entry["words"] += len(seg.get("words", []))
|
||||
entry["segments"].append(seg)
|
||||
|
||||
profiles: List[dict] = []
|
||||
for i, (sid, entry) in enumerate(
|
||||
sorted(by_speaker.items(), key=lambda kv: kv[1]["seconds"], reverse=True)
|
||||
):
|
||||
count = len(entry["segments"])
|
||||
# Longest lines identify a role far better than the first ones: a
|
||||
# question and an answer look alike at the start of a recording.
|
||||
longest = sorted(
|
||||
entry["segments"],
|
||||
key=lambda s: float(s.get("end", 0)) - float(s.get("start", 0)),
|
||||
reverse=True,
|
||||
)[:3]
|
||||
profiles.append({
|
||||
"id": sid,
|
||||
"name": f"Speaker {i + 1}",
|
||||
"speaking_seconds": round(entry["seconds"], 2),
|
||||
"share": round(entry["seconds"] / duration, 3) if duration > 0 else 0.0,
|
||||
"segment_count": count,
|
||||
"avg_segment": round(entry["seconds"] / count, 2) if count else 0.0,
|
||||
"word_count": entry["words"],
|
||||
"samples": [(s.get("text") or "").strip()[:160] for s in longest],
|
||||
})
|
||||
return profiles
|
||||
|
||||
|
||||
def select_peaks(
|
||||
words: Sequence[dict], percentile: float, floor: float
|
||||
) -> List[dict]:
|
||||
"""The most emphatic words: the top ``percentile`` fraction, above ``floor``.
|
||||
|
||||
Selection is relative on purpose. The emphasis index is a weighted
|
||||
average whose real range depends entirely on the material — a measured
|
||||
interview peaks around 0.5 while an energetic ad reaches much higher —
|
||||
so any fixed cutoff either floods one and selects nothing in the other.
|
||||
Asking for "the top 2%" instead yields a usable handful either way.
|
||||
|
||||
``floor`` is only a sanity guard for genuinely flat audio, where even
|
||||
the top of the distribution carries no emphasis worth cutting on.
|
||||
"""
|
||||
ranked = sorted(words, key=lambda w: w["emphasis"], reverse=True)
|
||||
keep = max(1, round(len(ranked) * percentile)) if ranked else 0
|
||||
return [w for w in ranked[:keep] if w["emphasis"] >= floor]
|
||||
|
||||
|
||||
def _summary(words: Sequence[dict], segments: Sequence[dict], speakers: Sequence[dict],
|
||||
duration: float, peak_percentile: float, emphasis_floor: float) -> dict:
|
||||
"""The top layer: the shape of the piece in a handful of numbers."""
|
||||
emphases = [w["emphasis"] for w in words]
|
||||
peaks = select_peaks(words, peak_percentile, emphasis_floor)
|
||||
return {
|
||||
"duration": round(duration, 3),
|
||||
"speaker_count": len(speakers),
|
||||
"segment_count": len(segments),
|
||||
"word_count": len(words),
|
||||
"avg_emphasis": round(sum(emphases) / len(emphases), 3) if emphases else 0.0,
|
||||
"peak_selection": f"top {peak_percentile:.0%} of words, minimum emphasis {emphasis_floor:.2f}",
|
||||
"peak_count": len(peaks),
|
||||
"peak_moments": [
|
||||
{
|
||||
"time": round(float(w.get("start", 0.0)), 3),
|
||||
"text": w.get("word", ""),
|
||||
"speaker": w.get("speaker_id", DEFAULT_SPEAKER),
|
||||
"emphasis": round(w["emphasis"], 3),
|
||||
}
|
||||
for w in sorted(peaks, key=lambda w: w["emphasis"], reverse=True)[:20]
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def build_voice_timeline(
|
||||
media_path: str,
|
||||
transcript: dict,
|
||||
hf_token: Optional[str] = None,
|
||||
num_speakers: str = "",
|
||||
weights: EmphasisWeights = EmphasisWeights(),
|
||||
peak_percentile: float = 0.02,
|
||||
emphasis_floor: float = 0.25,
|
||||
progress_cb: Optional[Callable[[float, str], None]] = None,
|
||||
) -> dict:
|
||||
"""Build the consolidated voice timeline for one media file.
|
||||
|
||||
Every analysis layer is optional and degrades independently: without
|
||||
librosa the acoustic values are ``0.0``; without a diarization token
|
||||
every word belongs to ``SPEAKER_00``. The document's shape never
|
||||
changes, so downstream consumers (the rules engine, or a model reading
|
||||
the JSON) can rely on it.
|
||||
"""
|
||||
def report(fraction: float, stage: str) -> None:
|
||||
if progress_cb:
|
||||
progress_cb(fraction, stage)
|
||||
|
||||
report(0.1, "Analisando tom e energia...")
|
||||
pitch_track = extract_pitch(media_path)
|
||||
energy_track = extract_energy(media_path)
|
||||
|
||||
report(0.5, "Calculando ênfase...")
|
||||
words = enrich_words(transcript.get("words", []), pitch_track, energy_track, weights)
|
||||
|
||||
report(0.7, "Identificando participantes...")
|
||||
tracks = diarize(media_path, hf_token, num_speakers) if hf_token else None
|
||||
segments, words = assign_speakers(transcript.get("segments", []), words, tracks)
|
||||
speakers = build_speakers(segments)
|
||||
|
||||
report(0.9, "Montando linha do tempo...")
|
||||
duration = float(transcript.get("duration", 0.0))
|
||||
segment_rows = _segment_rows(segments, words)
|
||||
return {
|
||||
"version": VOICE_TIMELINE_VERSION,
|
||||
"source": Path(media_path).name,
|
||||
"language": transcript.get("language", ""),
|
||||
# What actually ran, not what was installed — a consumer must be able
|
||||
# to tell "this speech is flat" from "the acoustics never loaded",
|
||||
# since both leave the same zeros in the data.
|
||||
"layers": {
|
||||
"transcript": bool(transcript.get("words")),
|
||||
"acoustics": pitch_track is not None or energy_track is not None,
|
||||
"speakers": tracks is not None,
|
||||
},
|
||||
"scales": VALUE_SCALES,
|
||||
"summary": _summary(
|
||||
words, segments, speakers, duration, peak_percentile, emphasis_floor
|
||||
),
|
||||
"speakers": speaker_profiles(segment_rows, duration),
|
||||
"segments": segment_rows,
|
||||
}
|
||||
|
||||
|
||||
def voice_timeline_path(media_path: str, output_dir: Optional[str] = None) -> Path:
|
||||
"""Where the ``_voice_timeline.json`` for ``media_path`` lives.
|
||||
|
||||
Mirrors ``_transcript.json``: next to the media, or in the chosen
|
||||
project folder when one is set.
|
||||
"""
|
||||
p = Path(media_path)
|
||||
if output_dir:
|
||||
directory = Path(output_dir).expanduser()
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
return directory / f"{p.stem}_voice_timeline.json"
|
||||
return p.with_name(p.stem + "_voice_timeline.json")
|
||||
|
||||
|
||||
def save_voice_timeline(timeline: dict, path: Path) -> None:
|
||||
"""Write the timeline as UTF-8 JSON (accented transcripts stay readable)."""
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
json.dump(timeline, f, ensure_ascii=False, indent=2)
|
||||
|
||||
|
||||
def load_voice_timeline(path: Path) -> Optional[dict]:
|
||||
"""Read a cached voice timeline, or ``None`` when absent/unreadable."""
|
||||
try:
|
||||
with open(path, encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
except (OSError, json.JSONDecodeError, UnicodeDecodeError):
|
||||
return None
|
||||
return data if isinstance(data, dict) and "segments" in data else None
|
||||
+355
-39
@@ -35,9 +35,11 @@ import unicodedata
|
||||
import uuid
|
||||
import xml.etree.ElementTree as ET
|
||||
from datetime import datetime
|
||||
from fractions import Fraction
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from .collision import blocking, validate_titles
|
||||
from .models import (
|
||||
_FCPXML_STANDARD_TIMEBASES,
|
||||
DynamicSubtitleConfig,
|
||||
@@ -50,7 +52,12 @@ from .models import (
|
||||
ValidationIssue,
|
||||
ValidationIssueType,
|
||||
)
|
||||
from .text_layout import LayoutBox, compose_sentence, layout_sentence
|
||||
from .text_layout import (
|
||||
TEXT_TEMPLATE_FONT_SCALE,
|
||||
LayoutBox,
|
||||
compose_sentence,
|
||||
layout_sentence,
|
||||
)
|
||||
from .transcribe import group_words_by_segment
|
||||
|
||||
# Maximum lengths for XML attribute values to prevent memory abuse
|
||||
@@ -154,6 +161,23 @@ _ASSET_CLIP_CHILD_ORDER = [
|
||||
_CHILD_ORDER_INDEX = {tag: i for i, tag in enumerate(_ASSET_CLIP_CHILD_ORDER)}
|
||||
|
||||
|
||||
# How close to the end of a clip a zoom must finish for the return to be
|
||||
# skipped. Within this margin the cut arrives before the eye registers the
|
||||
# move back, so the return reads as a twitch rather than a resolution.
|
||||
HOLD_AT_CUT_THRESHOLD = 1.0
|
||||
|
||||
# How close to the start of a clip a zoom must begin for the ramp-in to be
|
||||
# skipped and the shot to simply open already zoomed. Tighter than the end
|
||||
# margin on purpose: at the end the cut hides an unfinished return, but at
|
||||
# the start a ramp is visible from frame one and reads as the shot settling.
|
||||
START_AT_CUT_THRESHOLD = 0.5
|
||||
|
||||
|
||||
def _fmt_scale(value: float) -> str:
|
||||
"""Format a scale factor without trailing float noise (1.0 -> "1")."""
|
||||
return f"{value:.6f}".rstrip("0").rstrip(".") or "0"
|
||||
|
||||
|
||||
def _dtd_insert(parent: ET.Element, child: ET.Element) -> ET.Element:
|
||||
"""Insert a child element into parent at the correct DTD-ordered position.
|
||||
|
||||
@@ -541,10 +565,43 @@ def _check_timebases(root: ET.Element) -> List[ValidationIssue]:
|
||||
return issues
|
||||
|
||||
|
||||
def _document_frame_duration(root: ET.Element) -> Optional[Fraction]:
|
||||
"""The sequence's exact ``frameDuration`` as a fraction, if declared.
|
||||
|
||||
Read from the format the ``<sequence>`` references (falling back to the
|
||||
first declared format), so the value is the document's own timebase
|
||||
rather than an assumed rate.
|
||||
"""
|
||||
formats = {f.get('id'): f for f in root.findall('.//format') if f.get('id')}
|
||||
sequence = root.find('.//sequence')
|
||||
fmt = formats.get(sequence.get('format')) if sequence is not None else None
|
||||
if fmt is None:
|
||||
fmt = next(iter(formats.values()), None)
|
||||
if fmt is None:
|
||||
return None
|
||||
raw = fmt.get('frameDuration', '')
|
||||
if not (raw.endswith('s') and '/' in raw):
|
||||
return None
|
||||
numerator, denominator = raw[:-1].split('/', 1)
|
||||
try:
|
||||
value = Fraction(int(numerator), int(denominator))
|
||||
except (ValueError, ZeroDivisionError):
|
||||
return None
|
||||
return value if value > 0 else None
|
||||
|
||||
|
||||
def _check_frame_alignment(root: ET.Element, fps: float = 24.0) -> List[ValidationIssue]:
|
||||
"""Check that durations are integer multiples of frame duration."""
|
||||
"""Check that durations are integer multiples of the frame duration.
|
||||
|
||||
Uses the document's exact ``frameDuration`` fraction and rational
|
||||
arithmetic. Comparing against an integer fps instead would flag every
|
||||
NTSC project as broken: at 1001/24000s (23.976fps) a perfectly aligned
|
||||
duration is not an integer number of "24fps" frames, so whole timelines
|
||||
would be reported misaligned when nothing is wrong.
|
||||
"""
|
||||
issues = []
|
||||
fps_int = int(fps)
|
||||
frame_duration = _document_frame_duration(root)
|
||||
label = f"{1 / float(frame_duration):.3f}".rstrip('0').rstrip('.') if frame_duration else str(fps)
|
||||
for elem in root.iter():
|
||||
dur_str = elem.get('duration')
|
||||
if not dur_str or not dur_str.endswith('s'):
|
||||
@@ -553,14 +610,19 @@ def _check_frame_alignment(root: ET.Element, fps: float = 24.0) -> List[Validati
|
||||
continue
|
||||
try:
|
||||
tv = TimeValue.from_timecode(dur_str)
|
||||
frames = tv.to_seconds() * fps_int
|
||||
if abs(frames - round(frames)) > 0.01:
|
||||
if frame_duration is not None:
|
||||
frames = Fraction(tv.numerator, tv.denominator) / frame_duration
|
||||
aligned = frames.denominator == 1
|
||||
else:
|
||||
approx = tv.to_seconds() * fps
|
||||
aligned = abs(approx - round(approx)) <= 0.01
|
||||
if not aligned:
|
||||
issues.append(ValidationIssue(
|
||||
issue_type=ValidationIssueType.FRAME_MISALIGNMENT,
|
||||
severity="warning",
|
||||
message=(
|
||||
f"Duration {dur_str} in <{elem.tag}> "
|
||||
f"'{elem.get('name', '')}' is not frame-aligned at {fps_int}fps."
|
||||
f"'{elem.get('name', '')}' is not frame-aligned at {label}fps."
|
||||
),
|
||||
clip_name=elem.get('name'),
|
||||
))
|
||||
@@ -1030,13 +1092,21 @@ class FCPXMLModifier:
|
||||
elem.set(attr, val)
|
||||
return elem
|
||||
|
||||
def _require_clip(self, clip_id: str) -> ET.Element:
|
||||
def _require_clip(self, clip_id: 'str | ET.Element') -> ET.Element:
|
||||
"""Look up a clip by ID/name, raising if not found.
|
||||
|
||||
Centralises the get-or-raise pattern used by every clip-mutating
|
||||
method so the error message stays consistent and future
|
||||
enhancements (fuzzy matching, suggestions) only need one site.
|
||||
|
||||
An Element is returned as-is. That matters after ``split_clip`` or
|
||||
``cut_clip_ranges``: the resulting pieces all carry the *same* name,
|
||||
so a name lookup would always resolve to the first one and silently
|
||||
put the edit on the wrong piece. Callers holding the exact element
|
||||
pass it directly.
|
||||
"""
|
||||
if isinstance(clip_id, ET.Element):
|
||||
return clip_id
|
||||
clip = self.clips.get(clip_id)
|
||||
if clip is None:
|
||||
raise ValueError(f"Clip not found: {clip_id}")
|
||||
@@ -1421,7 +1491,7 @@ class FCPXMLModifier:
|
||||
|
||||
def add_marker(
|
||||
self,
|
||||
clip_id: str,
|
||||
clip_id: 'str | ET.Element',
|
||||
timecode: str,
|
||||
name: str,
|
||||
marker_type: "MarkerType | str" = MarkerType.STANDARD,
|
||||
@@ -1937,35 +2007,50 @@ class FCPXMLModifier:
|
||||
|
||||
def add_zoom(
|
||||
self,
|
||||
clip_id: str,
|
||||
clip_id: 'str | ET.Element',
|
||||
start: float,
|
||||
end: float,
|
||||
scale: float = 1.3,
|
||||
ease: float = 0.3,
|
||||
ease: float = 0.25,
|
||||
position: str = "0 0",
|
||||
ease_out: Optional[float] = None,
|
||||
hold_at_end: Optional[bool] = None,
|
||||
start_at_peak: Optional[bool] = None,
|
||||
) -> ET.Element:
|
||||
"""Add a smooth ease-in/ease-out punch-in zoom to a clip.
|
||||
"""Add a punch-in zoom to a clip, snapping back to its framing at the end.
|
||||
|
||||
Animates ``<adjust-transform>``'s ``scale`` param (per the FCPXML
|
||||
DTD: ``<param>`` + ``<keyframeAnimation>`` of ``<keyframe>``
|
||||
elements, ``interp="ease"``) from 100% up to *scale* and back down
|
||||
to 100%, entirely within ``[start, end]`` — clip-relative seconds
|
||||
(seconds from the clip's own head, same convention as
|
||||
``cut_clip_ranges``). The ease portions each last *ease* seconds;
|
||||
the zoom holds at *scale* in between.
|
||||
Animates ``<adjust-transform>``'s ``scale`` param (``<param>`` +
|
||||
``<keyframeAnimation>`` of ``<keyframe>``) from the clip's current
|
||||
scale up to *scale* times it, holds, then returns — all within
|
||||
``[start, end]`` — clip-relative seconds (same convention as
|
||||
``cut_clip_ranges``).
|
||||
|
||||
The two ends are deliberately asymmetric. *ease* ramps the zoom
|
||||
**in** over half a second by default, fast enough to land with the
|
||||
emphasised word. The way **out** is instant — a single frame — so
|
||||
the moment the impact phrase ends the shot is simply back to its
|
||||
normal framing and the video resumes its flow, with no drift
|
||||
drawing attention to itself. Pass *ease_out* to ramp the return
|
||||
gradually instead.
|
||||
|
||||
*hold_at_end* keeps the peak instead of returning, and
|
||||
*start_at_peak* opens already zoomed with no ramp. Left as ``None``
|
||||
both decide on their own from how close the window sits to the
|
||||
clip's edges: a cut is itself the transition, so ramping away from
|
||||
one — or back toward one — is motion the viewer reads as a wobble
|
||||
rather than as emphasis.
|
||||
"""
|
||||
if end <= start:
|
||||
raise ValueError(f"end ({end}) must be greater than start ({start})")
|
||||
if ease <= 0:
|
||||
raise ValueError(f"ease must be positive, got {ease}")
|
||||
if ease * 2 > (end - start):
|
||||
raise ValueError(
|
||||
f"ease ({ease}s x2 = {ease * 2}s) doesn't fit in the zoom "
|
||||
f"window ({end - start}s) — shorten ease or widen start/end"
|
||||
)
|
||||
if scale <= 0:
|
||||
raise ValueError(f"scale must be positive, got {scale}")
|
||||
|
||||
frame = float(self.frame_duration_fraction())
|
||||
ramp_out = frame if ease_out is None else ease_out
|
||||
if ramp_out <= 0:
|
||||
raise ValueError(f"ease_out must be positive, got {ease_out}")
|
||||
clip = self._require_clip(clip_id)
|
||||
clip_duration = self._parse_time(clip.get('duration', '0s')).to_seconds()
|
||||
if start < 0 or end > clip_duration:
|
||||
@@ -1974,26 +2059,146 @@ class FCPXMLModifier:
|
||||
f"duration (0 to {clip_duration:.3f}s)"
|
||||
)
|
||||
|
||||
# Replace rather than stack a prior zoom on the same clip.
|
||||
# Replace a prior zoom, but never the clip's framing. A clip can
|
||||
# already carry an <adjust-transform> holding the editor's own
|
||||
# reframe — rotation for footage shot sideways, position, a scale
|
||||
# that makes the shot work at all. Dropping it outright (the old
|
||||
# behaviour) silently destroyed that framing; on real footage the
|
||||
# zoomed section came back rotated. So: keep the static attributes,
|
||||
# and animate *relative to* the existing scale.
|
||||
base_x, base_y = 1.0, 1.0
|
||||
carried: dict = {}
|
||||
old_keyframes: list = []
|
||||
for stale in clip.findall('adjust-transform'):
|
||||
carried = {k: v for k, v in stale.attrib.items() if k != 'scale'}
|
||||
parts = (stale.get('scale') or '').split()
|
||||
if len(parts) == 2:
|
||||
try:
|
||||
base_x, base_y = float(parts[0]), float(parts[1])
|
||||
except ValueError:
|
||||
base_x, base_y = 1.0, 1.0
|
||||
else:
|
||||
# No static attribute — a PRIOR zoom on this same clip left
|
||||
# an animated <param name="scale"> instead, and the true
|
||||
# resting framing lives in its keyframes, not in 1.0.
|
||||
# Reading it as 1.0 here doesn't just miss the framing: it
|
||||
# replaces the earlier zoom's whole animation with a wrong
|
||||
# one, since this loop unconditionally removes `stale`
|
||||
# right after. The rest value is recoverable without
|
||||
# knowing which keyframe it is: MIN_ZOOM_SCALE == 1.0 means
|
||||
# every keyframed value is >= the rest scale, so the
|
||||
# smallest one keyframed is the rest value, peak or not.
|
||||
for old_param in stale.findall("param[@name='scale']"):
|
||||
xs, ys = [], []
|
||||
for kf in old_param.findall('.//keyframe'):
|
||||
kv = (kf.get('value') or '').split()
|
||||
if len(kv) == 2:
|
||||
try:
|
||||
xs.append(float(kv[0]))
|
||||
ys.append(float(kv[1]))
|
||||
except ValueError:
|
||||
pass
|
||||
# Kept for merging: a second zoom on the same clip
|
||||
# (two emphatic beats a cut didn't separate) should
|
||||
# stack alongside the first, not erase it — the
|
||||
# earlier peak is still a real editorial decision.
|
||||
old_keyframes.append((kf.get('time', '0s'), kf.get('value', '')))
|
||||
if xs and ys:
|
||||
base_x, base_y = min(xs), min(ys)
|
||||
clip.remove(stale)
|
||||
|
||||
transform = ET.Element('adjust-transform')
|
||||
for key, value in carried.items():
|
||||
transform.set(key, value)
|
||||
scale_param = ET.SubElement(transform, 'param')
|
||||
scale_param.set('name', 'scale')
|
||||
anim = ET.SubElement(scale_param, 'keyframeAnimation')
|
||||
|
||||
scale_value = f"{scale} {scale}"
|
||||
for seconds, value in (
|
||||
(start, "1 1"),
|
||||
(start + ease, scale_value),
|
||||
(end - ease, scale_value),
|
||||
(end, "1 1"),
|
||||
):
|
||||
# Keyframe times live in the clip's SOURCE timebase — the same origin
|
||||
# as its own ``start`` — not in clip-relative seconds. A clip whose
|
||||
# media starts at, say, 3109.9s of timecode looks for the animation
|
||||
# there; keyframes written at 0-5s land outside the clip entirely and
|
||||
# Final Cut imports the zoom as nothing at all, silently. Matches what
|
||||
# add_text_title already does, and only shows up on footage whose
|
||||
# start isn't 0s — every synthetic fixture starts at 0s and hides it.
|
||||
media_origin = self._parse_time(clip.get('start', '0s'))
|
||||
|
||||
rest_value = f"{_fmt_scale(base_x)} {_fmt_scale(base_y)}"
|
||||
scale_value = f"{_fmt_scale(base_x * scale)} {_fmt_scale(base_y * scale)}"
|
||||
|
||||
# A return that lands right before a cut is wasted motion: the next
|
||||
# clip begins on its own framing anyway, so all the viewer sees is a
|
||||
# twitch on the way out. When the zoom runs to the end of the clip,
|
||||
# hold the peak and let the cut do the resetting.
|
||||
holds_to_cut = (
|
||||
hold_at_end
|
||||
if hold_at_end is not None
|
||||
else (clip_duration - end) <= HOLD_AT_CUT_THRESHOLD
|
||||
)
|
||||
opens_at_peak = (
|
||||
start_at_peak
|
||||
if start_at_peak is not None
|
||||
else start <= START_AT_CUT_THRESHOLD
|
||||
)
|
||||
|
||||
# Only the ramps actually written have to fit in the window: a zoom
|
||||
# that opens at the peak spends no time ramping in, and one held to
|
||||
# the cut spends none ramping out.
|
||||
needed = (0.0 if opens_at_peak else ease) + (0.0 if holds_to_cut else ramp_out)
|
||||
if needed > (end - start):
|
||||
raise ValueError(
|
||||
f"the ramps ({needed}s) don't fit in the zoom window "
|
||||
f"({end - start}s) — shorten them or widen start/end"
|
||||
)
|
||||
|
||||
if opens_at_peak:
|
||||
# The cut already delivered the change of framing; ramping up
|
||||
# from it just looks like the shot settling.
|
||||
keyframes = [(start, scale_value)]
|
||||
else:
|
||||
keyframes = [(start, rest_value), (start + ease, scale_value)]
|
||||
if holds_to_cut:
|
||||
keyframes.append((end, scale_value))
|
||||
else:
|
||||
# Hold the peak right up to the end, then drop back on the very
|
||||
# next frame — the snap-back the edit wants, not a slow drift.
|
||||
keyframes.append((end - ramp_out, scale_value))
|
||||
keyframes.append((end, rest_value))
|
||||
|
||||
new_entries = [
|
||||
((media_origin + self.snap_seconds_to_frame(seconds)), value)
|
||||
for seconds, value in keyframes
|
||||
]
|
||||
new_start_time = new_entries[0][0]
|
||||
new_end_time = new_entries[-1][0]
|
||||
|
||||
# Two calls on the same clip mean two different things depending on
|
||||
# whether their windows overlap. Overlapping = redoing the *same*
|
||||
# zoom with new numbers — the old keyframes are stale and all of
|
||||
# them go. Disjoint = a second, separate beat that a cut didn't
|
||||
# separate onto its own clip — that one stacks alongside the first
|
||||
# instead of erasing it, since both are real editorial decisions.
|
||||
old_times = [self._parse_time(t) for t, _ in old_keyframes]
|
||||
old_span_overlaps_new = bool(old_times) and not (
|
||||
max(old_times) < new_start_time or min(old_times) > new_end_time
|
||||
)
|
||||
if old_span_overlaps_new:
|
||||
surviving_old: list = []
|
||||
else:
|
||||
surviving_old = [(self._parse_time(t), v) for t, v in old_keyframes]
|
||||
all_entries = sorted(surviving_old + new_entries, key=lambda e: e[0])
|
||||
|
||||
for time_value, value in all_entries:
|
||||
kf = ET.SubElement(anim, 'keyframe')
|
||||
kf.set('time', self.snap_seconds_to_frame(seconds).to_fcpxml())
|
||||
kf.set('time', time_value.to_fcpxml())
|
||||
kf.set('value', value)
|
||||
kf.set('interp', 'ease')
|
||||
# Only 'time' and 'value' — no 'interp', no 'curve'. The DTD allows
|
||||
# both, but Final Cut rejected 'interp' on this vector param
|
||||
# ("does not support the interpolation attribute") and discarded
|
||||
# the whole <param>. A hand-made zoom exported from FCP itself
|
||||
# writes bare keyframes and relies on the DTD default
|
||||
# (curve="smooth"), so we match that export exactly rather than
|
||||
# guess which attributes survive its importer.
|
||||
|
||||
if position != "0 0":
|
||||
pos_param = ET.SubElement(transform, 'param')
|
||||
@@ -2707,7 +2912,18 @@ class FCPXMLModifier:
|
||||
_TEXT_POSITION_KEY = '9999/10003/13260/3296672360/1/100/101'
|
||||
# Layout params the "Text" template ships with. These keys are the
|
||||
# template's own defaults and never vary between instances.
|
||||
#
|
||||
# "Build Out" is the one deliberate override: with "Apply Speed" set to
|
||||
# "2 (Per Object)" below, the template's whole built-in animation (build
|
||||
# in + build out) is always compressed to exactly fill the title's own
|
||||
# on-screen duration — so on a short word-length clip, build out was
|
||||
# eating time that build in needed to finish revealing the text before
|
||||
# the cut. Disabling build out hands that entire compressed window to
|
||||
# build in alone, which is what "sempre acelerado" turned out to mean:
|
||||
# no separate speed knob needed. Value captured from a real FCP export
|
||||
# with "Build Out" unchecked in the Inspector (see chat, 2026-08-18).
|
||||
_TEXT_TITLE_PARAMS = (
|
||||
('Build Out', '9999/10000/2/102', '0'),
|
||||
('Layout Method', '9999/10003/13260/3296672360/2/314', '1 (Paragraph)'),
|
||||
('Left Margin', '9999/10003/13260/3296672360/2/323', '-1210'),
|
||||
('Right Margin', '9999/10003/13260/3296672360/2/324', '1210'),
|
||||
@@ -2795,6 +3011,7 @@ class FCPXMLModifier:
|
||||
bold: bool = True,
|
||||
face: Optional[str] = None,
|
||||
kerning: Optional[float] = None,
|
||||
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
|
||||
) -> ET.Element:
|
||||
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
|
||||
|
||||
@@ -2849,13 +3066,31 @@ class FCPXMLModifier:
|
||||
style_def.set('id', ts_id)
|
||||
text_style = ET.SubElement(style_def, 'text-style')
|
||||
text_style.set('font', font)
|
||||
text_style.set('fontSize', str(font_size))
|
||||
# Text.moti sizes type in frame pixels but positions in canvas points.
|
||||
# See TEXT_TEMPLATE_FONT_SCALE: layout measures in points, so only the
|
||||
# emitted size (and its kerning, to keep the same letter spacing) is
|
||||
# converted here.
|
||||
scale = float(font_scale) or 1.0
|
||||
text_style.set('fontSize', f"{float(font_size) * scale:g}")
|
||||
text_style.set('fontColor', font_color)
|
||||
text_style.set('bold', '1' if bold else '0')
|
||||
if face:
|
||||
# FCP represents bold weight as the bold attribute — never as a
|
||||
# fontFace. Writing ``bold="0" fontFace="Bold"`` (the previous
|
||||
# behaviour) is contradictory and FCP refuses to render the text.
|
||||
# Italic, by contrast, IS a face: FCP writes both ``fontFace`` and
|
||||
# ``italic="1"``. See Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-19.
|
||||
face_lower = (face or '').strip().lower()
|
||||
if face_lower == 'bold':
|
||||
text_style.set('bold', '1')
|
||||
elif 'italic' in face_lower:
|
||||
text_style.set('fontFace', face)
|
||||
text_style.set('italic', '1')
|
||||
else:
|
||||
if bold:
|
||||
text_style.set('bold', '1')
|
||||
if face:
|
||||
text_style.set('fontFace', face)
|
||||
if kerning:
|
||||
text_style.set('kerning', f"{float(kerning):g}")
|
||||
text_style.set('kerning', f"{float(kerning) * scale:g}")
|
||||
text_style.set('alignment', 'center')
|
||||
text_style.set('lineSpacing', '-19')
|
||||
|
||||
@@ -3005,7 +3240,9 @@ class FCPXMLModifier:
|
||||
def lay_out(pending: List[Dict]):
|
||||
"""Place what fits; return (units, still-unplaced words)."""
|
||||
if phrase_mode:
|
||||
composition = compose_sentence(pending, config.style, box)
|
||||
composition = compose_sentence(
|
||||
pending, config.style, box, line_gap=config.line_gap,
|
||||
)
|
||||
return composition.blocks, composition.overflow
|
||||
layout = layout_sentence(pending, config.style, box)
|
||||
return layout.placed, layout.overflow
|
||||
@@ -3083,19 +3320,98 @@ class FCPXMLModifier:
|
||||
duration,
|
||||
lane=lane,
|
||||
name=f"caption_{uuid.uuid4().hex[:8]}",
|
||||
position=unit.position_param(),
|
||||
position=unit.position_param(config.text_scale),
|
||||
font=unit.font or config.style.font,
|
||||
font_size=int(round(unit.font_size)),
|
||||
font_color=unit.color or config.style.active_color,
|
||||
bold=config.style.bold,
|
||||
face=unit.face,
|
||||
kerning=unit.kerning,
|
||||
font_scale=config.text_scale,
|
||||
)
|
||||
_dtd_insert(parent, title)
|
||||
created.append(title)
|
||||
|
||||
if getattr(config, 'validate', False):
|
||||
report = self.validate_subtitle_layout()
|
||||
if blocking(report["severity"]):
|
||||
raise ValueError(
|
||||
"Subtitle layout validation failed: "
|
||||
+ str(report["summary"])
|
||||
)
|
||||
|
||||
return created
|
||||
|
||||
def validate_subtitle_layout(
|
||||
self,
|
||||
*,
|
||||
safe_margin_x: float = 0.05,
|
||||
safe_margin_y: float = 0.05,
|
||||
min_font_size: Optional[float] = None,
|
||||
min_distance: Optional[float] = None,
|
||||
max_distance: Optional[float] = None,
|
||||
) -> dict:
|
||||
"""Re-measure every ``<title>`` in the document and report collisions.
|
||||
|
||||
Reconstructs each title's on-screen box from the values the writer
|
||||
emitted (``fontSize``/``kerning``/``Position`` are already in template
|
||||
space), then checks for temporal+spatial collisions, frame/safe-area
|
||||
containment, and font fallbacks. This is the spec-16 validation pass the
|
||||
layout engine does not do on its own — it only guarantees non-overlap
|
||||
*by construction* while composing, and cannot see a hand-edited title.
|
||||
|
||||
Returns the ``collision.validate_titles`` report: ``severity`` (worst
|
||||
bucket), ``issues`` (spec-16 occurrences) and ``summary`` (counts).
|
||||
"""
|
||||
titles = []
|
||||
for elem in self.root.iter('title'):
|
||||
text_el = elem.find('text/text-style')
|
||||
text = (text_el.text or '').strip() if text_el is not None else ''
|
||||
style = elem.find('text-style-def/text-style')
|
||||
font = style.get('font') if style is not None else None
|
||||
face = style.get('fontFace') if style is not None else None
|
||||
font_size = (
|
||||
float(style.get('fontSize', '0')) if style is not None else 0.0
|
||||
)
|
||||
kerning = (
|
||||
float(style.get('kerning', '0') or 0)
|
||||
if style is not None else 0.0
|
||||
)
|
||||
|
||||
x = y = 0.0
|
||||
for param in elem.findall('param'):
|
||||
if param.get('name') == 'Position' and param.get('value'):
|
||||
parts = param.get('value').split()
|
||||
if len(parts) >= 2:
|
||||
x, y = float(parts[0]), float(parts[1])
|
||||
|
||||
start = self._parse_time(elem.get('offset', '0s')).to_seconds()
|
||||
duration = self._parse_time(elem.get('duration', '0s')).to_seconds()
|
||||
|
||||
titles.append({
|
||||
'text': text,
|
||||
'font': font,
|
||||
'face': face,
|
||||
'font_size': font_size,
|
||||
'kerning': kerning,
|
||||
'x': x,
|
||||
'y': y,
|
||||
'start': start,
|
||||
'end': start + duration,
|
||||
'group': start + duration,
|
||||
})
|
||||
|
||||
return validate_titles(
|
||||
titles,
|
||||
self.frame_width(),
|
||||
self.frame_height(),
|
||||
safe_margin_x=safe_margin_x,
|
||||
safe_margin_y=safe_margin_y,
|
||||
min_font_size=min_font_size,
|
||||
min_distance=min_distance,
|
||||
max_distance=max_distance,
|
||||
)
|
||||
|
||||
# ========================================================================
|
||||
# AUDIO CLIP OPERATIONS (v0.6.0)
|
||||
# ========================================================================
|
||||
|
||||
Reference in New Issue
Block a user