chore: atualização geral

This commit is contained in:
João Henrique
2026-08-19 16:35:29 -04:00
parent 8fca456ceb
commit e7748c2c58
66 changed files with 13037 additions and 4237 deletions
+355 -39
View File
@@ -35,9 +35,11 @@ import unicodedata
import uuid
import xml.etree.ElementTree as ET
from datetime import datetime
from fractions import Fraction
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
from .collision import blocking, validate_titles
from .models import (
_FCPXML_STANDARD_TIMEBASES,
DynamicSubtitleConfig,
@@ -50,7 +52,12 @@ from .models import (
ValidationIssue,
ValidationIssueType,
)
from .text_layout import LayoutBox, compose_sentence, layout_sentence
from .text_layout import (
TEXT_TEMPLATE_FONT_SCALE,
LayoutBox,
compose_sentence,
layout_sentence,
)
from .transcribe import group_words_by_segment
# Maximum lengths for XML attribute values to prevent memory abuse
@@ -154,6 +161,23 @@ _ASSET_CLIP_CHILD_ORDER = [
_CHILD_ORDER_INDEX = {tag: i for i, tag in enumerate(_ASSET_CLIP_CHILD_ORDER)}
# How close to the end of a clip a zoom must finish for the return to be
# skipped. Within this margin the cut arrives before the eye registers the
# move back, so the return reads as a twitch rather than a resolution.
HOLD_AT_CUT_THRESHOLD = 1.0
# How close to the start of a clip a zoom must begin for the ramp-in to be
# skipped and the shot to simply open already zoomed. Tighter than the end
# margin on purpose: at the end the cut hides an unfinished return, but at
# the start a ramp is visible from frame one and reads as the shot settling.
START_AT_CUT_THRESHOLD = 0.5
def _fmt_scale(value: float) -> str:
"""Format a scale factor without trailing float noise (1.0 -> "1")."""
return f"{value:.6f}".rstrip("0").rstrip(".") or "0"
def _dtd_insert(parent: ET.Element, child: ET.Element) -> ET.Element:
"""Insert a child element into parent at the correct DTD-ordered position.
@@ -541,10 +565,43 @@ def _check_timebases(root: ET.Element) -> List[ValidationIssue]:
return issues
def _document_frame_duration(root: ET.Element) -> Optional[Fraction]:
"""The sequence's exact ``frameDuration`` as a fraction, if declared.
Read from the format the ``<sequence>`` references (falling back to the
first declared format), so the value is the document's own timebase
rather than an assumed rate.
"""
formats = {f.get('id'): f for f in root.findall('.//format') if f.get('id')}
sequence = root.find('.//sequence')
fmt = formats.get(sequence.get('format')) if sequence is not None else None
if fmt is None:
fmt = next(iter(formats.values()), None)
if fmt is None:
return None
raw = fmt.get('frameDuration', '')
if not (raw.endswith('s') and '/' in raw):
return None
numerator, denominator = raw[:-1].split('/', 1)
try:
value = Fraction(int(numerator), int(denominator))
except (ValueError, ZeroDivisionError):
return None
return value if value > 0 else None
def _check_frame_alignment(root: ET.Element, fps: float = 24.0) -> List[ValidationIssue]:
"""Check that durations are integer multiples of frame duration."""
"""Check that durations are integer multiples of the frame duration.
Uses the document's exact ``frameDuration`` fraction and rational
arithmetic. Comparing against an integer fps instead would flag every
NTSC project as broken: at 1001/24000s (23.976fps) a perfectly aligned
duration is not an integer number of "24fps" frames, so whole timelines
would be reported misaligned when nothing is wrong.
"""
issues = []
fps_int = int(fps)
frame_duration = _document_frame_duration(root)
label = f"{1 / float(frame_duration):.3f}".rstrip('0').rstrip('.') if frame_duration else str(fps)
for elem in root.iter():
dur_str = elem.get('duration')
if not dur_str or not dur_str.endswith('s'):
@@ -553,14 +610,19 @@ def _check_frame_alignment(root: ET.Element, fps: float = 24.0) -> List[Validati
continue
try:
tv = TimeValue.from_timecode(dur_str)
frames = tv.to_seconds() * fps_int
if abs(frames - round(frames)) > 0.01:
if frame_duration is not None:
frames = Fraction(tv.numerator, tv.denominator) / frame_duration
aligned = frames.denominator == 1
else:
approx = tv.to_seconds() * fps
aligned = abs(approx - round(approx)) <= 0.01
if not aligned:
issues.append(ValidationIssue(
issue_type=ValidationIssueType.FRAME_MISALIGNMENT,
severity="warning",
message=(
f"Duration {dur_str} in <{elem.tag}> "
f"'{elem.get('name', '')}' is not frame-aligned at {fps_int}fps."
f"'{elem.get('name', '')}' is not frame-aligned at {label}fps."
),
clip_name=elem.get('name'),
))
@@ -1030,13 +1092,21 @@ class FCPXMLModifier:
elem.set(attr, val)
return elem
def _require_clip(self, clip_id: str) -> ET.Element:
def _require_clip(self, clip_id: 'str | ET.Element') -> ET.Element:
"""Look up a clip by ID/name, raising if not found.
Centralises the get-or-raise pattern used by every clip-mutating
method so the error message stays consistent and future
enhancements (fuzzy matching, suggestions) only need one site.
An Element is returned as-is. That matters after ``split_clip`` or
``cut_clip_ranges``: the resulting pieces all carry the *same* name,
so a name lookup would always resolve to the first one and silently
put the edit on the wrong piece. Callers holding the exact element
pass it directly.
"""
if isinstance(clip_id, ET.Element):
return clip_id
clip = self.clips.get(clip_id)
if clip is None:
raise ValueError(f"Clip not found: {clip_id}")
@@ -1421,7 +1491,7 @@ class FCPXMLModifier:
def add_marker(
self,
clip_id: str,
clip_id: 'str | ET.Element',
timecode: str,
name: str,
marker_type: "MarkerType | str" = MarkerType.STANDARD,
@@ -1937,35 +2007,50 @@ class FCPXMLModifier:
def add_zoom(
self,
clip_id: str,
clip_id: 'str | ET.Element',
start: float,
end: float,
scale: float = 1.3,
ease: float = 0.3,
ease: float = 0.25,
position: str = "0 0",
ease_out: Optional[float] = None,
hold_at_end: Optional[bool] = None,
start_at_peak: Optional[bool] = None,
) -> ET.Element:
"""Add a smooth ease-in/ease-out punch-in zoom to a clip.
"""Add a punch-in zoom to a clip, snapping back to its framing at the end.
Animates ``<adjust-transform>``'s ``scale`` param (per the FCPXML
DTD: ``<param>`` + ``<keyframeAnimation>`` of ``<keyframe>``
elements, ``interp="ease"``) from 100% up to *scale* and back down
to 100%, entirely within ``[start, end]`` — clip-relative seconds
(seconds from the clip's own head, same convention as
``cut_clip_ranges``). The ease portions each last *ease* seconds;
the zoom holds at *scale* in between.
Animates ``<adjust-transform>``'s ``scale`` param (``<param>`` +
``<keyframeAnimation>`` of ``<keyframe>``) from the clip's current
scale up to *scale* times it, holds, then returns — all within
``[start, end]`` — clip-relative seconds (same convention as
``cut_clip_ranges``).
The two ends are deliberately asymmetric. *ease* ramps the zoom
**in** over half a second by default, fast enough to land with the
emphasised word. The way **out** is instant — a single frame — so
the moment the impact phrase ends the shot is simply back to its
normal framing and the video resumes its flow, with no drift
drawing attention to itself. Pass *ease_out* to ramp the return
gradually instead.
*hold_at_end* keeps the peak instead of returning, and
*start_at_peak* opens already zoomed with no ramp. Left as ``None``
both decide on their own from how close the window sits to the
clip's edges: a cut is itself the transition, so ramping away from
one — or back toward one — is motion the viewer reads as a wobble
rather than as emphasis.
"""
if end <= start:
raise ValueError(f"end ({end}) must be greater than start ({start})")
if ease <= 0:
raise ValueError(f"ease must be positive, got {ease}")
if ease * 2 > (end - start):
raise ValueError(
f"ease ({ease}s x2 = {ease * 2}s) doesn't fit in the zoom "
f"window ({end - start}s) — shorten ease or widen start/end"
)
if scale <= 0:
raise ValueError(f"scale must be positive, got {scale}")
frame = float(self.frame_duration_fraction())
ramp_out = frame if ease_out is None else ease_out
if ramp_out <= 0:
raise ValueError(f"ease_out must be positive, got {ease_out}")
clip = self._require_clip(clip_id)
clip_duration = self._parse_time(clip.get('duration', '0s')).to_seconds()
if start < 0 or end > clip_duration:
@@ -1974,26 +2059,146 @@ class FCPXMLModifier:
f"duration (0 to {clip_duration:.3f}s)"
)
# Replace rather than stack a prior zoom on the same clip.
# Replace a prior zoom, but never the clip's framing. A clip can
# already carry an <adjust-transform> holding the editor's own
# reframe — rotation for footage shot sideways, position, a scale
# that makes the shot work at all. Dropping it outright (the old
# behaviour) silently destroyed that framing; on real footage the
# zoomed section came back rotated. So: keep the static attributes,
# and animate *relative to* the existing scale.
base_x, base_y = 1.0, 1.0
carried: dict = {}
old_keyframes: list = []
for stale in clip.findall('adjust-transform'):
carried = {k: v for k, v in stale.attrib.items() if k != 'scale'}
parts = (stale.get('scale') or '').split()
if len(parts) == 2:
try:
base_x, base_y = float(parts[0]), float(parts[1])
except ValueError:
base_x, base_y = 1.0, 1.0
else:
# No static attribute — a PRIOR zoom on this same clip left
# an animated <param name="scale"> instead, and the true
# resting framing lives in its keyframes, not in 1.0.
# Reading it as 1.0 here doesn't just miss the framing: it
# replaces the earlier zoom's whole animation with a wrong
# one, since this loop unconditionally removes `stale`
# right after. The rest value is recoverable without
# knowing which keyframe it is: MIN_ZOOM_SCALE == 1.0 means
# every keyframed value is >= the rest scale, so the
# smallest one keyframed is the rest value, peak or not.
for old_param in stale.findall("param[@name='scale']"):
xs, ys = [], []
for kf in old_param.findall('.//keyframe'):
kv = (kf.get('value') or '').split()
if len(kv) == 2:
try:
xs.append(float(kv[0]))
ys.append(float(kv[1]))
except ValueError:
pass
# Kept for merging: a second zoom on the same clip
# (two emphatic beats a cut didn't separate) should
# stack alongside the first, not erase it — the
# earlier peak is still a real editorial decision.
old_keyframes.append((kf.get('time', '0s'), kf.get('value', '')))
if xs and ys:
base_x, base_y = min(xs), min(ys)
clip.remove(stale)
transform = ET.Element('adjust-transform')
for key, value in carried.items():
transform.set(key, value)
scale_param = ET.SubElement(transform, 'param')
scale_param.set('name', 'scale')
anim = ET.SubElement(scale_param, 'keyframeAnimation')
scale_value = f"{scale} {scale}"
for seconds, value in (
(start, "1 1"),
(start + ease, scale_value),
(end - ease, scale_value),
(end, "1 1"),
):
# Keyframe times live in the clip's SOURCE timebase — the same origin
# as its own ``start`` — not in clip-relative seconds. A clip whose
# media starts at, say, 3109.9s of timecode looks for the animation
# there; keyframes written at 0-5s land outside the clip entirely and
# Final Cut imports the zoom as nothing at all, silently. Matches what
# add_text_title already does, and only shows up on footage whose
# start isn't 0s — every synthetic fixture starts at 0s and hides it.
media_origin = self._parse_time(clip.get('start', '0s'))
rest_value = f"{_fmt_scale(base_x)} {_fmt_scale(base_y)}"
scale_value = f"{_fmt_scale(base_x * scale)} {_fmt_scale(base_y * scale)}"
# A return that lands right before a cut is wasted motion: the next
# clip begins on its own framing anyway, so all the viewer sees is a
# twitch on the way out. When the zoom runs to the end of the clip,
# hold the peak and let the cut do the resetting.
holds_to_cut = (
hold_at_end
if hold_at_end is not None
else (clip_duration - end) <= HOLD_AT_CUT_THRESHOLD
)
opens_at_peak = (
start_at_peak
if start_at_peak is not None
else start <= START_AT_CUT_THRESHOLD
)
# Only the ramps actually written have to fit in the window: a zoom
# that opens at the peak spends no time ramping in, and one held to
# the cut spends none ramping out.
needed = (0.0 if opens_at_peak else ease) + (0.0 if holds_to_cut else ramp_out)
if needed > (end - start):
raise ValueError(
f"the ramps ({needed}s) don't fit in the zoom window "
f"({end - start}s) — shorten them or widen start/end"
)
if opens_at_peak:
# The cut already delivered the change of framing; ramping up
# from it just looks like the shot settling.
keyframes = [(start, scale_value)]
else:
keyframes = [(start, rest_value), (start + ease, scale_value)]
if holds_to_cut:
keyframes.append((end, scale_value))
else:
# Hold the peak right up to the end, then drop back on the very
# next frame — the snap-back the edit wants, not a slow drift.
keyframes.append((end - ramp_out, scale_value))
keyframes.append((end, rest_value))
new_entries = [
((media_origin + self.snap_seconds_to_frame(seconds)), value)
for seconds, value in keyframes
]
new_start_time = new_entries[0][0]
new_end_time = new_entries[-1][0]
# Two calls on the same clip mean two different things depending on
# whether their windows overlap. Overlapping = redoing the *same*
# zoom with new numbers — the old keyframes are stale and all of
# them go. Disjoint = a second, separate beat that a cut didn't
# separate onto its own clip — that one stacks alongside the first
# instead of erasing it, since both are real editorial decisions.
old_times = [self._parse_time(t) for t, _ in old_keyframes]
old_span_overlaps_new = bool(old_times) and not (
max(old_times) < new_start_time or min(old_times) > new_end_time
)
if old_span_overlaps_new:
surviving_old: list = []
else:
surviving_old = [(self._parse_time(t), v) for t, v in old_keyframes]
all_entries = sorted(surviving_old + new_entries, key=lambda e: e[0])
for time_value, value in all_entries:
kf = ET.SubElement(anim, 'keyframe')
kf.set('time', self.snap_seconds_to_frame(seconds).to_fcpxml())
kf.set('time', time_value.to_fcpxml())
kf.set('value', value)
kf.set('interp', 'ease')
# Only 'time' and 'value' — no 'interp', no 'curve'. The DTD allows
# both, but Final Cut rejected 'interp' on this vector param
# ("does not support the interpolation attribute") and discarded
# the whole <param>. A hand-made zoom exported from FCP itself
# writes bare keyframes and relies on the DTD default
# (curve="smooth"), so we match that export exactly rather than
# guess which attributes survive its importer.
if position != "0 0":
pos_param = ET.SubElement(transform, 'param')
@@ -2707,7 +2912,18 @@ class FCPXMLModifier:
_TEXT_POSITION_KEY = '9999/10003/13260/3296672360/1/100/101'
# Layout params the "Text" template ships with. These keys are the
# template's own defaults and never vary between instances.
#
# "Build Out" is the one deliberate override: with "Apply Speed" set to
# "2 (Per Object)" below, the template's whole built-in animation (build
# in + build out) is always compressed to exactly fill the title's own
# on-screen duration — so on a short word-length clip, build out was
# eating time that build in needed to finish revealing the text before
# the cut. Disabling build out hands that entire compressed window to
# build in alone, which is what "sempre acelerado" turned out to mean:
# no separate speed knob needed. Value captured from a real FCP export
# with "Build Out" unchecked in the Inspector (see chat, 2026-08-18).
_TEXT_TITLE_PARAMS = (
('Build Out', '9999/10000/2/102', '0'),
('Layout Method', '9999/10003/13260/3296672360/2/314', '1 (Paragraph)'),
('Left Margin', '9999/10003/13260/3296672360/2/323', '-1210'),
('Right Margin', '9999/10003/13260/3296672360/2/324', '1210'),
@@ -2795,6 +3011,7 @@ class FCPXMLModifier:
bold: bool = True,
face: Optional[str] = None,
kerning: Optional[float] = None,
font_scale: float = TEXT_TEMPLATE_FONT_SCALE,
) -> ET.Element:
"""Build a standalone ``<title>`` clip from the "Text" (Basic Text) template.
@@ -2849,13 +3066,31 @@ class FCPXMLModifier:
style_def.set('id', ts_id)
text_style = ET.SubElement(style_def, 'text-style')
text_style.set('font', font)
text_style.set('fontSize', str(font_size))
# Text.moti sizes type in frame pixels but positions in canvas points.
# See TEXT_TEMPLATE_FONT_SCALE: layout measures in points, so only the
# emitted size (and its kerning, to keep the same letter spacing) is
# converted here.
scale = float(font_scale) or 1.0
text_style.set('fontSize', f"{float(font_size) * scale:g}")
text_style.set('fontColor', font_color)
text_style.set('bold', '1' if bold else '0')
if face:
# FCP represents bold weight as the bold attribute — never as a
# fontFace. Writing ``bold="0" fontFace="Bold"`` (the previous
# behaviour) is contradictory and FCP refuses to render the text.
# Italic, by contrast, IS a face: FCP writes both ``fontFace`` and
# ``italic="1"``. See Engine/docs/05_EXPERIENCIAS.md, entry 2026-08-19.
face_lower = (face or '').strip().lower()
if face_lower == 'bold':
text_style.set('bold', '1')
elif 'italic' in face_lower:
text_style.set('fontFace', face)
text_style.set('italic', '1')
else:
if bold:
text_style.set('bold', '1')
if face:
text_style.set('fontFace', face)
if kerning:
text_style.set('kerning', f"{float(kerning):g}")
text_style.set('kerning', f"{float(kerning) * scale:g}")
text_style.set('alignment', 'center')
text_style.set('lineSpacing', '-19')
@@ -3005,7 +3240,9 @@ class FCPXMLModifier:
def lay_out(pending: List[Dict]):
"""Place what fits; return (units, still-unplaced words)."""
if phrase_mode:
composition = compose_sentence(pending, config.style, box)
composition = compose_sentence(
pending, config.style, box, line_gap=config.line_gap,
)
return composition.blocks, composition.overflow
layout = layout_sentence(pending, config.style, box)
return layout.placed, layout.overflow
@@ -3083,19 +3320,98 @@ class FCPXMLModifier:
duration,
lane=lane,
name=f"caption_{uuid.uuid4().hex[:8]}",
position=unit.position_param(),
position=unit.position_param(config.text_scale),
font=unit.font or config.style.font,
font_size=int(round(unit.font_size)),
font_color=unit.color or config.style.active_color,
bold=config.style.bold,
face=unit.face,
kerning=unit.kerning,
font_scale=config.text_scale,
)
_dtd_insert(parent, title)
created.append(title)
if getattr(config, 'validate', False):
report = self.validate_subtitle_layout()
if blocking(report["severity"]):
raise ValueError(
"Subtitle layout validation failed: "
+ str(report["summary"])
)
return created
def validate_subtitle_layout(
self,
*,
safe_margin_x: float = 0.05,
safe_margin_y: float = 0.05,
min_font_size: Optional[float] = None,
min_distance: Optional[float] = None,
max_distance: Optional[float] = None,
) -> dict:
"""Re-measure every ``<title>`` in the document and report collisions.
Reconstructs each title's on-screen box from the values the writer
emitted (``fontSize``/``kerning``/``Position`` are already in template
space), then checks for temporal+spatial collisions, frame/safe-area
containment, and font fallbacks. This is the spec-16 validation pass the
layout engine does not do on its own — it only guarantees non-overlap
*by construction* while composing, and cannot see a hand-edited title.
Returns the ``collision.validate_titles`` report: ``severity`` (worst
bucket), ``issues`` (spec-16 occurrences) and ``summary`` (counts).
"""
titles = []
for elem in self.root.iter('title'):
text_el = elem.find('text/text-style')
text = (text_el.text or '').strip() if text_el is not None else ''
style = elem.find('text-style-def/text-style')
font = style.get('font') if style is not None else None
face = style.get('fontFace') if style is not None else None
font_size = (
float(style.get('fontSize', '0')) if style is not None else 0.0
)
kerning = (
float(style.get('kerning', '0') or 0)
if style is not None else 0.0
)
x = y = 0.0
for param in elem.findall('param'):
if param.get('name') == 'Position' and param.get('value'):
parts = param.get('value').split()
if len(parts) >= 2:
x, y = float(parts[0]), float(parts[1])
start = self._parse_time(elem.get('offset', '0s')).to_seconds()
duration = self._parse_time(elem.get('duration', '0s')).to_seconds()
titles.append({
'text': text,
'font': font,
'face': face,
'font_size': font_size,
'kerning': kerning,
'x': x,
'y': y,
'start': start,
'end': start + duration,
'group': start + duration,
})
return validate_titles(
titles,
self.frame_width(),
self.frame_height(),
safe_margin_x=safe_margin_x,
safe_margin_y=safe_margin_y,
min_font_size=min_font_size,
min_distance=min_distance,
max_distance=max_distance,
)
# ========================================================================
# AUDIO CLIP OPERATIONS (v0.6.0)
# ========================================================================