chore: adiciona .gitignore e commit.command

This commit is contained in:
João Henrique
2026-08-18 08:25:29 -04:00
parent 68958fde00
commit 8fca456ceb
215 changed files with 65752 additions and 0 deletions
+139
View File
@@ -0,0 +1,139 @@
"""
FCPXML - Python library for reading, writing, and modifying Final Cut Pro XML files.
This package provides tools to:
- Parse FCPXML files into Python objects
- Modify existing FCPXML files (add markers, trim clips, reorder, etc.)
- Generate new FCPXML files from scratch
- Create AI-powered rough cuts from source material
- Compare timelines and export to other NLE formats
"""
from .diff import ClipDiff, MarkerDiff, TimelineDiff, compare_timelines
from .export import DaVinciExporter
from .models import (
MARKER_XML_TAGS,
AudioClip,
Clip,
CompoundClip,
ConnectedClip,
# Core models
Keyword,
Marker,
MarkerColor,
# Enums
MarkerType,
PacingConfig,
PacingStyle,
Project,
RoughCutResult,
# Rough cut models
SegmentSpec,
SilenceCandidate,
Timecode,
Timeline,
# Time handling
TimeValue,
Transition,
TransitionType,
VideoClip,
)
from .parser import FCPXMLParser, parse_fcpxml
from .rough_cut import (
RoughCutGenerator,
generate_rough_cut,
generate_segmented_rough_cut,
)
from .templates import (
BUILTIN_TEMPLATES,
ClipSpec,
Template,
TemplateSlot,
apply_template,
list_templates,
)
from .writer import (
FCP_EFFECTS,
FCPXMLModifier,
FCPXMLWriter,
add_marker_to_file,
build_marker_element,
list_effects,
modify_fcpxml,
trim_clip_in_file,
validate_fcpxml,
write_fcpxml,
)
__version__ = "0.6.35"
__author__ = "DareDev256"
__all__ = [
# Version
"__version__",
# Enums & constants
"MarkerType",
"MarkerColor",
"MARKER_XML_TAGS",
"TransitionType",
"PacingStyle",
# Time
"TimeValue",
"Timecode",
# Models
"Keyword",
"Marker",
"Clip",
"AudioClip",
"VideoClip",
"ConnectedClip",
"CompoundClip",
"SilenceCandidate",
"Transition",
"Timeline",
"Project",
"SegmentSpec",
"PacingConfig",
"RoughCutResult",
# Parser
"FCPXMLParser",
"parse_fcpxml",
# Writer
"FCPXMLWriter",
"FCPXMLModifier",
"modify_fcpxml",
"add_marker_to_file",
"trim_clip_in_file",
"write_fcpxml",
"build_marker_element",
"validate_fcpxml",
"FCP_EFFECTS",
"list_effects",
# Templates (v0.6.0)
"Template",
"TemplateSlot",
"ClipSpec",
"BUILTIN_TEMPLATES",
"list_templates",
"apply_template",
# Rough Cut
"RoughCutGenerator",
"generate_rough_cut",
"generate_segmented_rough_cut",
# Diff
"compare_timelines",
"TimelineDiff",
"ClipDiff",
"MarkerDiff",
# Export
"DaVinciExporter",
]
+150
View File
@@ -0,0 +1,150 @@
"""Speaker diarization for local transcripts — WHISPERX-inspired, lazy + graceful.
pyannote.audio (with an HF token granted access to the gated diarization
models) is optional: when it is absent or the token is missing, ``diarize``
returns ``None`` and callers fall back to ``SPEAKER_00`` for every word —
never crashing. The speaker-assignment helpers are pure functions over
``(start, end)`` tracks, so they are fully testable without any model.
This mirrors the reference WHISPERX app (``app._assign_speakers_to_words``),
adapted to our flat ``segments`` + ``words`` transcript shape.
"""
import logging
from typing import List, Optional, Sequence, Tuple
logger = logging.getLogger(__name__)
DEFAULT_SPEAKER = "SPEAKER_00"
# Raw speaker id -> canonical ``SPEAKER_NN`` label.
_DEFAULT_LABELS = ("SPEAKER_00", "SPEAKER_01", "SPEAKER_02", "SPEAKER_03")
def diarization_capability(token: Optional[str]) -> Tuple[bool, str]:
"""Whether speaker identification is available.
Returns ``(ok, message)``. ``ok`` is only ``True`` when pyannote.audio is
installed *and* a token is configured (both are needed for the gated
models).
"""
try:
import pyannote.audio # noqa: F401
except Exception:
return False, "Diarização indisponível: componente pyannote.audio ausente."
if not (token or "").strip():
return False, "Diarização indisponível: nenhum token HuggingFace configurado."
return True, "Identificação de participantes disponível."
def diarize(
path: str,
token: Optional[str],
num_speakers: str = "",
) -> Optional[List[Tuple[float, float, str]]]:
"""Run speaker diarization on ``path``.
Returns a list of ``(start, end, raw_speaker_id)`` turns, or ``None`` when
pyannote is unavailable, the token is missing/invalid, or analysis fails —
the graceful-degradation contract shared with ``transcribe``.
"""
if not diarization_capability(token)[0]:
return None
try:
from pyannote.audio import Pipeline
pipe = None
try:
pipe = Pipeline.from_pretrained("pyannote/speaker-diarization-3.1", token=token)
except TypeError:
# pyannote.audio < 4.0 used use_auth_token instead of token.
pipe = Pipeline.from_pretrained(
"pyannote/speaker-diarization-3.1", use_auth_token=token
)
if pipe is None:
return None
kwargs = {}
n = str(num_speakers or "").strip()
if n.isdigit() and int(n) > 0:
kwargs["num_speakers"] = int(n)
result = pipe(path, **kwargs)
# pyannote.audio >= 4.0 wraps the annotation; normalize to the raw one.
if hasattr(result, "exclusive_speaker_diarization"):
result = result.exclusive_speaker_diarization
elif hasattr(result, "speaker_diarization"):
result = result.speaker_diarization
tracks: List[Tuple[float, float, str]] = []
for turn, _, speaker in result.itertracks(yield_label=True):
tracks.append((float(turn.start), float(turn.end), str(speaker)))
return tracks or None
except Exception:
logger.warning("diarization failed for %s", path)
return None
def _overlap_speaker(
start: float, end: float, tracks: Sequence[Tuple[float, float, str]]
) -> Optional[str]:
"""Speaker with the largest summed time-overlap with ``[start, end]``."""
overlap: dict[str, float] = {}
for t_start, t_end, sid in tracks:
o = min(end, t_end) - max(start, t_start)
if o > 0:
overlap[sid] = overlap.get(sid, 0.0) + o
if not overlap:
return None
return max(overlap.items(), key=lambda kv: kv[1])[0]
def assign_speakers(
segments: Sequence[dict],
words: Sequence[dict],
tracks: Optional[Sequence[Tuple[float, float, str]]],
default: str = DEFAULT_SPEAKER,
) -> Tuple[List[dict], List[dict]]:
"""Attach ``speaker_id`` to every segment and word.
``tracks`` is the output of :func:`diarize` (may be ``None``). Raw speaker
ids are mapped to stable ``SPEAKER_NN`` labels in first-seen order. Without
tracks everything is assigned ``default``.
"""
norm_tracks: List[Tuple[float, float, str]] = []
if tracks:
label_map: dict[str, str] = {}
counter = 0
for t_start, t_end, raw in tracks:
if raw not in label_map:
label = _DEFAULT_LABELS[counter] if counter < len(_DEFAULT_LABELS) else f"SPEAKER_{counter:02d}"
label_map[raw] = label
counter += 1
norm_tracks.append((t_start, t_end, label_map[raw]))
out_segments: List[dict] = []
for seg in segments:
s = dict(seg)
s["speaker_id"] = (
_overlap_speaker(seg.get("start", 0.0), seg.get("end", 0.0), norm_tracks) or default
)
out_segments.append(s)
out_words: List[dict] = []
for w in words:
ww = dict(w)
ww["speaker_id"] = (
_overlap_speaker(w.get("start", 0.0), w.get("end", 0.0), norm_tracks) or default
)
out_words.append(ww)
return out_segments, out_words
def build_speakers(segments: Sequence[dict]) -> List[dict]:
"""Ordered ``[{"id", "name"}, ...]`` from the speakers present in segments."""
ids: List[str] = []
seen = set()
for seg in segments:
sid = seg.get("speaker_id", DEFAULT_SPEAKER)
if sid not in seen:
seen.add(sid)
ids.append(sid)
return [{"id": sid, "name": f"Speaker {i + 1}"} for i, sid in enumerate(ids)]
+269
View File
@@ -0,0 +1,269 @@
"""Timeline comparison engine for FCPXML files.
Compares two FCPXML timelines and reports differences in clips,
markers, transitions, and format settings.
"""
from dataclasses import dataclass, field
from typing import Dict, List, Optional, Tuple
from .parser import FCPXMLParser
@dataclass
class ClipDiff:
"""Difference record for a single clip."""
action: str # "added", "removed", "moved", "trimmed", "unchanged"
clip_name: str
details: str = ""
old_start: Optional[float] = None
new_start: Optional[float] = None
old_duration: Optional[float] = None
new_duration: Optional[float] = None
@dataclass
class MarkerDiff:
"""Difference record for a marker."""
action: str # "added", "removed", "moved"
marker_name: str
details: str = ""
old_position: Optional[float] = None
new_position: Optional[float] = None
@dataclass
class TimelineDiff:
"""Complete diff between two timelines."""
timeline_a_name: str
timeline_b_name: str
clip_diffs: List[ClipDiff] = field(default_factory=list)
marker_diffs: List[MarkerDiff] = field(default_factory=list)
transition_diffs: List[str] = field(default_factory=list)
format_changes: List[str] = field(default_factory=list)
@property
def total_changes(self) -> int:
changes = [d for d in self.clip_diffs if d.action != "unchanged"]
return len(changes) + len(self.marker_diffs) + len(self.transition_diffs) + len(self.format_changes)
@property
def has_changes(self) -> bool:
return self.total_changes > 0
def _clip_identity(clip) -> Tuple[str, float]:
"""Build identity key for a clip: (name, source_start rounded to 0.01s)."""
source_start = round(clip.source_start.seconds, 2) if clip.source_start else 0.0
return (clip.name, source_start)
def compare_timelines(filepath_a: str, filepath_b: str) -> TimelineDiff:
"""Compare two FCPXML files and return structured diff.
Uses clip identity (name + source in-point) to match clips between
timelines, then detects moved, trimmed, added, and removed clips.
Args:
filepath_a: Path to baseline FCPXML
filepath_b: Path to comparison FCPXML
Returns:
TimelineDiff with all detected changes
"""
parser_a = FCPXMLParser()
parser_b = FCPXMLParser()
project_a = parser_a.parse_file(filepath_a)
project_b = parser_b.parse_file(filepath_b)
tl_a = project_a.primary_timeline
tl_b = project_b.primary_timeline
if tl_a is None or tl_b is None:
return TimelineDiff(
timeline_a_name=tl_a.name if tl_a else "No timeline",
timeline_b_name=tl_b.name if tl_b else "No timeline",
)
diff = TimelineDiff(
timeline_a_name=tl_a.name,
timeline_b_name=tl_b.name,
)
# Compare clips
_compare_clips(tl_a.clips, tl_b.clips, diff)
# Compare markers
_compare_markers(tl_a, tl_b, diff)
# Compare transitions
_compare_transitions(tl_a.transitions, tl_b.transitions, diff)
# Compare format
_compare_format(tl_a, tl_b, diff)
return diff
def _compare_clips(clips_a, clips_b, diff: TimelineDiff):
"""Compare clip sequences between two timelines."""
# Build identity maps
map_a: Dict[Tuple[str, float], list] = {}
for clip in clips_a:
key = _clip_identity(clip)
map_a.setdefault(key, []).append(clip)
map_b: Dict[Tuple[str, float], list] = {}
for clip in clips_b:
key = _clip_identity(clip)
map_b.setdefault(key, []).append(clip)
all_keys = set(map_a.keys()) | set(map_b.keys())
for key in sorted(all_keys, key=lambda k: k[1]):
a_clips = map_a.get(key, [])
b_clips = map_b.get(key, [])
if not a_clips and b_clips:
for clip in b_clips:
diff.clip_diffs.append(ClipDiff(
action="added",
clip_name=clip.name,
details=f"Added at {clip.start.to_smpte()}",
new_start=clip.start.seconds,
new_duration=clip.duration_seconds,
))
elif a_clips and not b_clips:
for clip in a_clips:
diff.clip_diffs.append(ClipDiff(
action="removed",
clip_name=clip.name,
details=f"Removed from {clip.start.to_smpte()}",
old_start=clip.start.seconds,
old_duration=clip.duration_seconds,
))
else:
# Match clips pairwise
for i in range(max(len(a_clips), len(b_clips))):
if i >= len(a_clips):
diff.clip_diffs.append(ClipDiff(
action="added", clip_name=b_clips[i].name,
new_start=b_clips[i].start.seconds,
new_duration=b_clips[i].duration_seconds,
))
elif i >= len(b_clips):
diff.clip_diffs.append(ClipDiff(
action="removed", clip_name=a_clips[i].name,
old_start=a_clips[i].start.seconds,
old_duration=a_clips[i].duration_seconds,
))
else:
a, b = a_clips[i], b_clips[i]
moved = abs(a.start.seconds - b.start.seconds) > 0.04
trimmed = abs(a.duration_seconds - b.duration_seconds) > 0.04
if moved and trimmed:
diff.clip_diffs.append(ClipDiff(
action="moved",
clip_name=a.name,
details=(
f"Moved {a.start.to_smpte()} -> {b.start.to_smpte()}, "
f"trimmed {a.duration_seconds:.2f}s -> {b.duration_seconds:.2f}s"
),
old_start=a.start.seconds,
new_start=b.start.seconds,
old_duration=a.duration_seconds,
new_duration=b.duration_seconds,
))
elif moved:
diff.clip_diffs.append(ClipDiff(
action="moved",
clip_name=a.name,
details=f"Moved {a.start.to_smpte()} -> {b.start.to_smpte()}",
old_start=a.start.seconds,
new_start=b.start.seconds,
old_duration=a.duration_seconds,
new_duration=b.duration_seconds,
))
elif trimmed:
diff.clip_diffs.append(ClipDiff(
action="trimmed",
clip_name=a.name,
details=f"Duration {a.duration_seconds:.2f}s -> {b.duration_seconds:.2f}s",
old_start=a.start.seconds,
new_start=b.start.seconds,
old_duration=a.duration_seconds,
new_duration=b.duration_seconds,
))
def _compare_markers(tl_a, tl_b, diff: TimelineDiff):
"""Compare markers between two timelines."""
# Collect all markers including clip-level markers
markers_a = {}
for m in tl_a.markers:
markers_a[m.name] = m.start.seconds
for clip in tl_a.clips:
for m in clip.markers:
markers_a[m.name] = clip.start.seconds + m.start.seconds
markers_b = {}
for m in tl_b.markers:
markers_b[m.name] = m.start.seconds
for clip in tl_b.clips:
for m in clip.markers:
markers_b[m.name] = clip.start.seconds + m.start.seconds
all_names = set(markers_a.keys()) | set(markers_b.keys())
for name in sorted(all_names):
if name in markers_a and name not in markers_b:
diff.marker_diffs.append(MarkerDiff(
action="removed", marker_name=name,
old_position=markers_a[name],
))
elif name not in markers_a and name in markers_b:
diff.marker_diffs.append(MarkerDiff(
action="added", marker_name=name,
new_position=markers_b[name],
))
elif abs(markers_a[name] - markers_b[name]) > 1.0:
diff.marker_diffs.append(MarkerDiff(
action="moved", marker_name=name,
details=f"Position {markers_a[name]:.2f}s -> {markers_b[name]:.2f}s",
old_position=markers_a[name],
new_position=markers_b[name],
))
def _compare_transitions(trans_a, trans_b, diff: TimelineDiff):
"""Compare transitions between timelines."""
count_a = len(trans_a)
count_b = len(trans_b)
if count_a != count_b:
diff.transition_diffs.append(
f"Transition count changed: {count_a} -> {count_b}"
)
for i in range(min(count_a, count_b)):
a, b = trans_a[i], trans_b[i]
if a.name != b.name:
diff.transition_diffs.append(
f"Transition {i + 1}: {a.name} -> {b.name}"
)
if abs(a.duration.seconds - b.duration.seconds) > 0.04:
diff.transition_diffs.append(
f"Transition {i + 1} duration: {a.duration.seconds:.2f}s -> {b.duration.seconds:.2f}s"
)
def _compare_format(tl_a, tl_b, diff: TimelineDiff):
"""Compare format settings (resolution, frame rate)."""
if tl_a.width != tl_b.width or tl_a.height != tl_b.height:
diff.format_changes.append(
f"Resolution: {tl_a.width}x{tl_a.height} -> {tl_b.width}x{tl_b.height}"
)
if abs(tl_a.frame_rate - tl_b.frame_rate) > 0.01:
diff.format_changes.append(
f"Frame rate: {tl_a.frame_rate:.3f} -> {tl_b.frame_rate:.3f}"
)
+112
View File
@@ -0,0 +1,112 @@
"""Validate FCPXML against Apple's official DTDs.
Apple stopped publishing FCPXML DTDs online after version 1.10 — the only
authoritative spec is the set of DTD files shipped inside the Final Cut
Pro app bundle (FCPXMLv1_0.dtd through FCPXMLv1_14.dtd as of FCP 12.x).
This module locates those DTDs on the local machine and validates
generated FCPXML against them with ``xmllint``.
Validation is strictly opt-in/best-effort: on machines without Final Cut
Pro (CI, Linux, servers) every helper degrades to "DTDs unavailable"
rather than failing, so the server never *requires* FCP to be installed.
Set ``FCPXML_DTD_DIR`` to point at a directory of ``FCPXMLv*_*.dtd``
files to validate without an FCP install (e.g. DTDs copied to a CI
runner — note Apple's DTDs are Apple-copyrighted, so they are located
at runtime rather than redistributed with this project).
"""
import os
import shutil
import subprocess
from pathlib import Path
from typing import Optional, Tuple
# Where FCP 10.6+ keeps its DTDs (verified against FCP 12.2)
_INTERCHANGE_RESOURCES = Path(
"/Applications/Final Cut Pro.app/Contents/Frameworks/"
"Interchange.framework/Versions/A/Resources"
)
_XMLLINT_TIMEOUT_SECONDS = 30
def dtd_search_dir() -> Path:
"""Directory searched for Apple FCPXML DTDs (env-overridable)."""
return Path(os.environ.get("FCPXML_DTD_DIR", str(_INTERCHANGE_RESOURCES)))
def find_apple_dtd(version: str) -> Optional[Path]:
"""Locate the Apple DTD for an FCPXML *version* (e.g. ``"1.13"``).
Returns None when the DTD (or the whole directory) is absent —
typically a machine without Final Cut Pro installed.
"""
safe = version.strip()
if not safe.replace(".", "").isdigit():
return None
candidate = dtd_search_dir() / f"FCPXMLv{safe.replace('.', '_')}.dtd"
return candidate if candidate.is_file() else None
def available_dtd_versions() -> list[str]:
"""List FCPXML versions with a locally available Apple DTD."""
directory = dtd_search_dir()
if not directory.is_dir():
return []
versions = []
for dtd in directory.glob("FCPXMLv*_*.dtd"):
versions.append(dtd.stem.removeprefix("FCPXMLv").replace("_", "."))
return sorted(versions, key=lambda v: [int(p) for p in v.split(".")])
def validate_against_dtd(
fcpxml_path: str,
version: Optional[str] = None,
) -> Tuple[Optional[bool], str]:
"""Validate an FCPXML file against Apple's DTD via ``xmllint``.
Args:
fcpxml_path: Path to a ``.fcpxml`` file or ``.fcpxmld`` bundle.
version: FCPXML version to validate against. Defaults to the
file's own ``version`` attribute.
Returns:
``(ok, detail)`` where *ok* is True (valid), False (invalid), or
None (validation unavailable: no xmllint, no DTD on this
machine, or unreadable input — *detail* says which).
"""
path = Path(fcpxml_path)
if path.suffix.lower() == ".fcpxmld":
path = path / "Info.fcpxml"
if not path.is_file():
return None, f"File not found: {path}"
if shutil.which("xmllint") is None:
return None, "xmllint not available on PATH"
if version is None:
# Cheap version sniff without a full parse
from .safe_xml import safe_parse
version = safe_parse(str(path)).getroot().get("version", "1.13")
dtd = find_apple_dtd(version)
if dtd is None:
return None, (
f"No Apple DTD for FCPXML {version} found in {dtd_search_dir()} "
f"(Final Cut Pro not installed? Set FCPXML_DTD_DIR to override)"
)
# xmllint treats the --dtdvalid argument as a URI: raw paths containing
# spaces (e.g. ".../Final Cut Pro.app/...") fail entity resolution with
# "xmlSAX2ResolveEntity". Path.as_uri() percent-encodes them.
proc = subprocess.run(
["xmllint", "--noout", "--dtdvalid", dtd.resolve().as_uri(), str(path)],
capture_output=True,
text=True,
timeout=_XMLLINT_TIMEOUT_SECONDS,
)
if proc.returncode == 0:
return True, f"Valid against FCPXML {version} DTD ({dtd.name})"
return False, proc.stderr.strip() or f"xmllint exit code {proc.returncode}"
+226
View File
@@ -0,0 +1,226 @@
"""Export FCPXML to other NLE formats.
Supports:
- Simplified FCPXML v1.9 for DaVinci Resolve compatibility
- FCP7 XML (XMEML) for Premiere Pro / Resolve / Avid compatibility
"""
import copy
import xml.etree.ElementTree as ET
from typing import Any, Dict, List
from .parser import FCPXMLParser
from .safe_xml import serialize_xml
from .writer import _sanitize_xml_value
class DaVinciExporter:
"""Export FCPXML in formats compatible with other NLEs."""
def __init__(self, source_path: str):
"""Load the source FCPXML file.
Args:
source_path: Path to the source FCPXML file
"""
self.source_path = source_path
from .safe_xml import safe_parse
self.tree = safe_parse(source_path)
self.root = self.tree.getroot()
self.parser = FCPXMLParser()
self.project = self.parser.parse_file(source_path)
def export_simplified_fcpxml(self, output_path: str,
flatten_compounds: bool = True) -> str:
"""Generate simplified FCPXML v1.9 for DaVinci Resolve.
Strips features that cause Resolve import issues:
- Downgrades version to 1.9
- Optionally flattens compound clips
- Removes unsupported attributes
Args:
output_path: Destination file path
flatten_compounds: Whether to flatten compound clips
Returns:
Path to the generated file
"""
root = copy.deepcopy(self.root)
root.set('version', '1.9')
# Strip attributes that cause Resolve issues
unsupported_attrs = ['mcClipAngle', 'modDate', 'colorProcessing']
for elem in root.iter():
for attr in unsupported_attrs:
if attr in elem.attrib:
del elem.attrib[attr]
# Flatten compound clips (ref-clips pointing to media sequences)
if flatten_compounds:
for spine in root.findall('.//spine'):
for ref_clip in spine.findall('ref-clip'):
# Convert ref-clip to simple asset-clip
ref_clip.tag = 'asset-clip'
# Keep core attributes, strip compound-specific ones
for attr in list(ref_clip.attrib.keys()):
if attr not in ('ref', 'offset', 'duration', 'start', 'name', 'format'):
del ref_clip.attrib[attr]
return serialize_xml(root, output_path, '<!DOCTYPE fcpxml>')
def export_xmeml(self, output_path: str) -> str:
"""Generate FCP7 XML (XMEML) for maximum NLE compatibility.
Converts FCPXML's spine-based model to XMEML's track-based model:
- Primary storyline -> Video Track 1
- Connected clips lane +N -> Video Track N+1
- Connected clips lane -N -> Audio Track N+1
Args:
output_path: Destination file path
Returns:
Path to the generated file
"""
tl = self.project.primary_timeline
if tl is None:
raise ValueError("No timeline found in source FCPXML")
# Build track model from spine
video_tracks, audio_tracks = self._spine_to_tracks()
# Build XMEML document
xmeml = ET.Element('xmeml')
xmeml.set('version', '5')
sequence = ET.SubElement(xmeml, 'sequence')
ET.SubElement(sequence, 'name').text = _sanitize_xml_value(tl.name, 512)
total_frames = int(tl.duration.seconds * tl.frame_rate)
ET.SubElement(sequence, 'duration').text = str(total_frames)
rate = ET.SubElement(sequence, 'rate')
ET.SubElement(rate, 'timebase').text = str(int(tl.frame_rate))
is_ntsc = tl.frame_rate in (29.97, 23.976, 59.94)
ET.SubElement(rate, 'ntsc').text = 'TRUE' if is_ntsc else 'FALSE'
# Timecode
tc = ET.SubElement(sequence, 'timecode')
tc_rate = ET.SubElement(tc, 'rate')
ET.SubElement(tc_rate, 'timebase').text = str(int(tl.frame_rate))
ET.SubElement(tc_rate, 'ntsc').text = 'TRUE' if is_ntsc else 'FALSE'
ET.SubElement(tc, 'string').text = '00:00:00:00'
ET.SubElement(tc, 'frame').text = '0'
media = ET.SubElement(sequence, 'media')
# Video tracks
video = ET.SubElement(media, 'video')
format_elem = ET.SubElement(video, 'format')
sample = ET.SubElement(format_elem, 'samplecharacteristics')
ET.SubElement(sample, 'width').text = str(tl.width)
ET.SubElement(sample, 'height').text = str(tl.height)
for track_num in sorted(video_tracks.keys()):
track = ET.SubElement(video, 'track')
for clip_data in video_tracks[track_num]:
self._add_xmeml_clipitem(track, clip_data, tl.frame_rate)
# Audio tracks
audio = ET.SubElement(media, 'audio')
for track_num in sorted(audio_tracks.keys()):
track = ET.SubElement(audio, 'track')
for clip_data in audio_tracks[track_num]:
self._add_xmeml_clipitem(track, clip_data, tl.frame_rate)
# If no explicit audio tracks, create one from primary storyline
if not audio_tracks and video_tracks.get(0):
track = ET.SubElement(audio, 'track')
for clip_data in video_tracks[0]:
if clip_data.get('has_audio', True):
self._add_xmeml_clipitem(track, clip_data, tl.frame_rate)
return serialize_xml(xmeml, output_path, '<!DOCTYPE xmeml>')
def _spine_to_tracks(self) -> tuple:
"""Convert spine + connected clips to track-based model.
Returns:
Tuple of (video_tracks, audio_tracks) where each is a
dict mapping track number to list of clip data dicts.
"""
tl = self.project.primary_timeline
video_tracks: Dict[int, List[Dict[str, Any]]] = {0: []}
audio_tracks: Dict[int, List[Dict[str, Any]]] = {}
# Primary storyline -> track 0
for clip in tl.clips:
video_tracks[0].append({
'name': clip.name,
'start_seconds': clip.start.seconds,
'duration_seconds': clip.duration_seconds,
'source_start_seconds': clip.source_start.seconds if clip.source_start else 0,
'media_path': clip.media_path,
'has_audio': True,
})
# Connected clips -> higher tracks
for cc in tl.connected_clips:
if cc.lane > 0:
track_num = cc.lane
if track_num not in video_tracks:
video_tracks[track_num] = []
video_tracks[track_num].append({
'name': cc.name,
'start_seconds': cc.offset.seconds if cc.offset else 0,
'duration_seconds': cc.duration_seconds,
'source_start_seconds': cc.source_start.seconds if cc.source_start else 0,
'media_path': cc.media_path,
'has_audio': cc.clip_type not in ('title', 'video'),
})
elif cc.lane < 0:
track_num = abs(cc.lane) - 1
if track_num not in audio_tracks:
audio_tracks[track_num] = []
audio_tracks[track_num].append({
'name': cc.name,
'start_seconds': cc.offset.seconds if cc.offset else 0,
'duration_seconds': cc.duration_seconds,
'source_start_seconds': cc.source_start.seconds if cc.source_start else 0,
'media_path': cc.media_path,
'has_audio': True,
})
return video_tracks, audio_tracks
def _add_xmeml_clipitem(self, track: ET.Element,
clip_data: Dict[str, Any],
frame_rate: float):
"""Add a clipitem element to an XMEML track."""
clipitem = ET.SubElement(track, 'clipitem')
ET.SubElement(clipitem, 'name').text = _sanitize_xml_value(
clip_data['name'], 512
)
duration_frames = int(clip_data['duration_seconds'] * frame_rate)
ET.SubElement(clipitem, 'duration').text = str(duration_frames)
rate = ET.SubElement(clipitem, 'rate')
ET.SubElement(rate, 'timebase').text = str(int(frame_rate))
is_ntsc = frame_rate in (29.97, 23.976, 59.94)
ET.SubElement(rate, 'ntsc').text = 'TRUE' if is_ntsc else 'FALSE'
start_frame = int(clip_data['start_seconds'] * frame_rate)
source_in = int(clip_data['source_start_seconds'] * frame_rate)
source_out = source_in + duration_frames
ET.SubElement(clipitem, 'start').text = str(start_frame)
ET.SubElement(clipitem, 'end').text = str(start_frame + duration_frames)
ET.SubElement(clipitem, 'in').text = str(source_in)
ET.SubElement(clipitem, 'out').text = str(source_out)
if clip_data.get('media_path'):
file_elem = ET.SubElement(clipitem, 'file')
pathurl = ET.SubElement(file_elem, 'pathurl')
pathurl.text = _sanitize_xml_value(clip_data['media_path'], 2048)
+445
View File
@@ -0,0 +1,445 @@
"""Glyph advance widths for the fonts the subtitle rhythm uses.
Generated from the actual font files on the editing machine — the macOS
system fonts (/System/Library/Fonts/Helvetica.ttc, HelveticaNeue.ttc) and the
installed Playfair Display family (~/Library/Fonts) the editorial caption look
uses — as a fraction of the em size. Embedded rather than read at runtime so layout needs no font
library and stays identical on any machine.
Estimating these from a generic table was not good enough: the variants differ
by up to 21% of the em on some glyphs, and an estimate that runs low on even
one word puts two words on top of each other — the failure this table exists to
rule out. Measured error against these tables is zero for the covered glyphs.
Keys are "<family>" or "<family>-<face>", lowercased.
"""
METRICS = {
'helvetica': {
' ': 0.2778, '!': 0.2778, '"': 0.355, '#': 0.5562, '$': 0.5562, '%': 0.8892, '&':
0.667, "'": 0.1909, '(': 0.333, ')': 0.333, '*': 0.3892, '+': 0.584, ',': 0.2778,
'-': 0.333, '.': 0.2778, '/': 0.2778, '0': 0.5562, '1': 0.5562, '2': 0.5562, '3':
0.5562, '4': 0.5562, '5': 0.5562, '6': 0.5562, '7': 0.5562, '8': 0.5562, '9':
0.5562, ':': 0.2778, ';': 0.2778, '<': 0.584, '=': 0.584, '>': 0.584, '?': 0.5562,
'@': 1.0151, 'A': 0.667, 'B': 0.667, 'C': 0.7222, 'D': 0.7222, 'E': 0.667, 'F':
0.6108, 'G': 0.7778, 'H': 0.7222, 'I': 0.2778, 'J': 0.5, 'K': 0.667, 'L': 0.5562,
'M': 0.833, 'N': 0.7222, 'O': 0.7778, 'P': 0.667, 'Q': 0.7778, 'R': 0.7222, 'S':
0.667, 'T': 0.6108, 'U': 0.7222, 'V': 0.667, 'W': 0.9438, 'X': 0.667, 'Y': 0.667,
'Z': 0.6108, '[': 0.2778, '\\': 0.2778, ']': 0.2778, '_': 0.5562, 'a': 0.5562, 'b':
0.5562, 'c': 0.5, 'd': 0.5562, 'e': 0.5562, 'f': 0.2778, 'g': 0.5562, 'h': 0.5562,
'i': 0.2222, 'j': 0.2222, 'k': 0.5, 'l': 0.2222, 'm': 0.833, 'n': 0.5562, 'o':
0.5562, 'p': 0.5562, 'q': 0.5562, 'r': 0.333, 's': 0.5, 't': 0.2778, 'u': 0.5562,
'v': 0.5, 'w': 0.7222, 'x': 0.5, 'y': 0.5, 'z': 0.5, '{': 0.334, '|': 0.2598, '}':
0.334, '~': 0.584, 'ª': 0.3701, 'º': 0.3652, 'À': 0.667, 'Á': 0.667, 'Â': 0.667,
'Ã': 0.667, 'Ä': 0.667, 'Ç': 0.7222, 'È': 0.667, 'É': 0.667, 'Ê': 0.667, 'Ë': 0.667,
'Ì': 0.2778, 'Í': 0.2778, 'Î': 0.2778, 'Ï': 0.2778, 'Ñ': 0.7222, 'Ò': 0.7778, 'Ó':
0.7778, 'Ô': 0.7778, 'Õ': 0.7778, 'Ö': 0.7778, 'Ù': 0.7222, 'Ú': 0.7222, 'Û':
0.7222, 'Ü': 0.7222, 'à': 0.5562, 'á': 0.5562, 'â': 0.5562, 'ã': 0.5562, 'ä':
0.5562, 'ç': 0.5, 'è': 0.5562, 'é': 0.5562, 'ê': 0.5562, 'ë': 0.5562, 'ì': 0.2778,
'í': 0.2778, 'î': 0.2778, 'ï': 0.2778, 'ñ': 0.5562, 'ò': 0.5562, 'ó': 0.5562, 'ô':
0.5562, 'õ': 0.5562, 'ö': 0.5562, 'ù': 0.5562, 'ú': 0.5562, 'û': 0.5562, 'ü': 0.5562
},
'helvetica neue': {
' ': 0.278, '!': 0.259, '"': 0.426, '#': 0.556, '$': 0.556, '%': 1.0, '&': 0.63,
"'": 0.278, '(': 0.259, ')': 0.259, '*': 0.352, '+': 0.6, ',': 0.278, '-': 0.389,
'.': 0.278, '/': 0.333, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
'<': 0.6, '=': 0.6, '>': 0.6, '?': 0.556, '@': 0.8, 'A': 0.648, 'B': 0.685, 'C':
0.722, 'D': 0.704, 'E': 0.611, 'F': 0.574, 'G': 0.759, 'H': 0.722, 'I': 0.259, 'J':
0.519, 'K': 0.667, 'L': 0.556, 'M': 0.871, 'N': 0.722, 'O': 0.76, 'P': 0.648, 'Q':
0.76, 'R': 0.685, 'S': 0.648, 'T': 0.574, 'U': 0.722, 'V': 0.611, 'W': 0.926, 'X':
0.611, 'Y': 0.648, 'Z': 0.611, '[': 0.259, '\\': 0.333, ']': 0.259, '_': 0.5, 'a':
0.537, 'b': 0.593, 'c': 0.537, 'd': 0.593, 'e': 0.537, 'f': 0.296, 'g': 0.574, 'h':
0.556, 'i': 0.222, 'j': 0.222, 'k': 0.519, 'l': 0.222, 'm': 0.853, 'n': 0.556, 'o':
0.574, 'p': 0.593, 'q': 0.593, 'r': 0.333, 's': 0.5, 't': 0.315, 'u': 0.556, 'v':
0.5, 'w': 0.758, 'x': 0.518, 'y': 0.5, 'z': 0.48, '{': 0.333, '|': 0.222, '}':
0.333, '~': 0.6, 'ª': 0.378, 'º': 0.384, 'À': 0.648, 'Á': 0.648, 'Â': 0.648, 'Ã':
0.648, 'Ä': 0.648, 'Ç': 0.722, 'È': 0.611, 'É': 0.611, 'Ê': 0.611, 'Ë': 0.611, 'Ì':
0.259, 'Í': 0.259, 'Î': 0.259, 'Ï': 0.259, 'Ñ': 0.722, 'Ò': 0.76, 'Ó': 0.76, 'Ô':
0.76, 'Õ': 0.76, 'Ö': 0.76, 'Ù': 0.722, 'Ú': 0.722, 'Û': 0.722, 'Ü': 0.722, 'à':
0.537, 'á': 0.537, 'â': 0.537, 'ã': 0.537, 'ä': 0.537, 'ç': 0.537, 'è': 0.537, 'é':
0.537, 'ê': 0.537, 'ë': 0.537, 'ì': 0.222, 'í': 0.222, 'î': 0.222, 'ï': 0.222, 'ñ':
0.556, 'ò': 0.574, 'ó': 0.574, 'ô': 0.574, 'õ': 0.574, 'ö': 0.574, 'ù': 0.556, 'ú':
0.556, 'û': 0.556, 'ü': 0.556
},
'helvetica neue-bold': {
' ': 0.278, '!': 0.278, '"': 0.463, '#': 0.556, '$': 0.556, '%': 1.0, '&': 0.685,
"'": 0.278, '(': 0.296, ')': 0.296, '*': 0.407, '+': 0.6, ',': 0.278, '-': 0.407,
'.': 0.278, '/': 0.371, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
'<': 0.6, '=': 0.6, '>': 0.6, '?': 0.556, '@': 0.8, 'A': 0.685, 'B': 0.704, 'C':
0.741, 'D': 0.741, 'E': 0.648, 'F': 0.593, 'G': 0.759, 'H': 0.741, 'I': 0.295, 'J':
0.556, 'K': 0.722, 'L': 0.593, 'M': 0.907, 'N': 0.741, 'O': 0.778, 'P': 0.667, 'Q':
0.778, 'R': 0.722, 'S': 0.649, 'T': 0.611, 'U': 0.741, 'V': 0.63, 'W': 0.944, 'X':
0.667, 'Y': 0.667, 'Z': 0.648, '[': 0.333, '\\': 0.371, ']': 0.333, '_': 0.5, 'a':
0.574, 'b': 0.611, 'c': 0.574, 'd': 0.611, 'e': 0.574, 'f': 0.333, 'g': 0.611, 'h':
0.593, 'i': 0.258, 'j': 0.278, 'k': 0.574, 'l': 0.258, 'm': 0.906, 'n': 0.593, 'o':
0.611, 'p': 0.611, 'q': 0.611, 'r': 0.389, 's': 0.537, 't': 0.352, 'u': 0.593, 'v':
0.52, 'w': 0.814, 'x': 0.537, 'y': 0.519, 'z': 0.519, '{': 0.333, '|': 0.223, '}':
0.333, '~': 0.6, 'ª': 0.344, 'º': 0.367, 'À': 0.685, 'Á': 0.685, 'Â': 0.685, 'Ã':
0.685, 'Ä': 0.685, 'Ç': 0.741, 'È': 0.648, 'É': 0.648, 'Ê': 0.648, 'Ë': 0.648, 'Ì':
0.295, 'Í': 0.295, 'Î': 0.295, 'Ï': 0.295, 'Ñ': 0.741, 'Ò': 0.778, 'Ó': 0.778, 'Ô':
0.778, 'Õ': 0.778, 'Ö': 0.778, 'Ù': 0.741, 'Ú': 0.741, 'Û': 0.741, 'Ü': 0.741, 'à':
0.574, 'á': 0.574, 'â': 0.574, 'ã': 0.574, 'ä': 0.574, 'ç': 0.574, 'è': 0.574, 'é':
0.574, 'ê': 0.574, 'ë': 0.574, 'ì': 0.258, 'í': 0.258, 'î': 0.258, 'ï': 0.258, 'ñ':
0.593, 'ò': 0.611, 'ó': 0.611, 'ô': 0.611, 'õ': 0.611, 'ö': 0.611, 'ù': 0.593, 'ú':
0.593, 'û': 0.593, 'ü': 0.593
},
'helvetica neue-italic': {
' ': 0.278, '!': 0.259, '"': 0.426, '#': 0.556, '$': 0.556, '%': 0.926, '&': 0.63,
"'": 0.278, '(': 0.259, ')': 0.259, '*': 0.352, '+': 0.6, ',': 0.278, '-': 0.389,
'.': 0.278, '/': 0.333, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
'<': 0.6, '=': 0.6, '>': 0.6, '?': 0.556, '@': 0.8, 'A': 0.667, 'B': 0.685, 'C':
0.722, 'D': 0.704, 'E': 0.611, 'F': 0.574, 'G': 0.759, 'H': 0.722, 'I': 0.259, 'J':
0.519, 'K': 0.667, 'L': 0.556, 'M': 0.87, 'N': 0.722, 'O': 0.759, 'P': 0.648, 'Q':
0.759, 'R': 0.685, 'S': 0.648, 'T': 0.574, 'U': 0.722, 'V': 0.611, 'W': 0.926, 'X':
0.611, 'Y': 0.611, 'Z': 0.611, '[': 0.259, '\\': 0.333, ']': 0.259, '_': 0.5, 'a':
0.519, 'b': 0.593, 'c': 0.537, 'd': 0.593, 'e': 0.537, 'f': 0.296, 'g': 0.574, 'h':
0.556, 'i': 0.222, 'j': 0.222, 'k': 0.481, 'l': 0.222, 'm': 0.852, 'n': 0.556, 'o':
0.574, 'p': 0.593, 'q': 0.593, 'r': 0.333, 's': 0.481, 't': 0.315, 'u': 0.556, 'v':
0.481, 'w': 0.759, 'x': 0.481, 'y': 0.481, 'z': 0.444, '{': 0.333, '|': 0.222, '}':
0.333, '~': 0.6, 'ª': 0.311, 'º': 0.344, 'À': 0.667, 'Á': 0.667, 'Â': 0.667, 'Ã':
0.667, 'Ä': 0.667, 'Ç': 0.722, 'È': 0.611, 'É': 0.611, 'Ê': 0.611, 'Ë': 0.611, 'Ì':
0.259, 'Í': 0.259, 'Î': 0.259, 'Ï': 0.259, 'Ñ': 0.722, 'Ò': 0.759, 'Ó': 0.759, 'Ô':
0.759, 'Õ': 0.759, 'Ö': 0.759, 'Ù': 0.722, 'Ú': 0.722, 'Û': 0.722, 'Ü': 0.722, 'à':
0.519, 'á': 0.519, 'â': 0.519, 'ã': 0.519, 'ä': 0.519, 'ç': 0.537, 'è': 0.537, 'é':
0.537, 'ê': 0.537, 'ë': 0.537, 'ì': 0.222, 'í': 0.222, 'î': 0.222, 'ï': 0.222, 'ñ':
0.556, 'ò': 0.574, 'ó': 0.574, 'ô': 0.574, 'õ': 0.574, 'ö': 0.574, 'ù': 0.556, 'ú':
0.556, 'û': 0.556, 'ü': 0.556
},
'helvetica neue-light': {
' ': 0.278, '!': 0.241, '"': 0.37, '#': 0.556, '$': 0.556, '%': 0.889, '&': 0.611,
"'": 0.278, '(': 0.241, ')': 0.241, '*': 0.352, '+': 0.6, ',': 0.278, '-': 0.37,
'.': 0.278, '/': 0.333, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
'<': 0.6, '=': 0.6, '>': 0.6, '?': 0.537, '@': 0.8, 'A': 0.63, 'B': 0.667, 'C':
0.704, 'D': 0.685, 'E': 0.593, 'F': 0.537, 'G': 0.741, 'H': 0.704, 'I': 0.222, 'J':
0.5, 'K': 0.648, 'L': 0.537, 'M': 0.821, 'N': 0.704, 'O': 0.741, 'P': 0.63, 'Q':
0.741, 'R': 0.667, 'S': 0.63, 'T': 0.556, 'U': 0.685, 'V': 0.593, 'W': 0.907, 'X':
0.574, 'Y': 0.611, 'Z': 0.574, '[': 0.241, '\\': 0.333, ']': 0.241, '_': 0.5, 'a':
0.519, 'b': 0.574, 'c': 0.519, 'd': 0.574, 'e': 0.519, 'f': 0.259, 'g': 0.556, 'h':
0.537, 'i': 0.185, 'j': 0.185, 'k': 0.5, 'l': 0.185, 'm': 0.833, 'n': 0.537, 'o':
0.556, 'p': 0.574, 'q': 0.574, 'r': 0.315, 's': 0.481, 't': 0.296, 'u': 0.537, 'v':
0.463, 'w': 0.741, 'x': 0.481, 'y': 0.463, 'z': 0.463, '{': 0.333, '|': 0.222, '}':
0.333, '~': 0.6, 'ª': 0.311, 'º': 0.334, 'À': 0.63, 'Á': 0.63, 'Â': 0.63, 'Ã': 0.63,
'Ä': 0.63, 'Ç': 0.704, 'È': 0.593, 'É': 0.593, 'Ê': 0.593, 'Ë': 0.593, 'Ì': 0.222,
'Í': 0.222, 'Î': 0.222, 'Ï': 0.222, 'Ñ': 0.704, 'Ò': 0.741, 'Ó': 0.741, 'Ô': 0.741,
'Õ': 0.741, 'Ö': 0.741, 'Ù': 0.685, 'Ú': 0.685, 'Û': 0.685, 'Ü': 0.685, 'à': 0.519,
'á': 0.519, 'â': 0.519, 'ã': 0.519, 'ä': 0.519, 'ç': 0.519, 'è': 0.519, 'é': 0.519,
'ê': 0.519, 'ë': 0.519, 'ì': 0.185, 'í': 0.185, 'î': 0.185, 'ï': 0.185, 'ñ': 0.537,
'ò': 0.556, 'ó': 0.556, 'ô': 0.556, 'õ': 0.556, 'ö': 0.556, 'ù': 0.537, 'ú': 0.537,
'û': 0.537, 'ü': 0.537
},
'helvetica neue-light italic': {
' ': 0.278, '!': 0.278, '"': 0.37, '#': 0.556, '$': 0.556, '%': 0.833, '&': 0.611,
"'": 0.278, '(': 0.259, ')': 0.259, '*': 0.352, '+': 0.6, ',': 0.278, '-': 0.37,
'.': 0.278, '/': 0.333, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
'<': 0.6, '=': 0.6, '>': 0.6, '?': 0.537, '@': 0.8, 'A': 0.63, 'B': 0.667, 'C':
0.704, 'D': 0.685, 'E': 0.574, 'F': 0.537, 'G': 0.741, 'H': 0.704, 'I': 0.222, 'J':
0.5, 'K': 0.648, 'L': 0.537, 'M': 0.852, 'N': 0.704, 'O': 0.741, 'P': 0.63, 'Q':
0.741, 'R': 0.648, 'S': 0.63, 'T': 0.556, 'U': 0.704, 'V': 0.593, 'W': 0.907, 'X':
0.574, 'Y': 0.574, 'Z': 0.574, '[': 0.241, '\\': 0.333, ']': 0.241, '_': 0.5, 'a':
0.519, 'b': 0.574, 'c': 0.519, 'd': 0.574, 'e': 0.519, 'f': 0.259, 'g': 0.556, 'h':
0.537, 'i': 0.185, 'j': 0.185, 'k': 0.463, 'l': 0.185, 'm': 0.833, 'n': 0.537, 'o':
0.556, 'p': 0.574, 'q': 0.574, 'r': 0.315, 's': 0.481, 't': 0.296, 'u': 0.537, 'v':
0.463, 'w': 0.741, 'x': 0.463, 'y': 0.463, 'z': 0.426, '{': 0.333, '|': 0.222, '}':
0.333, '~': 0.6, 'ª': 0.311, 'º': 0.334, 'À': 0.63, 'Á': 0.63, 'Â': 0.63, 'Ã': 0.63,
'Ä': 0.63, 'Ç': 0.704, 'È': 0.574, 'É': 0.574, 'Ê': 0.574, 'Ë': 0.574, 'Ì': 0.222,
'Í': 0.222, 'Î': 0.222, 'Ï': 0.222, 'Ñ': 0.704, 'Ò': 0.741, 'Ó': 0.741, 'Ô': 0.741,
'Õ': 0.741, 'Ö': 0.741, 'Ù': 0.704, 'Ú': 0.704, 'Û': 0.704, 'Ü': 0.704, 'à': 0.519,
'á': 0.519, 'â': 0.519, 'ã': 0.519, 'ä': 0.519, 'ç': 0.519, 'è': 0.519, 'é': 0.519,
'ê': 0.519, 'ë': 0.519, 'ì': 0.185, 'í': 0.185, 'î': 0.185, 'ï': 0.185, 'ñ': 0.537,
'ò': 0.556, 'ó': 0.556, 'ô': 0.556, 'õ': 0.556, 'ö': 0.556, 'ù': 0.537, 'ú': 0.537,
'û': 0.537, 'ü': 0.537
},
'helvetica-bold': {
' ': 0.2778, '!': 0.333, '"': 0.4741, '#': 0.5562, '$': 0.5562, '%': 0.8892, '&':
0.7222, "'": 0.2378, '(': 0.333, ')': 0.333, '*': 0.3892, '+': 0.584, ',': 0.2778,
'-': 0.333, '.': 0.2778, '/': 0.2778, '0': 0.5562, '1': 0.5562, '2': 0.5562, '3':
0.5562, '4': 0.5562, '5': 0.5562, '6': 0.5562, '7': 0.5562, '8': 0.5562, '9':
0.5562, ':': 0.333, ';': 0.333, '<': 0.584, '=': 0.584, '>': 0.584, '?': 0.6108,
'@': 0.9751, 'A': 0.7222, 'B': 0.7222, 'C': 0.7222, 'D': 0.7222, 'E': 0.667, 'F':
0.6108, 'G': 0.7778, 'H': 0.7222, 'I': 0.2778, 'J': 0.5562, 'K': 0.7222, 'L':
0.6108, 'M': 0.833, 'N': 0.7222, 'O': 0.7778, 'P': 0.667, 'Q': 0.7778, 'R': 0.7222,
'S': 0.667, 'T': 0.6108, 'U': 0.7222, 'V': 0.667, 'W': 0.9438, 'X': 0.667, 'Y':
0.667, 'Z': 0.6108, '[': 0.333, '\\': 0.2778, ']': 0.333, '_': 0.5562, 'a': 0.5562,
'b': 0.6108, 'c': 0.5562, 'd': 0.6108, 'e': 0.5562, 'f': 0.333, 'g': 0.6108, 'h':
0.6108, 'i': 0.2778, 'j': 0.2778, 'k': 0.5562, 'l': 0.2778, 'm': 0.8892, 'n':
0.6108, 'o': 0.6108, 'p': 0.6108, 'q': 0.6108, 'r': 0.3892, 's': 0.5562, 't': 0.333,
'u': 0.6108, 'v': 0.5562, 'w': 0.7778, 'x': 0.5562, 'y': 0.5562, 'z': 0.5, '{':
0.3892, '|': 0.2798, '}': 0.3892, '~': 0.584, 'ª': 0.3701, 'º': 0.3652, 'À': 0.7222,
'Á': 0.7222, 'Â': 0.7222, 'Ã': 0.7222, 'Ä': 0.7222, 'Ç': 0.7222, 'È': 0.667, 'É':
0.667, 'Ê': 0.667, 'Ë': 0.667, 'Ì': 0.2778, 'Í': 0.2778, 'Î': 0.2778, 'Ï': 0.2778,
'Ñ': 0.7222, 'Ò': 0.7778, 'Ó': 0.7778, 'Ô': 0.7778, 'Õ': 0.7778, 'Ö': 0.7778, 'Ù':
0.7222, 'Ú': 0.7222, 'Û': 0.7222, 'Ü': 0.7222, 'à': 0.5562, 'á': 0.5562, 'â':
0.5562, 'ã': 0.5562, 'ä': 0.5562, 'ç': 0.5562, 'è': 0.5562, 'é': 0.5562, 'ê':
0.5562, 'ë': 0.5562, 'ì': 0.2778, 'í': 0.2778, 'î': 0.2778, 'ï': 0.2778, 'ñ':
0.6108, 'ò': 0.6108, 'ó': 0.6108, 'ô': 0.6108, 'õ': 0.6108, 'ö': 0.6108, 'ù':
0.6108, 'ú': 0.6108, 'û': 0.6108, 'ü': 0.6108
},
'helvetica-light': {
' ': 0.278, '!': 0.333, '"': 0.278, '#': 0.556, '$': 0.556, '%': 0.889, '&': 0.667,
"'": 0.222, '(': 0.333, ')': 0.333, '*': 0.389, '+': 0.66, ',': 0.278, '-': 0.333,
'.': 0.278, '/': 0.278, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
'<': 0.66, '=': 0.66, '>': 0.66, '?': 0.5, '@': 0.8, 'A': 0.667, 'B': 0.667, 'C':
0.722, 'D': 0.722, 'E': 0.611, 'F': 0.556, 'G': 0.778, 'H': 0.722, 'I': 0.278, 'J':
0.5, 'K': 0.667, 'L': 0.556, 'M': 0.833, 'N': 0.722, 'O': 0.778, 'P': 0.611, 'Q':
0.778, 'R': 0.667, 'S': 0.611, 'T': 0.556, 'U': 0.722, 'V': 0.611, 'W': 0.889, 'X':
0.611, 'Y': 0.611, 'Z': 0.611, '[': 0.333, '\\': 0.278, ']': 0.333, '_': 0.5, 'a':
0.556, 'b': 0.611, 'c': 0.556, 'd': 0.611, 'e': 0.556, 'f': 0.278, 'g': 0.611, 'h':
0.556, 'i': 0.222, 'j': 0.222, 'k': 0.5, 'l': 0.222, 'm': 0.833, 'n': 0.556, 'o':
0.556, 'p': 0.611, 'q': 0.611, 'r': 0.333, 's': 0.5, 't': 0.278, 'u': 0.556, 'v':
0.5, 'w': 0.722, 'x': 0.5, 'y': 0.5, 'z': 0.5, '{': 0.333, '|': 0.222, '}': 0.333,
'~': 0.66, 'ª': 0.334, 'º': 0.334, 'À': 0.667, 'Á': 0.667, 'Â': 0.667, 'Ã': 0.667,
'Ä': 0.667, 'Ç': 0.722, 'È': 0.611, 'É': 0.611, 'Ê': 0.611, 'Ë': 0.611, 'Ì': 0.278,
'Í': 0.278, 'Î': 0.278, 'Ï': 0.278, 'Ñ': 0.722, 'Ò': 0.778, 'Ó': 0.778, 'Ô': 0.778,
'Õ': 0.778, 'Ö': 0.778, 'Ù': 0.722, 'Ú': 0.722, 'Û': 0.722, 'Ü': 0.722, 'à': 0.556,
'á': 0.556, 'â': 0.556, 'ã': 0.556, 'ä': 0.556, 'ç': 0.556, 'è': 0.556, 'é': 0.556,
'ê': 0.556, 'ë': 0.556, 'ì': 0.222, 'í': 0.222, 'î': 0.222, 'ï': 0.222, 'ñ': 0.556,
'ò': 0.556, 'ó': 0.556, 'ô': 0.556, 'õ': 0.556, 'ö': 0.556, 'ù': 0.556, 'ú': 0.556,
'û': 0.556, 'ü': 0.556
},
'helvetica-oblique': {
' ': 0.2778, '!': 0.2778, '"': 0.355, '#': 0.5562, '$': 0.5562, '%': 0.8892, '&':
0.667, "'": 0.1909, '(': 0.333, ')': 0.333, '*': 0.3892, '+': 0.584, ',': 0.2778,
'-': 0.333, '.': 0.2778, '/': 0.2778, '0': 0.5562, '1': 0.5562, '2': 0.5562, '3':
0.5562, '4': 0.5562, '5': 0.5562, '6': 0.5562, '7': 0.5562, '8': 0.5562, '9':
0.5562, ':': 0.2778, ';': 0.2778, '<': 0.584, '=': 0.584, '>': 0.584, '?': 0.5562,
'@': 1.0151, 'A': 0.667, 'B': 0.667, 'C': 0.7222, 'D': 0.7222, 'E': 0.667, 'F':
0.6108, 'G': 0.7778, 'H': 0.7222, 'I': 0.2778, 'J': 0.5, 'K': 0.667, 'L': 0.5562,
'M': 0.833, 'N': 0.7222, 'O': 0.7778, 'P': 0.667, 'Q': 0.7778, 'R': 0.7222, 'S':
0.667, 'T': 0.6108, 'U': 0.7222, 'V': 0.667, 'W': 0.9438, 'X': 0.667, 'Y': 0.667,
'Z': 0.6108, '[': 0.2778, '\\': 0.2778, ']': 0.2778, '_': 0.5562, 'a': 0.5562, 'b':
0.5562, 'c': 0.5, 'd': 0.5562, 'e': 0.5562, 'f': 0.2778, 'g': 0.5562, 'h': 0.5562,
'i': 0.2222, 'j': 0.2222, 'k': 0.5, 'l': 0.2222, 'm': 0.833, 'n': 0.5562, 'o':
0.5562, 'p': 0.5562, 'q': 0.5562, 'r': 0.333, 's': 0.5, 't': 0.2778, 'u': 0.5562,
'v': 0.5, 'w': 0.7222, 'x': 0.5, 'y': 0.5, 'z': 0.5, '{': 0.334, '|': 0.2598, '}':
0.334, '~': 0.584, 'ª': 0.3701, 'º': 0.3652, 'À': 0.667, 'Á': 0.667, 'Â': 0.667,
'Ã': 0.667, 'Ä': 0.667, 'Ç': 0.7222, 'È': 0.667, 'É': 0.667, 'Ê': 0.667, 'Ë': 0.667,
'Ì': 0.2778, 'Í': 0.2778, 'Î': 0.2778, 'Ï': 0.2778, 'Ñ': 0.7222, 'Ò': 0.7778, 'Ó':
0.7778, 'Ô': 0.7778, 'Õ': 0.7778, 'Ö': 0.7778, 'Ù': 0.7222, 'Ú': 0.7222, 'Û':
0.7222, 'Ü': 0.7222, 'à': 0.5562, 'á': 0.5562, 'â': 0.5562, 'ã': 0.5562, 'ä':
0.5562, 'ç': 0.5, 'è': 0.5562, 'é': 0.5562, 'ê': 0.5562, 'ë': 0.5562, 'ì': 0.2778,
'í': 0.2778, 'î': 0.2778, 'ï': 0.2778, 'ñ': 0.5562, 'ò': 0.5562, 'ó': 0.5562, 'ô':
0.5562, 'õ': 0.5562, 'ö': 0.5562, 'ù': 0.5562, 'ú': 0.5562, 'û': 0.5562, 'ü': 0.5562
},
'playfair display': {
' ': 0.249, '!': 0.262, '"': 0.298, '#': 0.637, '$': 0.523, '%': 0.742, '&':
0.837, "'": 0.189, '(': 0.294, ')': 0.294, '*': 0.492, '+': 0.587, ',': 0.261,
'-': 0.499, '.': 0.252, '/': 0.372, '0': 0.6, '1': 0.37, '2': 0.479, '3': 0.451,
'4': 0.494, '5': 0.415, '6': 0.517, '7': 0.406, '8': 0.52, '9': 0.509, ':':
0.266, ';': 0.285, '<': 0.608, '=': 0.647, '>': 0.608, '?': 0.459, '@': 0.878,
'A': 0.63, 'B': 0.623, 'C': 0.688, 'D': 0.723, 'E': 0.609, 'F': 0.569, 'G':
0.703, 'H': 0.757, 'I': 0.339, 'J': 0.326, 'K': 0.655, 'L': 0.588, 'M': 0.871,
'N': 0.706, 'O': 0.741, 'P': 0.585, 'Q': 0.741, 'R': 0.648, 'S': 0.54, 'T':
0.621, 'U': 0.682, 'V': 0.63, 'W': 0.905, 'X': 0.634, 'Y': 0.591, 'Z': 0.594,
'[': 0.296, '\\': 0.372, ']': 0.296, '_': 0.591, 'a': 0.498, 'b': 0.563, 'c':
0.486, 'd': 0.581, 'e': 0.506, 'f': 0.332, 'g': 0.528, 'h': 0.588, 'i': 0.293,
'j': 0.267, 'k': 0.55, 'l': 0.286, 'm': 0.888, 'n': 0.597, 'o': 0.548, 'p':
0.581, 'q': 0.563, 'r': 0.445, 's': 0.447, 't': 0.354, 'u': 0.581, 'v': 0.489,
'w': 0.77, 'x': 0.519, 'y': 0.504, 'z': 0.476, '{': 0.304, '|': 0.222, '}':
0.304, '~': 0.649, 'ª': 0.463, 'º': 0.454, 'À': 0.63, 'Á': 0.63, 'Â': 0.63, 'Ã':
0.63, 'Ä': 0.63, 'Ç': 0.688, 'È': 0.609, 'É': 0.609, 'Ê': 0.609, 'Ë': 0.609,
'Ì': 0.339, 'Í': 0.339, 'Î': 0.339, 'Ï': 0.339, 'Ñ': 0.706, 'Ò': 0.741, 'Ó':
0.741, 'Ô': 0.741, 'Õ': 0.741, 'Ö': 0.741, 'Ù': 0.682, 'Ú': 0.682, 'Û': 0.682,
'Ü': 0.682, 'à': 0.498, 'á': 0.498, 'â': 0.498, 'ã': 0.498, 'ä': 0.498, 'ç':
0.486, 'è': 0.506, 'é': 0.506, 'ê': 0.506, 'ë': 0.506, 'ì': 0.293, 'í': 0.293,
'î': 0.293, 'ï': 0.293, 'ñ': 0.597, 'ò': 0.548, 'ó': 0.548, 'ô': 0.548, 'õ':
0.548, 'ö': 0.548, 'ù': 0.581, 'ú': 0.581, 'û': 0.581, 'ü': 0.581
},
'playfair display-italic': {
' ': 0.253, '!': 0.263, '"': 0.296, '#': 0.639, '$': 0.551, '%': 0.766, '&':
0.817, "'": 0.187, '(': 0.361, ')': 0.361, '*': 0.495, '+': 0.585, ',': 0.252,
'-': 0.504, '.': 0.244, '/': 0.468, '0': 0.565, '1': 0.342, '2': 0.503, '3':
0.485, '4': 0.515, '5': 0.459, '6': 0.527, '7': 0.487, '8': 0.544, '9': 0.513,
':': 0.278, ';': 0.287, '<': 0.602, '=': 0.663, '>': 0.602, '?': 0.438, '@':
0.848, 'A': 0.629, 'B': 0.632, 'C': 0.67, 'D': 0.725, 'E': 0.61, 'F': 0.566,
'G': 0.705, 'H': 0.755, 'I': 0.341, 'J': 0.636, 'K': 0.675, 'L': 0.586, 'M':
0.868, 'N': 0.77, 'O': 0.729, 'P': 0.589, 'Q': 0.781, 'R': 0.664, 'S': 0.554,
'T': 0.659, 'U': 0.682, 'V': 0.632, 'W': 0.902, 'X': 0.699, 'Y': 0.67, 'Z':
0.652, '[': 0.3, '\\': 0.492, ']': 0.3, '_': 0.744, 'a': 0.523, 'b': 0.504, 'c':
0.432, 'd': 0.542, 'e': 0.429, 'f': 0.321, 'g': 0.449, 'h': 0.536, 'i': 0.302,
'j': 0.276, 'k': 0.515, 'l': 0.277, 'm': 0.809, 'n': 0.58, 'o': 0.488, 'p':
0.551, 'q': 0.507, 'r': 0.426, 's': 0.381, 't': 0.336, 'u': 0.567, 'v': 0.533,
'w': 0.685, 'x': 0.526, 'y': 0.481, 'z': 0.445, '{': 0.306, '|': 0.318, '}':
0.306, '~': 0.649, 'ª': 0.454, 'º': 0.411, 'À': 0.629, 'Á': 0.629, 'Â': 0.629,
'Ã': 0.629, 'Ä': 0.629, 'Ç': 0.67, 'È': 0.61, 'É': 0.61, 'Ê': 0.61, 'Ë': 0.61,
'Ì': 0.341, 'Í': 0.341, 'Î': 0.341, 'Ï': 0.341, 'Ñ': 0.77, 'Ò': 0.729, 'Ó':
0.729, 'Ô': 0.729, 'Õ': 0.729, 'Ö': 0.729, 'Ù': 0.682, 'Ú': 0.682, 'Û': 0.682,
'Ü': 0.682, 'à': 0.523, 'á': 0.523, 'â': 0.523, 'ã': 0.523, 'ä': 0.523, 'ç':
0.432, 'è': 0.429, 'é': 0.429, 'ê': 0.429, 'ë': 0.429, 'ì': 0.302, 'í': 0.302,
'î': 0.302, 'ï': 0.302, 'ñ': 0.58, 'ò': 0.488, 'ó': 0.488, 'ô': 0.488, 'õ':
0.488, 'ö': 0.488, 'ù': 0.567, 'ú': 0.567, 'û': 0.567, 'ü': 0.567
},
'playfair display-medium italic': {
' ': 0.254, '!': 0.28, '"': 0.328, '#': 0.634, '$': 0.571, '%': 0.776, '&':
0.832, "'": 0.201, '(': 0.373, ')': 0.373, '*': 0.491, '+': 0.582, ',': 0.264,
'-': 0.505, '.': 0.256, '/': 0.466, '0': 0.584, '1': 0.352, '2': 0.514, '3':
0.496, '4': 0.522, '5': 0.463, '6': 0.54, '7': 0.492, '8': 0.554, '9': 0.527,
':': 0.289, ';': 0.297, '<': 0.599, '=': 0.659, '>': 0.6, '?': 0.454, '@':
0.853, 'A': 0.637, 'B': 0.648, 'C': 0.678, 'D': 0.74, 'E': 0.615, 'F': 0.573,
'G': 0.716, 'H': 0.773, 'I': 0.357, 'J': 0.652, 'K': 0.697, 'L': 0.592, 'M':
0.884, 'N': 0.788, 'O': 0.745, 'P': 0.607, 'Q': 0.798, 'R': 0.68, 'S': 0.565,
'T': 0.666, 'U': 0.686, 'V': 0.642, 'W': 0.929, 'X': 0.713, 'Y': 0.691, 'Z':
0.654, '[': 0.321, '\\': 0.492, ']': 0.321, '_': 0.744, 'a': 0.535, 'b': 0.518,
'c': 0.442, 'd': 0.554, 'e': 0.442, 'f': 0.328, 'g': 0.457, 'h': 0.547, 'i':
0.309, 'j': 0.284, 'k': 0.527, 'l': 0.285, 'm': 0.825, 'n': 0.588, 'o': 0.502,
'p': 0.562, 'q': 0.522, 'r': 0.441, 's': 0.394, 't': 0.342, 'u': 0.577, 'v':
0.53, 'w': 0.683, 'x': 0.539, 'y': 0.481, 'z': 0.451, '{': 0.324, '|': 0.335,
'}': 0.324, '~': 0.648, 'ª': 0.48, 'º': 0.438, 'À': 0.637, 'Á': 0.637, 'Â':
0.637, 'Ã': 0.637, 'Ä': 0.637, 'Ç': 0.678, 'È': 0.615, 'É': 0.615, 'Ê': 0.615,
'Ë': 0.615, 'Ì': 0.357, 'Í': 0.357, 'Î': 0.357, 'Ï': 0.357, 'Ñ': 0.788, 'Ò':
0.745, 'Ó': 0.745, 'Ô': 0.745, 'Õ': 0.745, 'Ö': 0.745, 'Ù': 0.686, 'Ú': 0.686,
'Û': 0.686, 'Ü': 0.686, 'à': 0.535, 'á': 0.535, 'â': 0.535, 'ã': 0.535, 'ä':
0.535, 'ç': 0.442, 'è': 0.442, 'é': 0.442, 'ê': 0.442, 'ë': 0.442, 'ì': 0.309,
'í': 0.309, 'î': 0.309, 'ï': 0.309, 'ñ': 0.588, 'ò': 0.502, 'ó': 0.502, 'ô':
0.502, 'õ': 0.502, 'ö': 0.502, 'ù': 0.577, 'ú': 0.577, 'û': 0.577, 'ü': 0.577
},
'playfair display-semibold italic': {
' ': 0.255, '!': 0.297, '"': 0.359, '#': 0.63, '$': 0.59, '%': 0.786, '&':
0.848, "'": 0.215, '(': 0.385, ')': 0.385, '*': 0.488, '+': 0.579, ',': 0.275,
'-': 0.505, '.': 0.267, '/': 0.463, '0': 0.603, '1': 0.362, '2': 0.524, '3':
0.507, '4': 0.529, '5': 0.467, '6': 0.553, '7': 0.498, '8': 0.564, '9': 0.541,
':': 0.3, ';': 0.307, '<': 0.596, '=': 0.654, '>': 0.598, '?': 0.47, '@': 0.858,
'A': 0.645, 'B': 0.663, 'C': 0.685, 'D': 0.756, 'E': 0.621, 'F': 0.581, 'G':
0.727, 'H': 0.791, 'I': 0.373, 'J': 0.668, 'K': 0.718, 'L': 0.598, 'M': 0.9,
'N': 0.806, 'O': 0.761, 'P': 0.625, 'Q': 0.815, 'R': 0.695, 'S': 0.576, 'T':
0.673, 'U': 0.69, 'V': 0.651, 'W': 0.955, 'X': 0.727, 'Y': 0.712, 'Z': 0.656,
'[': 0.342, '\\': 0.491, ']': 0.342, '_': 0.744, 'a': 0.547, 'b': 0.532, 'c':
0.453, 'd': 0.566, 'e': 0.456, 'f': 0.335, 'g': 0.464, 'h': 0.558, 'i': 0.316,
'j': 0.293, 'k': 0.54, 'l': 0.293, 'm': 0.84, 'n': 0.596, 'o': 0.516, 'p':
0.573, 'q': 0.537, 'r': 0.455, 's': 0.407, 't': 0.348, 'u': 0.587, 'v': 0.527,
'w': 0.681, 'x': 0.552, 'y': 0.482, 'z': 0.457, '{': 0.343, '|': 0.352, '}':
0.342, '~': 0.647, 'ª': 0.506, 'º': 0.465, 'À': 0.645, 'Á': 0.645, 'Â': 0.645,
'Ã': 0.645, 'Ä': 0.645, 'Ç': 0.685, 'È': 0.621, 'É': 0.621, 'Ê': 0.621, 'Ë':
0.621, 'Ì': 0.373, 'Í': 0.373, 'Î': 0.373, 'Ï': 0.373, 'Ñ': 0.806, 'Ò': 0.761,
'Ó': 0.761, 'Ô': 0.761, 'Õ': 0.761, 'Ö': 0.761, 'Ù': 0.69, 'Ú': 0.69, 'Û': 0.69,
'Ü': 0.69, 'à': 0.547, 'á': 0.547, 'â': 0.547, 'ã': 0.547, 'ä': 0.547, 'ç':
0.453, 'è': 0.456, 'é': 0.456, 'ê': 0.456, 'ë': 0.456, 'ì': 0.315, 'í': 0.315,
'î': 0.315, 'ï': 0.315, 'ñ': 0.596, 'ò': 0.516, 'ó': 0.516, 'ô': 0.516, 'õ':
0.516, 'ö': 0.516, 'ù': 0.587, 'ú': 0.587, 'û': 0.587, 'ü': 0.587
},
'playfair display-bold italic': {
' ': 0.255, '!': 0.314, '"': 0.391, '#': 0.625, '$': 0.61, '%': 0.796, '&':
0.863, "'": 0.228, '(': 0.396, ')': 0.397, '*': 0.484, '+': 0.575, ',': 0.287,
'-': 0.506, '.': 0.279, '/': 0.461, '0': 0.621, '1': 0.371, '2': 0.535, '3':
0.517, '4': 0.535, '5': 0.472, '6': 0.567, '7': 0.503, '8': 0.575, '9': 0.554,
':': 0.311, ';': 0.318, '<': 0.594, '=': 0.65, '>': 0.595, '?': 0.486, '@':
0.863, 'A': 0.653, 'B': 0.679, 'C': 0.693, 'D': 0.771, 'E': 0.626, 'F': 0.588,
'G': 0.739, 'H': 0.81, 'I': 0.39, 'J': 0.683, 'K': 0.74, 'L': 0.605, 'M': 0.917,
'N': 0.825, 'O': 0.777, 'P': 0.643, 'Q': 0.831, 'R': 0.711, 'S': 0.587, 'T':
0.681, 'U': 0.694, 'V': 0.661, 'W': 0.982, 'X': 0.741, 'Y': 0.734, 'Z': 0.658,
'[': 0.362, '\\': 0.491, ']': 0.363, '_': 0.744, 'a': 0.558, 'b': 0.547, 'c':
0.463, 'd': 0.579, 'e': 0.469, 'f': 0.341, 'g': 0.472, 'h': 0.57, 'i': 0.322,
'j': 0.301, 'k': 0.552, 'l': 0.3, 'm': 0.856, 'n': 0.603, 'o': 0.529, 'p':
0.583, 'q': 0.551, 'r': 0.47, 's': 0.42, 't': 0.353, 'u': 0.597, 'v': 0.525,
'w': 0.68, 'x': 0.565, 'y': 0.482, 'z': 0.464, '{': 0.361, '|': 0.369, '}':
0.361, '~': 0.645, 'ª': 0.532, 'º': 0.491, 'À': 0.653, 'Á': 0.653, 'Â': 0.653,
'Ã': 0.653, 'Ä': 0.653, 'Ç': 0.693, 'È': 0.626, 'É': 0.626, 'Ê': 0.626, 'Ë':
0.626, 'Ì': 0.39, 'Í': 0.39, 'Î': 0.39, 'Ï': 0.39, 'Ñ': 0.825, 'Ò': 0.777, 'Ó':
0.777, 'Ô': 0.777, 'Õ': 0.777, 'Ö': 0.777, 'Ù': 0.694, 'Ú': 0.694, 'Û': 0.694,
'Ü': 0.694, 'à': 0.558, 'á': 0.558, 'â': 0.558, 'ã': 0.558, 'ä': 0.558, 'ç':
0.463, 'è': 0.469, 'é': 0.469, 'ê': 0.469, 'ë': 0.469, 'ì': 0.322, 'í': 0.322,
'î': 0.322, 'ï': 0.322, 'ñ': 0.603, 'ò': 0.529, 'ó': 0.529, 'ô': 0.529, 'õ':
0.529, 'ö': 0.529, 'ù': 0.597, 'ú': 0.597, 'û': 0.597, 'ü': 0.597
},
'playfair display-bold': {
' ': 0.233, '!': 0.288, '"': 0.369, '#': 0.62, '$': 0.574, '%': 0.728, '&':
0.898, "'": 0.209, '(': 0.318, ')': 0.318, '*': 0.481, '+': 0.574, ',': 0.271,
'-': 0.481, '.': 0.27, '/': 0.368, '0': 0.645, '1': 0.383, '2': 0.523, '3':
0.488, '4': 0.528, '5': 0.459, '6': 0.561, '7': 0.455, '8': 0.561, '9': 0.553,
':': 0.29, ';': 0.296, '<': 0.602, '=': 0.635, '>': 0.602, '?': 0.508, '@':
0.884, 'A': 0.667, 'B': 0.672, 'C': 0.705, 'D': 0.776, 'E': 0.634, 'F': 0.592,
'G': 0.735, 'H': 0.8, 'I': 0.376, 'J': 0.364, 'K': 0.714, 'L': 0.611, 'M':
0.927, 'N': 0.722, 'O': 0.793, 'P': 0.641, 'Q': 0.793, 'R': 0.692, 'S': 0.584,
'T': 0.668, 'U': 0.689, 'V': 0.668, 'W': 0.968, 'X': 0.691, 'Y': 0.623, 'Z':
0.602, '[': 0.322, '\\': 0.368, ']': 0.322, '_': 0.591, 'a': 0.518, 'b': 0.586,
'c': 0.496, 'd': 0.598, 'e': 0.504, 'f': 0.352, 'g': 0.553, 'h': 0.608, 'i':
0.315, 'j': 0.297, 'k': 0.585, 'l': 0.311, 'm': 0.905, 'n': 0.614, 'o': 0.565,
'p': 0.6, 'q': 0.585, 'r': 0.467, 's': 0.462, 't': 0.352, 'u': 0.606, 'v':
0.508, 'w': 0.787, 'x': 0.536, 'y': 0.519, 'z': 0.474, '{': 0.333, '|': 0.253,
'}': 0.333, '~': 0.668, 'ª': 0.482, 'º': 0.468, 'À': 0.667, 'Á': 0.667, 'Â':
0.667, 'Ã': 0.667, 'Ä': 0.667, 'Ç': 0.705, 'È': 0.634, 'É': 0.634, 'Ê': 0.634,
'Ë': 0.634, 'Ì': 0.376, 'Í': 0.376, 'Î': 0.376, 'Ï': 0.376, 'Ñ': 0.722, 'Ò':
0.793, 'Ó': 0.793, 'Ô': 0.793, 'Õ': 0.793, 'Ö': 0.793, 'Ù': 0.689, 'Ú': 0.689,
'Û': 0.689, 'Ü': 0.689, 'à': 0.518, 'á': 0.518, 'â': 0.518, 'ã': 0.518, 'ä':
0.518, 'ç': 0.496, 'è': 0.504, 'é': 0.504, 'ê': 0.504, 'ë': 0.504, 'ì': 0.315,
'í': 0.315, 'î': 0.315, 'ï': 0.315, 'ñ': 0.614, 'ò': 0.565, 'ó': 0.565, 'ô':
0.565, 'õ': 0.565, 'ö': 0.565, 'ù': 0.606, 'ú': 0.606, 'û': 0.606, 'ü': 0.606
},
}
# Vertical extents, also measured off the font files, as a fraction of the em:
# ``ascent``/``descent`` are the line box Final Cut centres on the title's
# Position when vertical alignment is Middle; the rest are the real INK
# extremes of the glyph classes a line can contain. Horizontal metrics keep
# two words from colliding side by side; these keep two LINES from colliding,
# which cap-height alone could not — a display italic's accents and descenders
# reach far past it (Playfair's ç bottoms out at -0.241 em, its à tops out at
# 1.007), and that is exactly the pair that touched in the reference layout.
VERTICAL_METRICS = {
'helvetica neue': {
'ascent': 0.9520, 'descent': -0.2130,
'cap': 0.7140, 'x_height': 0.5290, 'ascender': 0.7220,
'accent_upper': 0.9470, 'accent_lower': 0.7700, 'descender': -0.2090,
},
'helvetica neue-bold': {
'ascent': 0.9750, 'descent': -0.2170,
'cap': 0.7140, 'x_height': 0.5310, 'ascender': 0.7140,
'accent_upper': 0.9550, 'accent_lower': 0.7850, 'descender': -0.2170,
},
'helvetica neue-italic': {
'ascent': 0.9570, 'descent': -0.2130,
'cap': 0.7140, 'x_height': 0.5290, 'ascender': 0.7220,
'accent_upper': 0.8980, 'accent_lower': 0.7310, 'descender': -0.2090,
},
'helvetica neue-light': {
'ascent': 0.9310, 'descent': -0.2130,
'cap': 0.7140, 'x_height': 0.5261, 'ascender': 0.7140,
'accent_upper': 0.9090, 'accent_lower': 0.7090, 'descender': -0.1880,
},
'helvetica': {
'ascent': 0.7700, 'descent': -0.2300,
'cap': 0.7173, 'x_height': 0.5381, 'ascender': 0.7275,
'accent_upper': 0.9346, 'accent_lower': 0.7334, 'descender': -0.2251,
},
'helvetica-bold': {
'ascent': 0.7700, 'descent': -0.2300,
'cap': 0.7197, 'x_height': 0.5493, 'ascender': 0.7271,
'accent_upper': 0.9453, 'accent_lower': 0.7500, 'descender': -0.2300,
},
'playfair display': {
'ascent': 1.0820, 'descent': -0.2510,
'cap': 0.7080, 'x_height': 0.5290, 'ascender': 0.7840,
'accent_upper': 0.9362, 'accent_lower': 0.8204, 'descender': -0.1880,
},
'playfair display-bold': {
'ascent': 1.0820, 'descent': -0.2510,
'cap': 0.7090, 'x_height': 0.5310, 'ascender': 0.7830,
'accent_upper': 0.9422, 'accent_lower': 0.8301, 'descender': -0.1880,
},
'playfair display-italic': {
'ascent': 1.0820, 'descent': -0.2510,
'cap': 0.8170, 'x_height': 0.5280, 'ascender': 0.7950,
'accent_upper': 0.9365, 'accent_lower': 0.8204, 'descender': -0.1950,
},
'playfair display-medium italic': {
'ascent': 1.0820, 'descent': -0.2510,
'cap': 0.8310, 'x_height': 0.5290, 'ascender': 0.7950,
'accent_upper': 0.9433, 'accent_lower': 0.8276, 'descender': -0.1940,
},
'playfair display-semibold italic': {
'ascent': 1.0820, 'descent': -0.2510,
'cap': 0.8450, 'x_height': 0.5300, 'ascender': 0.7950,
'accent_upper': 0.9492, 'accent_lower': 0.8348, 'descender': -0.1920,
},
'playfair display-bold italic': {
'ascent': 1.0820, 'descent': -0.2510,
'cap': 0.8580, 'x_height': 0.5310, 'ascender': 0.7950,
'accent_upper': 0.9563, 'accent_lower': 0.8427, 'descender': -0.1910,
},
}
+273
View File
@@ -0,0 +1,273 @@
"""Live mode v1 — officially-supported control of a running Final Cut Pro.
This module is the first piece of the dual-mode (XML + Live) architecture
(see docs/CAPABILITY-AUDIT-2026-06.md). Everything here rides Apple's
sanctioned surfaces only — no injection, no private APIs, no accessibility
scripting:
- **push_to_fcp** — programmatic FCPXML *import* via the Open Document
Apple event, Apple's documented zero-click ingestion path. Behavior is
steered by an ``<import-options>`` element injected into the document
(library location, suppress warnings, copy assets).
Live-verified findings (FCP 12.2, 2026-06-11) that shape the contract:
1. **Truly zero-click requires a ``.fcpbundle`` library location.** Given
a new ``.fcpbundle`` path, FCP silently creates the library + a dated
event and imports. With NO location, or a ``.fcplibrary``/bare path,
FCP raises a modal "Open Library" picker — a *required choice* that
``suppress warnings`` does not cover — and the Apple event blocks until
a human answers. ``inject_import_options`` therefore normalises the
location to ``.fcpbundle``.
2. **Media-identity collisions** — importing a project whose media already
exists in the target library fails with "the media already exists with a
unique identifier". Push into a fresh library, or reuse the exact asset
IDs FCP already holds.
- **list_fcp_libraries** — FCP 12's AppleScript dictionary is read-only
library inspection (suite ``com.apple.FinalCut.library.inspection``);
we use it to enumerate open libraries → events → projects.
The asymmetry is structural: import is scriptable, but Apple offers NO
programmatic export — reading back the user's current timeline still
requires a manual File > Export XML. Live mode therefore *pushes*;
round-trips come back through the XML tools.
macOS notes: ``osascript`` targeting Final Cut Pro requires the host
process to hold an Apple Events automation grant (System Settings →
Privacy & Security → Automation) — the first call triggers the consent
prompt. ``tell application "Final Cut Pro"`` launches FCP if it is not
already running; ``list_fcp_libraries`` checks first and declines to
launch, while ``push_to_fcp`` launching FCP is the point.
"""
import subprocess
import xml.etree.ElementTree as ET
from pathlib import Path
from typing import Any, Dict, List, Optional
_OSASCRIPT_TIMEOUT_SECONDS = 120 # FCP cold-launch + import can be slow
_FCP_BUNDLE_ID = "com.apple.FinalCut"
# Field separators for AppleScript list output — ASCII unit/record
# separators cannot appear in user-facing library/project names.
_FIELD_SEP = "\x1f"
_RECORD_SEP = "\x1e"
def _applescript_quote(value: str) -> str:
"""Escape a string for embedding in a double-quoted AppleScript literal."""
return value.replace("\\", "\\\\").replace('"', '\\"')
def _run_osascript(script: str) -> subprocess.CompletedProcess:
return subprocess.run(
["osascript", "-e", script],
capture_output=True,
text=True,
timeout=_OSASCRIPT_TIMEOUT_SECONDS,
)
def fcp_is_running() -> bool:
"""True when a Final Cut Pro process is active (no launch side-effect)."""
proc = subprocess.run(
["pgrep", "-x", "Final Cut Pro"], capture_output=True, text=True
)
return proc.returncode == 0
def inject_import_options(
fcpxml_path: str,
output_path: str,
library_location: Optional[str] = None,
suppress_warnings: bool = True,
copy_assets: Optional[bool] = None,
) -> str:
"""Write a copy of *fcpxml_path* with an ``<import-options>`` element.
The DTD requires ``import-options`` as the FIRST child of ``<fcpxml>``
(``<!ELEMENT fcpxml (import-options?, resources?, ...)>``). Any
existing import-options element is replaced.
Args:
fcpxml_path: Source ``.fcpxml`` document.
output_path: Where to write the import-ready copy.
library_location: Path or ``file://`` URL of the target library
(FCP creates the library if none exists there).
suppress_warnings: Suppress non-fatal import warning dialogs.
copy_assets: True = copy media into the library, False = link
in place, None = let FCP use its default.
Returns:
*output_path*.
"""
from .safe_xml import safe_parse
from .writer import write_fcpxml
tree = safe_parse(fcpxml_path)
root = tree.getroot()
for stale in root.findall('import-options'):
root.remove(stale)
options = ET.Element('import-options')
if library_location:
loc = library_location
if not loc.startswith('file://'):
from urllib.parse import quote
resolved = Path(loc).expanduser()
# Live-verified on FCP 12.2: FCP auto-creates a library ONLY when
# the location carries the .fcpbundle extension. A bare directory
# or a .fcplibrary path triggers the modal "Open Library" picker
# (a required choice that `suppress warnings` does NOT dismiss),
# which blocks the Apple event. Normalise to .fcpbundle.
if resolved.suffix.lower() != '.fcpbundle':
resolved = resolved.with_suffix('.fcpbundle')
loc = 'file://' + quote(str(resolved.resolve()))
ET.SubElement(options, 'option', key='library location', value=loc)
ET.SubElement(
options, 'option',
key='suppress warnings', value='1' if suppress_warnings else '0',
)
if copy_assets is not None:
ET.SubElement(
options, 'option',
key='copy assets', value='1' if copy_assets else '0',
)
root.insert(0, options)
write_fcpxml(root, output_path)
return output_path
def push_to_fcp(
fcpxml_path: str,
library_location: Optional[str] = None,
suppress_warnings: bool = True,
copy_assets: Optional[bool] = None,
import_copy_path: Optional[str] = None,
) -> Dict[str, Any]:
"""Send an FCPXML document to Final Cut Pro via the Open Document event.
This is Apple's documented programmatic-import path: FCP ingests the
document without any clicks, creating libraries/events as directed by
``<import-options>``. Launches FCP when it isn't running.
Args:
fcpxml_path: ``.fcpxml`` file or ``.fcpxmld`` bundle to import.
library_location: Target library path/URL (created if absent).
suppress_warnings: Suppress non-fatal import warning dialogs.
copy_assets: Copy media into the library vs. link in place.
import_copy_path: Where to write the options-injected copy for
flat files (defaults handled by the caller; required when
options are used on a flat file).
Returns:
Dict with ``sent`` (path actually opened), ``launched_fcp``
(whether FCP was started by this call), and ``stdout``.
Raises:
RuntimeError: When osascript fails (most commonly a missing
Automation permission grant for the host process).
"""
path = Path(fcpxml_path)
send_path = path
# Bundles: open directly (option injection inside a copied bundle is
# a v0.10 refinement); flat files get an import-ready copy so the
# user's original is never touched.
if path.suffix.lower() != '.fcpxmld' and import_copy_path:
send_path = Path(
inject_import_options(
str(path),
import_copy_path,
library_location=library_location,
suppress_warnings=suppress_warnings,
copy_assets=copy_assets,
)
)
was_running = fcp_is_running()
script = (
'tell application "Final Cut Pro"\n'
'activate\n'
f'open POSIX file "{_applescript_quote(str(send_path.resolve()))}"\n'
'end tell'
)
proc = _run_osascript(script)
if proc.returncode != 0:
stderr = proc.stderr.strip()
hint = ""
if "-1743" in stderr or "not allowed" in stderr.lower():
hint = (
" — grant Automation permission: System Settings → "
"Privacy & Security → Automation → allow your terminal/MCP "
"host to control Final Cut Pro, then retry"
)
raise RuntimeError(f"osascript failed: {stderr}{hint}")
return {
"sent": str(send_path),
"launched_fcp": not was_running,
"stdout": proc.stdout.strip(),
}
def list_fcp_libraries(allow_launch: bool = False) -> List[Dict[str, Any]]:
"""Enumerate open libraries → events → projects via AppleScript.
Uses FCP 12's read-only scripting dictionary. By default this
refuses to launch FCP (``tell application`` would start it);
pass ``allow_launch=True`` to override.
Returns:
List of ``{name, events: [{name, projects: [str, ...]}, ...]}``.
Raises:
RuntimeError: FCP not running (and *allow_launch* False), or
osascript failure.
"""
if not allow_launch and not fcp_is_running():
raise RuntimeError(
"Final Cut Pro is not running (pass allow_launch=true to start it)"
)
script = (
'set fieldSep to (ASCII character 31)\n'
'set recSep to (ASCII character 30)\n'
'set out to ""\n'
'tell application "Final Cut Pro"\n'
' repeat with lib in libraries\n'
' set libName to name of lib\n'
' repeat with evt in (events of lib)\n'
' set evtName to name of evt\n'
' set projNames to ""\n'
' repeat with proj in (projects of evt)\n'
' set projNames to projNames & (name of proj) & fieldSep\n'
' end repeat\n'
' set out to out & libName & fieldSep & evtName & fieldSep '
'& projNames & recSep\n'
' end repeat\n'
' if (count of events of lib) is 0 then\n'
' set out to out & libName & fieldSep & recSep\n'
' end if\n'
' end repeat\n'
'end tell\n'
'return out'
)
proc = _run_osascript(script)
if proc.returncode != 0:
raise RuntimeError(f"osascript failed: {proc.stderr.strip()}")
libraries: Dict[str, Dict[str, Any]] = {}
for record in proc.stdout.strip().split(_RECORD_SEP):
record = record.strip('\n')
if not record:
continue
fields = record.split(_FIELD_SEP)
lib_name = fields[0]
lib = libraries.setdefault(lib_name, {"name": lib_name, "events": []})
if len(fields) >= 2 and fields[1]:
projects = [p for p in fields[2:] if p]
lib["events"].append({"name": fields[1], "projects": projects})
return list(libraries.values())
+177
View File
@@ -0,0 +1,177 @@
"""Media intelligence — real analysis of source media referenced by timelines.
v0.10 slice 1: audio silence detection via ffmpeg's silencedetect filter.
No new Python dependencies: ffmpeg is invoked as a bounded subprocess
(list-form arguments, validated numeric parameters, hard timeout), and
detection degrades gracefully — ``detect_silence`` returns ``None`` when
ffmpeg is unavailable or the file cannot be analyzed, so callers can fall
back or report instead of crashing.
"""
import logging
import re
import shutil
import subprocess
from pathlib import Path
from typing import List, Optional, Tuple
logger = logging.getLogger(__name__)
# Hard ceiling on a single ffmpeg analysis pass. Decoding audio-only is far
# faster than realtime, so this covers multi-hour media while still bounding
# an adversarial/corrupt file that makes the decoder hang.
PROBE_TIMEOUT_SECONDS = 120
# silencedetect prints times as plain seconds on stderr; starts can be
# slightly negative (encoder priming samples), so allow a leading minus.
_SILENCE_START_RE = re.compile(r"silence_start:\s*(-?\d+(?:\.\d+)?)")
_SILENCE_END_RE = re.compile(r"silence_end:\s*(-?\d+(?:\.\d+)?)")
_DURATION_RE = re.compile(r"Duration:\s*(\d+):(\d\d):(\d\d(?:\.\d+)?)")
def parse_silencedetect_output(
stderr: str, total_duration: Optional[float] = None
) -> List[Tuple[float, float]]:
"""Parse ffmpeg silencedetect stderr into (start, end) ranges in seconds.
A trailing ``silence_start`` with no matching ``silence_end`` (media that
ends silent) is closed at ``total_duration`` when known, otherwise dropped.
"""
ranges: List[Tuple[float, float]] = []
pending: Optional[float] = None
for line in stderr.splitlines():
start_match = _SILENCE_START_RE.search(line)
if start_match:
pending = max(0.0, float(start_match.group(1)))
continue
end_match = _SILENCE_END_RE.search(line)
if end_match and pending is not None:
ranges.append((pending, float(end_match.group(1))))
pending = None
if pending is not None and total_duration is not None and total_duration > pending:
ranges.append((pending, total_duration))
return ranges
def map_silence_to_timeline(
silences: List[Tuple[float, float]],
source_start: float,
clip_duration: float,
timeline_offset: float,
) -> List[Tuple[float, float]]:
"""Map source-time silence ranges onto the timeline.
A clip uses ``[source_start, source_start + clip_duration)`` of its source
media and sits at ``timeline_offset``. Ranges outside the used window are
excluded; ranges overlapping its edges are clamped.
"""
source_end = source_start + clip_duration
mapped: List[Tuple[float, float]] = []
for start, end in silences:
clamped_start = max(start, source_start)
clamped_end = min(end, source_end)
if clamped_end <= clamped_start:
continue
mapped.append((
timeline_offset + (clamped_start - source_start),
timeline_offset + (clamped_end - source_start),
))
return mapped
def detect_beats(
path: str, max_analysis_seconds: float = 1200.0
) -> Optional[dict]:
"""Detect musical beats in an audio file via librosa's beat tracker.
librosa is an optional dependency (``pip install 'fcp-mcp-server[intelligence]'``);
without it, or when the file is missing/unreadable, this returns ``None``
so callers can degrade to a helpful message instead of crashing.
Returns:
``{'bpm': float, 'beats': [seconds, ...]}`` or ``None``.
"""
file_path = Path(path)
if not file_path.is_file():
return None
try:
import librosa
except ImportError:
logger.info("librosa not installed; beat detection unavailable")
return None
try:
# duration cap bounds memory on adversarially long media
y, sr = librosa.load(str(file_path), sr=None, mono=True,
duration=max_analysis_seconds)
tempo, frames = librosa.beat.beat_track(y=y, sr=sr)
beats = librosa.frames_to_time(frames, sr=sr)
except Exception:
logger.warning("librosa beat analysis failed for %s", file_path)
return None
bpm = float(tempo[0] if hasattr(tempo, "__len__") else tempo)
return {"bpm": bpm, "beats": [float(b) for b in beats]}
def media_src_to_path(src: str) -> str:
"""Convert an FCPXML media src (``file://`` URL or plain path) to a filesystem path."""
if src.startswith("file://"):
from urllib.parse import unquote, urlparse
return unquote(urlparse(src).path)
return src
def _parse_total_duration(stderr: str) -> Optional[float]:
match = _DURATION_RE.search(stderr)
if not match:
return None
hours, minutes, seconds = match.groups()
return int(hours) * 3600 + int(minutes) * 60 + float(seconds)
def detect_silence(
path: str, noise_db: float = -30.0, min_duration: float = 0.5
) -> Optional[List[Tuple[float, float]]]:
"""Detect silence in an audio/video file's first audio stream.
Returns (start, end) ranges in source seconds, or ``None`` when the file
is missing, ffmpeg is unavailable, or analysis fails. Raises ``ValueError``
on out-of-bounds parameters (they end up in a subprocess argument, so they
are validated, not trusted).
"""
if not (-120.0 <= noise_db <= 0.0):
raise ValueError(f"noise_db must be between -120 and 0 dB, got {noise_db}")
if not (0 < min_duration <= 3600):
raise ValueError(f"min_duration must be between 0 and 3600 seconds, got {min_duration}")
file_path = Path(path)
if not file_path.is_file():
return None
if shutil.which("ffmpeg") is None:
logger.info("ffmpeg not found on PATH; silence detection unavailable")
return None
try:
result = subprocess.run(
[
"ffmpeg", "-hide_banner", "-nostdin",
"-i", str(file_path),
# -vn: silence detection only needs the audio stream. Without it
# ffmpeg decodes the full video track into the null muxer, which
# blows past PROBE_TIMEOUT_SECONDS on long/high-bitrate files.
"-vn",
"-af", f"silencedetect=noise={float(noise_db)}dB:d={float(min_duration)}",
"-f", "null", "-",
],
capture_output=True,
text=True,
timeout=PROBE_TIMEOUT_SECONDS,
)
except (OSError, subprocess.TimeoutExpired):
logger.warning("ffmpeg silence analysis failed for %s", file_path)
return None
if result.returncode != 0:
return None
return parse_silencedetect_output(
result.stderr, total_duration=_parse_total_duration(result.stderr)
)
+361
View File
@@ -0,0 +1,361 @@
"""Explicit transcription model management (Hex-inspired design).
Adapts the model-management concept from the Hex macOS app to the fcp-mcp-server
stack (Python + MCP), keeping our conventions: rational, allowlisted model
names, lazy optional imports, graceful degradation, and I/O confined to the
model cache directory.
This module is the skeleton of the manager. The catalog and cache primitives
are real; the MCP handlers and wiring into ``transcribe()`` come in a later
phase (see docs/TRANSCRIPTION-MODELS.md).
Model weights are the same Systran/faster-whisper artifacts that
``transcribe.py`` already loads, so install status here matches what
``WhisperModel(model_size, ...)`` would download on its own.
"""
import json
import logging
import re
import threading
from pathlib import Path
from typing import Callable, List, Optional
from .transcribe import ALLOWED_MODELS
logger = logging.getLogger(__name__)
# faster-whisper (CTranslate2) caches HF snapshots under ~/.cache/huggingface/hub
# as models--Systran--faster-whisper-<size>. This is the on-disk truth for
# "is downloaded".
_HF_REPO = "Systran/faster-whisper"
# Default cache root — matches what faster-whisper (HF hub) uses on its own,
# so a default install lines up with anything already on disk.
_DEFAULT_MODELS_DIR = Path.home() / ".cache" / "huggingface" / "hub"
# Config file for the persisted model selection + models dir.
_CONFIG_DIR = Path.home() / ".fcp-mcp-server"
_CONFIG_FILE = _CONFIG_DIR / "config.json"
# Path-traversal guard: only allow [A-Za-z0-9._-] in an internal model name
# (already enforced by ALLOWED_MODELS, but the cache-dir helper is belt + braces).
_SAFE_NAME_RE = re.compile(r"^[\w.-]+$")
# Progress callback signature used across the module.
ProgressCallback = Callable[[float], None]
def get_models_dir() -> Path:
"""The configured models root (falls back to the default HF hub cache)."""
configured = _load_config().get("models_dir")
if configured:
return Path(configured)
return _DEFAULT_MODELS_DIR
def save_models_dir(path: str) -> str:
"""Persist the models root directory. Returns the stored value."""
if not path.strip():
raise ValueError("models_dir cannot be empty")
_CONFIG_DIR.mkdir(parents=True, exist_ok=True)
data = _load_config()
data["models_dir"] = str(Path(path).expanduser())
_write_config(data)
return data["models_dir"]
def _load_config() -> dict:
"""Read the full config JSON (never raises; returns {} on error)."""
try:
return json.loads(_CONFIG_FILE.read_text(encoding="utf-8"))
except (OSError, ValueError):
return {}
def _write_config(data: dict) -> None:
_CONFIG_DIR.mkdir(parents=True, exist_ok=True)
_CONFIG_FILE.write_text(json.dumps(data, indent=2), encoding="utf-8")
def _hf_snapshot_dir(model_size: str) -> Path:
"""Cache folder for a given model size's HF snapshot, under the models root.
``Systran/faster-whisper`` -> ``<root>/models--Systran--faster-whisper-<size>``.
"""
name = "-".join(_HF_REPO.replace("/", "--").split("-")) + f"-{model_size}"
return get_models_dir() / f"models--{name}"
def model_cache_dir(model_size: str) -> Path:
"""The on-disk cache directory for ``model_size`` (as downloaded on disk)."""
if not _SAFE_NAME_RE.match(model_size):
raise ValueError(f"Unsafe model name: {model_size!r}")
return _hf_snapshot_dir(model_size)
def _load_bundled_catalog() -> Optional[list[dict]]:
"""Read the curated catalog from the bundled ``models.json`` (lazy, cached)."""
global _catalog
if _catalog is not None:
return _catalog
path = Path(__file__).with_name("models.json")
try:
with path.open(encoding="utf-8") as fh:
_catalog = json.load(fh)
except (OSError, ValueError):
logger.warning("Failed to load bundled models.json (%s)", path)
_catalog = []
return _catalog
_catalog: Optional[list[dict]] = None
def load_catalog() -> list[dict]:
"""All curated models (or ``[]`` if the bundled file is unreadable)."""
return list(_load_bundled_catalog() or [])
def get_catalog_model(internal_name: str) -> Optional[dict]:
"""The catalog entry whose ``internal_name`` equals ``internal_name``."""
for entry in load_catalog():
if entry.get("internal_name") == internal_name:
return entry
return None
def is_model_downloaded(model_size: str) -> bool:
"""True when the model's cache snapshot exists and isn't an empty dir.
Never raises on I/O; reports ``False`` for missing/unreadable dirs so
callers can offer a download instead of crashing.
"""
try:
d = model_cache_dir(model_size)
except ValueError:
return False
if not d.is_dir():
return False
try:
return any(d.iterdir())
except OSError:
return False
def list_installed_models() -> List[str]:
"""Model sizes present in the cache, filtered to the known allowlist."""
root = get_models_dir()
if not root.is_dir():
return []
installed: List[str] = []
for entry in root.iterdir():
if not entry.is_dir():
continue
for size in ALLOWED_MODELS:
if entry.name == _hf_snapshot_dir(size).name and is_model_downloaded(size):
installed.append(size)
break
# Stable, de-duped order following the allowlist.
return [s for s in ALLOWED_MODELS if s in installed]
def download_model(
model_size: str,
*,
progress_cb: Optional[ProgressCallback] = None,
cancel_event: Optional[threading.Event] = None,
) -> Optional[Path]:
"""Download a model snapshot to the cache.
Validates ``model_size`` against ``ALLOWED_MODELS`` and returns ``None``
(with a logged install hint) when ``huggingface_hub`` is unavailable —
the same graceful-degradation contract as ``transcribe()``.
``cancel_event`` (a ``threading.Event``) aborts the download on the next
progress tick if set; the partially-downloaded snapshot is removed.
"""
if model_size not in ALLOWED_MODELS:
raise ValueError(
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
)
try:
from huggingface_hub import snapshot_download
except ImportError:
logger.info(
"huggingface_hub not installed; install the [transcribe] extra "
"to enable model download"
)
return None
target = model_cache_dir(model_size)
# tqdm hook that both reports progress and honors cancellation.
from tqdm import tqdm
class _CancelableTqdm(tqdm):
def update(self, n=1):
if cancel_event is not None and cancel_event.is_set():
raise _DownloadCancelledError(model_size)
super().update(n)
def _on_progress(current: int, total: int) -> None:
if cancel_event is not None and cancel_event.is_set():
raise _DownloadCancelledError(model_size)
if progress_cb is not None and total > 0:
progress_cb(current / total)
try:
snapshot_download(
repo_id=_HF_REPO,
revision=model_size,
# write into the same snapshot folder faster-whisper expects,
# under the configured models root.
cache_dir=str(get_models_dir()),
local_dir=str(target),
tqdm_class=_CancelableTqdm,
local_dir_use_symlinks=False,
)
# Local snapshot already landed in `target`; simpler than a temp+move.
except _DownloadCancelledError:
logger.info("Download cancelled for model %s", model_size)
_remove_dir(target)
return None
except Exception:
logger.warning("Failed to download model %s", model_size)
_remove_dir(target)
return None
return target
class _DownloadCancelledError(Exception):
"""Internal signal raised to abort a model download."""
def _remove_dir(path: Path) -> None:
import shutil
try:
if path.is_dir():
shutil.rmtree(path, ignore_errors=True)
except OSError:
logger.warning("Failed to clean up partial download %s", path)
def delete_model(model_size: str) -> bool:
"""Remove a model's cache snapshot from disk. Returns True if something was removed."""
try:
d = model_cache_dir(model_size)
except ValueError:
return False
if not d.is_dir():
return False
try:
import shutil
shutil.rmtree(d, ignore_errors=True)
except OSError:
logger.warning("Failed to delete model cache %s", d)
return False
return not d.exists()
def _load_selection() -> str:
"""The persisted selected model size (or ``""`` when absent/invalid)."""
return str(_load_config().get("selected_model", ""))
def save_selected_model(model_size: str) -> str:
"""Persist the selected model size. Validates against ``ALLOWED_MODELS``.
Returns the persisted value so callers can confirm it round-tripped.
"""
if model_size not in ALLOWED_MODELS:
raise ValueError(
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
)
data = _load_config()
data["selected_model"] = model_size
_write_config(data)
return model_size
def load_selected_model() -> str:
"""The effective selected model, with a safe fallback.
- Returns the persisted selection when it's in ``ALLOWED_MODELS``.
- If the persisted model isn't on disk but another is installed, returns that
installed one (never clears the user's persisted value on a false-negative
availability scan, mirroring Hex's rule).
- Otherwise returns ``""`` so callers can fall back to ``"base"``.
"""
selected = _load_selection()
if selected and selected in ALLOWED_MODELS and is_model_downloaded(selected):
return selected
installed = list_installed_models()
if installed:
return installed[0]
return ""
def load_hf_token() -> str:
"""The persisted HuggingFace token for diarization (or ``""``)."""
return str(_load_config().get("hf_token", ""))
def save_hf_token(token: str) -> str:
"""Persist the HuggingFace token used for speaker diarization."""
data = _load_config()
data["hf_token"] = str(token or "").strip()
_write_config(data)
return data["hf_token"]
# Transcription languages, matching the codes used across the app UI.
# ``""`` / ``"auto"`` means "detect automatically".
ALLOWED_LANGUAGES = {
"auto",
"pt",
"en",
"es",
"fr",
"de",
"it",
"nl",
"ja",
"ko",
"zh",
}
def load_transcript_language() -> str:
"""The persisted transcription language (``""``/``"auto"`` = auto-detect)."""
lang = str(_load_config().get("language", ""))
return lang if lang in ALLOWED_LANGUAGES else "auto"
def save_transcript_language(lang: str) -> str:
"""Persist the transcription language. Returns the stored value."""
val = str(lang or "auto").strip().lower()
if val not in ALLOWED_LANGUAGES:
raise ValueError(
f"language must be one of {', '.join(sorted(ALLOWED_LANGUAGES))}, got {lang!r}"
)
data = _load_config()
data["language"] = val
_write_config(data)
return val
def load_num_speakers() -> str:
"""The persisted expected participant count (``""`` = auto-detect)."""
return str(_load_config().get("num_speakers", ""))
def save_num_speakers(num: str) -> str:
"""Persist the expected participant count (empty string = auto)."""
val = str(num or "").strip()
data = _load_config()
data["num_speakers"] = val
_write_config(data)
return val
+66
View File
@@ -0,0 +1,66 @@
[
{
"display_name": "Whisper Tiny",
"internal_name": "tiny",
"language": "Multilingual",
"accuracy": 2,
"speed": 5,
"storage": "76 MB"
},
{
"display_name": "Whisper Tiny EN",
"internal_name": "tiny.en",
"language": "English",
"accuracy": 2,
"speed": 5,
"storage": "76 MB"
},
{
"display_name": "Whisper Base",
"internal_name": "base",
"language": "Multilingual",
"accuracy": 3,
"speed": 4,
"storage": "145 MB"
},
{
"display_name": "Whisper Base EN",
"internal_name": "base.en",
"language": "English",
"accuracy": 3,
"speed": 4,
"storage": "145 MB"
},
{
"display_name": "Whisper Small",
"internal_name": "small",
"language": "Multilingual",
"accuracy": 4,
"speed": 3,
"storage": "484 MB"
},
{
"display_name": "Whisper Medium",
"internal_name": "medium",
"language": "Multilingual",
"accuracy": 5,
"speed": 2,
"storage": "1.53 GB"
},
{
"display_name": "Whisper Large v3",
"internal_name": "large-v3",
"language": "Multilingual",
"accuracy": 5,
"speed": 1,
"storage": "3.08 GB"
},
{
"display_name": "Whisper Distil v3",
"internal_name": "distil-large-v3",
"language": "English",
"accuracy": 5,
"speed": 4,
"storage": "1.5 GB"
}
]
+1073
View File
File diff suppressed because it is too large Load Diff
+367
View File
@@ -0,0 +1,367 @@
"""
FCPXML Parser - Reads Final Cut Pro XML files into Python objects.
"""
import xml.etree.ElementTree as ET
from pathlib import Path
from typing import Any, Dict, Optional
from .models import (
MARKER_XML_TAGS,
Clip,
ConnectedClip,
Keyword,
Marker,
MarkerType,
Project,
Timecode,
Timeline,
TimeValue,
Transition,
)
from .safe_xml import safe_fromstring, safe_parse
# Maximum FCPXML file size (50 MB) — prevents memory exhaustion from crafted files
_MAX_FILE_SIZE_BYTES = 50 * 1024 * 1024
# Tags that represent connected clip elements (includes 'title' for text overlays)
_CONNECTED_CLIP_TAGS = ('asset-clip', 'clip', 'video', 'audio', 'title', 'ref-clip')
class FCPXMLParser:
"""Parser for Final Cut Pro FCPXML files. Supports versions 1.8 - 1.14.
Unknown elements introduced by newer FCPXML versions (e.g. 1.13's
``adjust-stereo-3D`` / ``hidden-clip-marker``, 1.14's smart-collection
search rules) are ignored on read and preserved untouched by the
modify path, which operates on the raw ElementTree.
"""
def __init__(self):
self.resources: Dict[str, Dict[str, Any]] = {}
self.formats: Dict[str, Dict[str, Any]] = {}
self.frame_rate: float = 24.0
def _tc(self, elem: ET.Element, attr: str, default: str = '0s') -> Timecode:
"""Parse a rational time attribute from an XML element.
Centralises the ``Timecode.from_rational(elem.get(attr), frame_rate)``
pattern that repeats across every clip/marker/transition parser.
"""
return Timecode.from_rational(elem.get(attr, default), self.frame_rate)
def parse_file(self, filepath: str) -> Project:
"""Parse an FCPXML file and return a Project object.
Enforces a file size limit to prevent memory exhaustion from
maliciously large XML files.
"""
path = Path(filepath)
if path.suffix == '.fcpxmld':
fcpxml_path = path / 'Info.fcpxml'
if not fcpxml_path.exists():
raise FileNotFoundError(f"Info.fcpxml not found in bundle: {filepath}")
filepath = str(fcpxml_path)
path = Path(filepath)
file_size = path.stat().st_size
if file_size > _MAX_FILE_SIZE_BYTES:
raise ValueError(
f"FCPXML file exceeds maximum size "
f"({file_size / 1024 / 1024:.1f} MB > "
f"{_MAX_FILE_SIZE_BYTES / 1024 / 1024:.0f} MB limit)"
)
tree = safe_parse(filepath)
return self._parse_fcpxml(tree.getroot())
def parse_string(self, xml_string: str) -> Project:
"""Parse FCPXML from a string."""
return self._parse_fcpxml(safe_fromstring(xml_string))
def _parse_fcpxml(self, root: ET.Element) -> Project:
"""Parse the root fcpxml element."""
version = root.get('version', '1.11')
resources_elem = root.find('resources')
if resources_elem is not None:
self._parse_resources(resources_elem)
timelines = []
for library in root.findall('.//library'):
for event in library.findall('event'):
for project in event.findall('project'):
timeline = self._parse_project(project)
if timeline:
timelines.append(timeline)
if not timelines:
for project in root.findall('.//project'):
timeline = self._parse_project(project)
if timeline:
timelines.append(timeline)
project_name = timelines[0].name if timelines else "Untitled"
return Project(name=project_name, timelines=timelines, fcpxml_version=version)
def _parse_resources(self, resources: ET.Element):
"""Parse the resources section."""
for fmt in resources.findall('format'):
fmt_id = fmt.get('id', '')
self.formats[fmt_id] = {
'id': fmt_id, 'name': fmt.get('name', ''),
'width': int(fmt.get('width', 1920)),
'height': int(fmt.get('height', 1080)),
'frameDuration': fmt.get('frameDuration', '1/24s')
}
frame_dur = fmt.get('frameDuration', '1/24s')
if '/' in frame_dur:
parts = frame_dur.rstrip('s').split('/', 1)
num, denom = int(parts[0]), int(parts[1])
if num <= 0:
raise ValueError(f"Invalid frameDuration numerator: {frame_dur}")
if denom <= 0:
raise ValueError(f"Invalid frameDuration denominator: {frame_dur}")
self.frame_rate = denom / num
for asset in resources.findall('asset'):
asset_id = asset.get('id', '')
self.resources[asset_id] = {
'id': asset_id, 'name': asset.get('name', ''),
'src': asset.get('src', '') or (media_rep.get('src', '') if (media_rep := asset.find('media-rep')) is not None else ''),
'start': asset.get('start', '0s'),
'duration': asset.get('duration', '0s'),
'hasVideo': asset.get('hasVideo', '1') == '1',
'hasAudio': asset.get('hasAudio', '1') == '1',
}
def _parse_project(self, project: ET.Element) -> Optional[Timeline]:
"""Parse a project element into a Timeline."""
name = project.get('name', 'Untitled')
sequence = project.find('sequence')
if sequence is None:
return None
format_ref = sequence.get('format', '')
fmt = self.formats.get(format_ref, {})
timeline = Timeline(
name=name,
duration=self._tc(sequence, 'duration'),
frame_rate=self.frame_rate,
width=fmt.get('width', 1920),
height=fmt.get('height', 1080)
)
spine = sequence.find('spine')
if spine is not None:
self._parse_spine(spine, timeline)
timeline.markers.extend(self._collect_markers(sequence))
return timeline
def _parse_spine(self, spine: ET.Element, timeline: Timeline):
"""Parse the spine (primary storyline) including connected clips."""
current_offset = 0
for elem in spine:
tag = elem.tag
if tag in ('asset-clip', 'clip', 'video', 'mc-clip', 'sync-clip', 'ref-clip'):
clip = self._parse_clip(elem, current_offset)
if clip:
timeline.clips.append(clip)
self._parse_connected_clips(elem, clip, timeline)
current_offset += clip.duration.frames
elif tag == 'gap':
gap_frames = self._tc(elem, 'duration').frames
self._parse_gap_connected_clips(elem, current_offset, timeline)
current_offset += gap_frames
elif tag == 'transition':
transition = self._parse_transition(elem, current_offset)
if transition:
timeline.transitions.append(transition)
def _parse_clip(self, elem: ET.Element, offset: int) -> Optional[Clip]:
"""Parse a clip element."""
name = elem.get('name', 'Untitled Clip')
duration = self._tc(elem, 'duration')
source_start = self._tc(elem, 'start')
ref = elem.get('ref', '')
media_path = self.resources.get(ref, {}).get('src', '')
clip = Clip(
name=name,
start=Timecode(frames=offset, frame_rate=self.frame_rate),
duration=duration,
source_start=source_start,
media_path=media_path,
audio_role=elem.get('audioRole', ''),
video_role=elem.get('videoRole', ''),
)
clip.markers.extend(self._collect_markers(elem))
for keyword_elem in elem.findall('keyword'):
keyword = self._parse_keyword(keyword_elem)
if keyword:
clip.keywords.append(keyword)
return clip
def _parse_marker_element(self, elem: ET.Element) -> Optional[Marker]:
"""Parse any marker element (<marker> or <chapter-marker>).
Type detection is delegated to MarkerType.from_xml_element which
owns the completed-attribute semantics. This means the parser
doesn't need separate methods for each tag.
"""
return Marker(
name=elem.get('value', ''),
start=self._tc(elem, 'start'),
duration=self._tc(elem, 'duration', '1/24s'),
marker_type=MarkerType.from_xml_element(elem),
note=elem.get('note', '')
)
def _collect_markers(self, elem: ET.Element) -> list:
"""Collect all markers (standard + chapter) from an element in a single pass.
Iterates children once, selecting recognised marker tags via
MARKER_XML_TAGS rather than making a separate findall per tag.
"""
return [
marker
for child in elem
if child.tag in MARKER_XML_TAGS
for marker in [self._parse_marker_element(child)]
if marker is not None
]
def _parse_keyword(self, elem: ET.Element) -> Optional[Keyword]:
"""Parse a keyword element."""
return Keyword(
value=elem.get('value', ''),
start=self._tc(elem, 'start') if elem.get('start') else None,
duration=self._tc(elem, 'duration') if elem.get('duration') else None,
)
def _parse_transition(self, elem: ET.Element, offset: int) -> Optional[Transition]:
"""Parse a transition element."""
return Transition(
name=elem.get('name', 'Cross Dissolve'),
duration=self._tc(elem, 'duration', '1s'),
start=Timecode(frames=offset, frame_rate=self.frame_rate)
)
def get_library_clips(self, keywords: Optional[list] = None) -> list:
"""
Get all available clips from the library (assets in resources section).
Args:
keywords: Optional list of keywords to filter by
Returns:
List of dicts with asset metadata: name, asset_id, duration_seconds, src
"""
result = []
for asset_id, asset_data in self.resources.items():
# Parse duration to seconds
duration_str = asset_data.get('duration', '0s')
duration_seconds = self._parse_duration_to_seconds(duration_str)
clip_info = {
'asset_id': asset_id,
'name': asset_data.get('name', ''),
'duration_seconds': duration_seconds,
'src': asset_data.get('src', ''),
'has_video': asset_data.get('hasVideo', True),
'has_audio': asset_data.get('hasAudio', True),
}
result.append(clip_info)
# Filter by keywords if provided
if keywords:
# For now, assets don't have keywords directly - return empty if filtering
# In real FCPXML, keywords are typically on clips in events, not assets
return []
return result
def _iter_connected_elements(self, parent_elem: ET.Element, parent_name: str):
"""Yield ``(element, lane, parent_name)`` tuples for connected clips.
Shared iteration logic for both spine-clip and gap-attached connected
clips — walks direct children with a ``lane`` attribute and
``<storyline>`` wrappers, yielding parsed :class:`ConnectedClip`
objects without prescribing where they get stored.
"""
for child in parent_elem:
lane = child.get('lane')
if lane is not None and child.tag in _CONNECTED_CLIP_TAGS:
connected = self._parse_one_connected_clip(
child, int(lane), parent_name)
if connected:
yield connected
elif child.tag == 'storyline':
lane_val = int(child.get('lane', '1'))
for sub_elem in child:
if sub_elem.tag in _CONNECTED_CLIP_TAGS:
connected = self._parse_one_connected_clip(
sub_elem, lane_val, parent_name)
if connected:
yield connected
def _parse_connected_clips(self, parent_elem: ET.Element,
parent_clip: Clip, timeline: Timeline):
"""Parse connected clips attached to a primary storyline clip."""
for connected in self._iter_connected_elements(parent_elem, parent_clip.name):
parent_clip.connected_clips.append(connected)
timeline.connected_clips.append(connected)
def _parse_gap_connected_clips(self, gap_elem: ET.Element,
gap_offset: int, timeline: Timeline):
"""Parse connected clips attached to gap elements."""
for connected in self._iter_connected_elements(gap_elem, f"gap@{gap_offset}"):
timeline.connected_clips.append(connected)
def _parse_one_connected_clip(self, elem: ET.Element, lane: int,
parent_name: str) -> Optional[ConnectedClip]:
"""Parse a single connected clip element."""
name = elem.get('name', 'Untitled')
duration = self._tc(elem, 'duration')
start = self._tc(elem, 'start')
offset = self._tc(elem, 'offset')
ref = elem.get('ref', '')
media_path = self.resources.get(ref, {}).get('src', '')
role = elem.get('audioRole', '') or elem.get('videoRole', '')
connected = ConnectedClip(
name=name, start=start, duration=duration,
lane=lane, offset=offset, source_start=start,
media_path=media_path, clip_type=elem.tag, role=role,
ref_id=ref, parent_clip_name=parent_name,
)
connected.markers.extend(self._collect_markers(elem))
for keyword_elem in elem.findall('keyword'):
keyword = self._parse_keyword(keyword_elem)
if keyword:
connected.keywords.append(keyword)
return connected
def _parse_duration_to_seconds(self, duration_str: str) -> float:
"""Convert FCPXML duration string to seconds.
Delegates to TimeValue.from_timecode() which handles rational
format (``"150/30s"``), plain seconds (``"10s"``), timecode
(``HH:MM:SS:FF``), and frame counts (``"15f"``).
"""
try:
return TimeValue.from_timecode(duration_str).to_seconds()
except (ValueError, ZeroDivisionError):
# Zero-denominator or unparseable → 0.0 (matches prior behaviour)
return 0.0
def parse_fcpxml(filepath: str) -> Project:
"""Convenience function to parse an FCPXML file."""
return FCPXMLParser().parse_file(filepath)
+798
View File
@@ -0,0 +1,798 @@
"""
Auto Rough Cut - AI-powered timeline assembly from source clips.
The flagship feature: select clips by keywords, set a target duration
and pacing style, and get a complete rough cut assembled automatically.
"""
import random
import uuid
import xml.etree.ElementTree as ET
from datetime import datetime
from pathlib import Path
from typing import Any, Dict, List, Optional
from .models import (
MontageConfig,
PacingConfig,
PacingCurve,
RoughCutResult,
SegmentSpec,
TimeValue,
)
from .writer import _create_asset_element, write_fcpxml
class RoughCutGenerator:
"""
Generates rough cuts from source FCPXML files.
Usage:
generator = RoughCutGenerator("source.fcpxml")
result = generator.generate(
output_path="rough_cut.fcpxml",
target_duration="00:03:00:00",
pacing="medium",
segments=[
SegmentSpec(name="Intro", keywords=["intro"], duration_seconds=30),
SegmentSpec(name="Interview", keywords=["interview", "talking"]),
SegmentSpec(name="Outro", keywords=["outro"], duration_seconds=15),
]
)
"""
def __init__(self, source_fcpxml: str):
"""Load source FCPXML for clip selection."""
self.source_path = Path(source_fcpxml)
from .safe_xml import safe_parse
self.tree = safe_parse(source_fcpxml)
self.root = self.tree.getroot()
self.fps = self._detect_fps()
self._index_clips()
self._index_resources()
def _detect_fps(self) -> float:
"""Extract frame rate from format resource."""
for fmt in self.root.findall('.//format'):
frame_dur = fmt.get('frameDuration', '1/30s')
if '/' in frame_dur:
num, denom = frame_dur.replace('s', '').split('/')
return int(denom) / int(num)
return 30.0
def _index_clips(self) -> None:
"""Index all clips with their metadata."""
self.clips: List[Dict[str, Any]] = []
for clip in self.root.findall('.//asset-clip'):
clip_data = self._extract_clip_data(clip, 'asset-clip')
if clip_data:
self.clips.append(clip_data)
for clip in self.root.findall('.//clip'):
clip_data = self._extract_clip_data(clip, 'clip')
if clip_data:
self.clips.append(clip_data)
for video in self.root.findall('.//video'):
clip_data = self._extract_clip_data(video, 'video')
if clip_data:
self.clips.append(clip_data)
def _extract_clip_data(self, elem: ET.Element, clip_type: str) -> Optional[Dict[str, Any]]:
"""Extract clip metadata into a dictionary."""
name = elem.get('name', 'Untitled')
duration_str = elem.get('duration', '0s')
start_str = elem.get('start', '0s')
ref = elem.get('ref', '')
duration = TimeValue.from_timecode(duration_str, self.fps)
# Skip very short clips
if duration.to_seconds() < 0.1:
return None
# Extract keywords
keywords = []
for kw in elem.findall('keyword'):
keywords.append(kw.get('value', ''))
# Check if favorited
is_favorite = elem.get('rating', '') == '1' or elem.get('isFavorite', '') == '1'
is_rejected = elem.get('rating', '') == '-1' or elem.get('isRejected', '') == '1'
return {
'element': elem,
'name': name,
'type': clip_type,
'ref': ref,
'duration': duration,
'start': TimeValue.from_timecode(start_str, self.fps),
'keywords': keywords,
'is_favorite': is_favorite,
'is_rejected': is_rejected,
'used': False, # Track if already used in rough cut
}
def _index_resources(self) -> None:
"""Index all resources (assets, formats)."""
self.resources = {}
self.formats = {}
for asset in self.root.findall('.//asset'):
self.resources[asset.get('id', '')] = asset
for fmt in self.root.findall('.//format'):
self.formats[fmt.get('id', '')] = fmt
def generate(
self,
output_path: str,
target_duration: str = "00:03:00:00",
pacing: str = "medium",
segments: Optional[List[SegmentSpec]] = None,
keywords: Optional[List[str]] = None,
priority: str = "best",
exclude_rejected: bool = True,
favorites_only: bool = False,
add_transitions: bool = False,
transition_duration: str = "00:00:00:15"
) -> RoughCutResult:
"""
Generate a rough cut.
Args:
output_path: Where to save the rough cut FCPXML
target_duration: Target timeline length (timecode or "3m30s")
pacing: "slow", "medium", "fast", or "dynamic"
segments: Optional segment breakdown with keywords per section
keywords: Simple mode - just filter by these keywords
priority: How to select clips - "best", "favorites", "longest", "shortest", "random"
exclude_rejected: Skip rejected clips
favorites_only: Only use favorited clips
add_transitions: Add cross-dissolves between clips
transition_duration: Length of transitions
Returns:
RoughCutResult with output details
"""
target_time = self._parse_duration(target_duration)
pacing_config = PacingConfig(pacing=pacing)
# Filter available clips
available_clips = self._filter_clips(
keywords=keywords,
exclude_rejected=exclude_rejected,
favorites_only=favorites_only
)
if not available_clips:
raise ValueError("No clips match the filter criteria")
# Build the sequence
if segments:
selected_clips = self._select_clips_by_segments(
available_clips, segments, target_time, pacing_config
)
else:
selected_clips = self._select_clips_simple(
available_clips, target_time, pacing_config, priority
)
# Generate output FCPXML
actual_duration = self._build_output(
selected_clips, output_path, add_transitions, transition_duration
)
return RoughCutResult(
output_path=output_path,
clips_used=len(selected_clips),
clips_available=len(available_clips),
target_duration=target_time.to_seconds(),
actual_duration=actual_duration,
segments=len(segments) if segments else 1,
average_clip_duration=actual_duration / len(selected_clips) if selected_clips else 0
)
def _parse_duration(self, duration: str) -> TimeValue:
"""Parse various duration formats."""
# Handle shorthand like "3m30s" or "2m"
if 'm' in duration and ':' not in duration:
parts = duration.lower().replace('s', '').split('m')
minutes = int(parts[0]) if parts[0] else 0
seconds = float(parts[1]) if len(parts) > 1 and parts[1] else 0
total_seconds = minutes * 60 + seconds
return TimeValue.from_seconds(total_seconds, self.fps)
return TimeValue.from_timecode(duration, self.fps)
def _filter_clips(
self,
keywords: Optional[List[str]] = None,
exclude_rejected: bool = True,
favorites_only: bool = False
) -> List[Dict[str, Any]]:
"""Filter clips by criteria."""
result = []
for clip in self.clips:
# Skip rejected
if exclude_rejected and clip['is_rejected']:
continue
# Filter to favorites only
if favorites_only and not clip['is_favorite']:
continue
# Filter by keywords
if keywords:
clip_kws = set(k.lower() for k in clip['keywords'])
filter_kws = set(k.lower() for k in keywords)
if not clip_kws & filter_kws:
continue
result.append(clip)
return result
def _select_clips_simple(
self,
clips: List[Dict[str, Any]],
target_duration: TimeValue,
pacing: PacingConfig,
priority: str
) -> List[Dict[str, Any]]:
"""Select clips to fill target duration."""
selected = []
current_duration = TimeValue.zero()
target_seconds = target_duration.to_seconds()
min_clip, max_clip = pacing.get_duration_range()
# Sort clips by priority
if priority == 'favorites':
clips = sorted(clips, key=lambda c: (not c['is_favorite'], -c['duration'].to_seconds()))
elif priority == 'longest':
clips = sorted(clips, key=lambda c: -c['duration'].to_seconds())
elif priority == 'shortest':
clips = sorted(clips, key=lambda c: c['duration'].to_seconds())
elif priority == 'random':
random.shuffle(clips)
else: # 'best' - mix of favorites first, then by duration
clips = sorted(clips, key=lambda c: (not c['is_favorite'], -c['duration'].to_seconds()))
for clip in clips:
if current_duration.to_seconds() >= target_seconds:
break
remaining = target_seconds - current_duration.to_seconds()
clip_dur = clip['duration'].to_seconds()
# Calculate usable duration for this clip
if clip_dur > max_clip:
use_duration = min(max_clip, remaining)
elif clip_dur < min_clip:
use_duration = clip_dur # Use short clips as-is
else:
use_duration = min(clip_dur, remaining)
if use_duration <= 0:
continue
# Create selection with in/out points
selection = {
**clip,
'use_duration': TimeValue.from_seconds(use_duration, self.fps),
'in_point': clip['start'],
'out_point': clip['start'] + TimeValue.from_seconds(use_duration, self.fps),
}
selected.append(selection)
current_duration = current_duration + TimeValue.from_seconds(use_duration, self.fps)
return selected
def _select_clips_by_segments(
self,
clips: List[Dict[str, Any]],
segments: List[SegmentSpec],
target_duration: TimeValue,
pacing: PacingConfig
) -> List[Dict[str, Any]]:
"""Select clips organized by segment structure."""
selected = []
# Calculate duration per segment
total_specified = sum(s.duration_seconds for s in segments if s.duration_seconds > 0)
unspecified_count = sum(1 for s in segments if s.duration_seconds <= 0)
if unspecified_count > 0:
remaining_duration = target_duration.to_seconds() - total_specified
per_segment = max(0, remaining_duration / unspecified_count)
else:
per_segment = 0
for segment in segments:
seg_duration = segment.duration_seconds if segment.duration_seconds > 0 else per_segment
# Filter clips for this segment
if segment.keywords:
seg_clips = [c for c in clips if
set(k.lower() for k in c['keywords']) &
set(k.lower() for k in segment.keywords)]
else:
seg_clips = clips.copy()
# Remove already used clips
seg_clips = [c for c in seg_clips if not c.get('used_in_rough', False)]
# Select clips for segment
seg_target = TimeValue.from_seconds(seg_duration, self.fps)
seg_selected = self._select_clips_simple(
seg_clips, seg_target, pacing, segment.priority
)
# Mark as used on originals so they're excluded from later segments
used_names = {sel['name'] for sel in seg_selected}
for c in clips:
if c['name'] in used_names:
c['used_in_rough'] = True
for sel in seg_selected:
sel['segment'] = segment.name
selected.extend(seg_selected)
return selected
def _build_output(
self,
clips: List[Dict[str, Any]],
output_path: str,
add_transitions: bool,
transition_duration: str
) -> float:
"""Build the output FCPXML."""
# Create new FCPXML structure
root = ET.Element('fcpxml', version='1.13')
resources = ET.SubElement(root, 'resources')
# Copy relevant resources
format_id = None
for fmt_id, fmt in self.formats.items():
resources.append(fmt)
format_id = fmt_id
break
# Copy needed assets — use media-rep form for new assets
used_refs = set(c.get('ref') for c in clips if c.get('ref'))
for ref in used_refs:
if ref in self.resources:
orig = self.resources[ref]
# Extract src from either attribute or media-rep child
src = orig.get('src', '')
if not src:
mr = orig.find('media-rep')
if mr is not None:
src = mr.get('src', '')
_create_asset_element(
resources,
asset_id=orig.get('id', ref),
name=orig.get('name', ''),
src=src,
duration=orig.get('duration', '0s'),
start=orig.get('start', '0s'),
has_video=orig.get('hasVideo', '1'),
has_audio=orig.get('hasAudio', '1'),
uid=orig.get('uid'),
)
# Create library structure
library = ET.SubElement(root, 'library',
location="file:///Users/editor/Movies/RoughCut.fcpbundle/")
event = ET.SubElement(library, 'event',
name="Rough Cut", uid=str(uuid.uuid4()).upper())
project = ET.SubElement(event, 'project',
name="Rough Cut", uid=str(uuid.uuid4()).upper(),
modDate=datetime.now().strftime("%Y-%m-%d %H:%M:%S -0500"))
# Calculate total duration
total_duration = TimeValue.zero()
for clip in clips:
total_duration = total_duration + clip['use_duration']
# Create sequence
sequence = ET.SubElement(project, 'sequence',
format=format_id or "r1",
duration=total_duration.to_fcpxml(),
tcStart="0s", tcFormat="NDF",
audioLayout="stereo", audioRate="48k")
spine = ET.SubElement(sequence, 'spine')
# Add clips to spine
current_offset = TimeValue.zero()
trans_dur = TimeValue.from_timecode(transition_duration, self.fps) if add_transitions else None
for i, clip in enumerate(clips):
ET.SubElement(spine, 'asset-clip',
ref=clip.get('ref', 'r1'),
offset=current_offset.to_fcpxml(),
name=clip['name'],
start=clip['in_point'].to_fcpxml(),
duration=clip['use_duration'].to_fcpxml(),
format=format_id or "r1",
tcFormat="NDF")
# Add transition before clip (except first)
if add_transitions and i > 0 and trans_dur:
half_trans = trans_dur * 0.5
trans_offset = current_offset - half_trans
transition = ET.Element('transition',
name="Cross Dissolve",
offset=trans_offset.to_fcpxml(),
duration=trans_dur.to_fcpxml())
ET.SubElement(transition, 'filter-video',
name="Cross Dissolve")
spine.insert(-1, transition)
current_offset = current_offset + clip['use_duration']
# Write output
write_fcpxml(root, output_path)
return total_duration.to_seconds()
# ========================================================================
# MONTAGE GENERATION (v0.3.0)
# ========================================================================
def generate_montage(
self,
output_path: str,
target_duration: str,
pacing_curve: str = "accelerating",
start_duration: float = 2.0,
end_duration: float = 0.5,
keywords: Optional[List[str]] = None,
exclude_rejected: bool = True,
add_transitions: bool = False
) -> Dict[str, Any]:
"""
Generate a rapid-fire montage with dynamic pacing curves.
Creates montages where clip duration varies over time:
- accelerating: Starts slow (2s), ends fast (0.5s) - builds energy
- decelerating: Starts fast, ends slow - winds down
- pyramid: Slow → fast → slow - dramatic arc
- constant: Same duration throughout
Args:
output_path: Where to save the montage FCPXML
target_duration: Total montage length (e.g., "30s", "00:00:30:00")
pacing_curve: "accelerating", "decelerating", "pyramid", or "constant"
start_duration: Clip duration at start (seconds)
end_duration: Clip duration at end (seconds)
keywords: Filter clips by keywords
exclude_rejected: Skip rejected clips
add_transitions: Add quick dissolves between clips
Returns:
Dict with output_path, clips_used, actual_duration, pacing_curve
"""
# Parse pacing curve
curve_map = {
'accelerating': PacingCurve.ACCELERATING,
'decelerating': PacingCurve.DECELERATING,
'pyramid': PacingCurve.PYRAMID,
'constant': PacingCurve.CONSTANT
}
curve = curve_map.get(pacing_curve, PacingCurve.ACCELERATING)
config = MontageConfig(
target_duration=self._parse_duration(target_duration).to_seconds(),
pacing_curve=curve,
start_duration=start_duration,
end_duration=end_duration,
min_duration=0.2,
max_duration=max(start_duration, end_duration) + 1.0
)
# Filter available clips
available_clips = self._filter_clips(
keywords=keywords,
exclude_rejected=exclude_rejected
)
if not available_clips:
raise ValueError("No clips match the filter criteria")
# Select clips with pacing curve
selected_clips = self._select_clips_for_montage(available_clips, config)
# Build output
actual_duration = self._build_output(
selected_clips, output_path, add_transitions, "00:00:00:06"
)
return {
'output_path': output_path,
'clips_used': len(selected_clips),
'clips_available': len(available_clips),
'target_duration': config.target_duration,
'actual_duration': actual_duration,
'pacing_curve': pacing_curve,
'start_clip_duration': selected_clips[0]['use_duration'].to_seconds() if selected_clips else 0,
'end_clip_duration': selected_clips[-1]['use_duration'].to_seconds() if selected_clips else 0
}
def _select_clips_for_montage(
self,
clips: List[Dict[str, Any]],
config: MontageConfig
) -> List[Dict[str, Any]]:
"""Select clips with dynamic pacing based on montage config."""
selected = []
current_duration = 0.0
target = config.target_duration
clip_index = 0
# Shuffle clips for variety
available = clips.copy()
random.shuffle(available)
while current_duration < target and clip_index < len(available):
# Calculate position in montage (0.0 to 1.0)
position = current_duration / target if target > 0 else 0
# Get target clip duration for this position
target_clip_duration = config.get_duration_at_position(position)
# Find best matching clip
clip = available[clip_index]
clip_dur = clip['duration'].to_seconds()
# Determine actual duration to use
remaining = target - current_duration
use_duration = min(target_clip_duration, clip_dur, remaining)
use_duration = max(use_duration, config.min_duration)
if use_duration <= 0:
clip_index += 1
continue
# Create selection
selection = {
**clip,
'use_duration': TimeValue.from_seconds(use_duration, self.fps),
'in_point': clip['start'],
'out_point': clip['start'] + TimeValue.from_seconds(use_duration, self.fps),
'montage_position': position
}
selected.append(selection)
current_duration += use_duration
clip_index += 1
return selected
# ========================================================================
# A/B ROLL GENERATION (v0.3.0)
# ========================================================================
def generate_ab_roll(
self,
output_path: str,
target_duration: str,
a_keywords: List[str],
b_keywords: List[str],
a_duration: str = "5s",
b_duration: str = "3s",
start_with: str = "a",
exclude_rejected: bool = True,
add_transitions: bool = True
) -> Dict[str, Any]:
"""
Generate classic documentary-style A/B roll edit.
Alternates between A-roll (main content, usually interviews) and
B-roll (cutaway footage, usually visuals).
Args:
output_path: Where to save the FCPXML
target_duration: Total duration (e.g., "3m", "00:03:00:00")
a_keywords: Keywords for A-roll clips (e.g., ["interview", "talking"])
b_keywords: Keywords for B-roll clips (e.g., ["broll", "cutaway"])
a_duration: How long each A-roll segment should be
b_duration: How long each B-roll cutaway should be
start_with: Start with "a" or "b" roll
exclude_rejected: Skip rejected clips
add_transitions: Add cross-dissolves between clips
Returns:
Dict with output details including a_segments and b_segments counts
"""
target_time = self._parse_duration(target_duration)
a_dur = self._parse_duration(a_duration)
b_dur = self._parse_duration(b_duration)
# Filter A and B clips
a_clips = self._filter_clips(keywords=a_keywords, exclude_rejected=exclude_rejected)
b_clips = self._filter_clips(keywords=b_keywords, exclude_rejected=exclude_rejected)
if not a_clips:
raise ValueError(f"No A-roll clips found with keywords: {a_keywords}")
if not b_clips:
raise ValueError(f"No B-roll clips found with keywords: {b_keywords}")
# Build alternating sequence
selected_clips = self._build_ab_sequence(
a_clips, b_clips, target_time, a_dur, b_dur, start_with
)
# Build output
actual_duration = self._build_output(
selected_clips, output_path, add_transitions, "00:00:00:12"
)
a_count = sum(1 for c in selected_clips if c.get('roll_type') == 'A')
b_count = sum(1 for c in selected_clips if c.get('roll_type') == 'B')
return {
'output_path': output_path,
'clips_used': len(selected_clips),
'a_clips_available': len(a_clips),
'b_clips_available': len(b_clips),
'a_segments': a_count,
'b_segments': b_count,
'target_duration': target_time.to_seconds(),
'actual_duration': actual_duration,
'a_duration_setting': a_duration,
'b_duration_setting': b_duration
}
def _build_ab_sequence(
self,
a_clips: List[Dict[str, Any]],
b_clips: List[Dict[str, Any]],
target_duration: TimeValue,
a_dur: TimeValue,
b_dur: TimeValue,
start_with: str
) -> List[Dict[str, Any]]:
"""Build alternating A/B sequence."""
selected = []
current_duration = TimeValue.zero()
target_seconds = target_duration.to_seconds()
# Track which clips we've used
a_index = 0
b_index = 0
current_roll = start_with.lower()
while current_duration.to_seconds() < target_seconds:
remaining = target_seconds - current_duration.to_seconds()
if current_roll == 'a':
if a_index >= len(a_clips):
a_index = 0 # Loop if needed
clip = a_clips[a_index]
target_dur = min(a_dur.to_seconds(), remaining)
clip_dur = clip['duration'].to_seconds()
use_dur = min(target_dur, clip_dur)
selection = {
**clip,
'use_duration': TimeValue.from_seconds(use_dur, self.fps),
'in_point': clip['start'],
'out_point': clip['start'] + TimeValue.from_seconds(use_dur, self.fps),
'roll_type': 'A'
}
selected.append(selection)
current_duration = current_duration + TimeValue.from_seconds(use_dur, self.fps)
a_index += 1
current_roll = 'b'
else: # B-roll
if b_index >= len(b_clips):
b_index = 0 # Loop if needed
clip = b_clips[b_index]
target_dur = min(b_dur.to_seconds(), remaining)
clip_dur = clip['duration'].to_seconds()
use_dur = min(target_dur, clip_dur)
selection = {
**clip,
'use_duration': TimeValue.from_seconds(use_dur, self.fps),
'in_point': clip['start'],
'out_point': clip['start'] + TimeValue.from_seconds(use_dur, self.fps),
'roll_type': 'B'
}
selected.append(selection)
current_duration = current_duration + TimeValue.from_seconds(use_dur, self.fps)
b_index += 1
current_roll = 'a'
# Safety check to prevent infinite loop
if len(selected) > 1000:
break
return selected
# ============================================================================
# CONVENIENCE FUNCTIONS
# ============================================================================
def generate_rough_cut(
source_fcpxml: str,
output_path: str,
target_duration: str = "3m",
pacing: str = "medium",
keywords: Optional[List[str]] = None,
priority: str = "best"
) -> RoughCutResult:
"""
One-liner rough cut generation.
Example:
result = generate_rough_cut(
"source_clips.fcpxml",
"rough_cut.fcpxml",
target_duration="3m",
pacing="fast",
keywords=["broll", "action"]
)
"""
generator = RoughCutGenerator(source_fcpxml)
return generator.generate(
output_path=output_path,
target_duration=target_duration,
pacing=pacing,
keywords=keywords,
priority=priority
)
def generate_segmented_rough_cut(
source_fcpxml: str,
output_path: str,
segments: List[Dict[str, Any]],
pacing: str = "medium",
add_transitions: bool = True
) -> RoughCutResult:
"""
Generate rough cut with defined segments.
Example:
result = generate_segmented_rough_cut(
"interview.fcpxml",
"rough_cut.fcpxml",
segments=[
{"name": "Intro", "keywords": ["intro"], "duration": 30},
{"name": "Main", "keywords": ["interview"], "duration": 180},
{"name": "Outro", "keywords": ["outro"], "duration": 20},
],
pacing="medium",
add_transitions=True
)
"""
segment_specs = []
for seg in segments:
segment_specs.append(SegmentSpec(
name=seg.get('name', 'Segment'),
keywords=seg.get('keywords', []),
duration_seconds=seg.get('duration', 0),
priority=seg.get('priority', 'best')
))
# Calculate total duration
total = sum(seg.get('duration', 0) for seg in segments)
generator = RoughCutGenerator(source_fcpxml)
return generator.generate(
output_path=output_path,
target_duration=f"{total}s",
pacing=pacing,
segments=segment_specs,
add_transitions=add_transitions
)
+113
View File
@@ -0,0 +1,113 @@
"""
Safe XML parsing — defused against XXE, billion laughs, and entity expansion.
Centralizes all XML parsing so every entry point (parser, writer, export,
rough_cut) uses the same hardened functions. Drop-in replacements for
ET.parse() and ET.fromstring().
Blocks:
- External entity injection (XXE): file:///etc/passwd, http:// callbacks
- Billion laughs / entity expansion: exponential DTD bombs
- DTD retrieval: remote DTD loading
Security note: All flags are set explicitly rather than relying on library
defaults — this ensures protection survives dependency upgrades that might
change default behavior.
IMPORTANT: Never use stdlib xml.etree.ElementTree.parse() or .fromstring()
directly in this codebase. Always import from this module instead.
"""
import xml.etree.ElementTree as ET
from xml.dom.minidom import Document
import defusedxml.ElementTree as _safe_ET
import defusedxml.minidom as _safe_minidom
# Explicit security flags — pinned so a defusedxml upgrade that changes
# defaults cannot silently weaken the boundary.
#
# forbid_dtd is False because FCPXML files legitimately include
# <!DOCTYPE fcpxml> — blocking all DTDs would reject every real FCP export.
# forbid_entities and forbid_external block the dangerous payloads
# (entity expansion bombs, external entity reads, remote DTD fetches).
_SECURITY_FLAGS = {
"forbid_dtd": False,
"forbid_entities": True,
"forbid_external": True,
}
def safe_parse(source: str) -> ET.ElementTree:
"""Parse an XML file with XXE and entity-expansion protection.
All DTD processing, entity definitions, and external references are
rejected outright. Returns a standard ElementTree so downstream code
is unchanged.
"""
return _safe_ET.parse(source, **_SECURITY_FLAGS)
def safe_fromstring(text: str) -> ET.Element:
"""Parse an XML string with XXE and entity-expansion protection.
All DTD processing, entity definitions, and external references are
rejected outright. Returns a standard Element so downstream code is
unchanged.
"""
return _safe_ET.fromstring(text, **_SECURITY_FLAGS)
def safe_parse_string(text: str) -> Document:
"""Parse an XML string into a minidom Document with XXE protection.
Drop-in replacement for xml.dom.minidom.parseString(). Used in the
pretty-print path (writer.py, export.py) to maintain defense-in-depth
even when re-parsing XML that was already produced by safe_parse().
Why this matters: if a future refactor passes unsanitized XML through
serialize_xml(), stdlib minidom would silently process external
entities and DTD bombs. Using defusedxml.minidom closes that gap.
"""
return _safe_minidom.parseString(text)
def serialize_xml(root: ET.Element, filepath: str, doctype: str = "") -> str:
"""Pretty-print an ElementTree root and write to disk.
Single serialization pipeline shared by all XML output paths
(write_fcpxml for FCPXML, DaVinciExporter for XMEML / simplified exports).
Consolidates the ET.tostring → minidom → toprettyxml → strip blanks →
replace declaration → write-to-file sequence that was previously
duplicated across writer.py and export.py.
Args:
root: The XML root Element to serialize.
filepath: Destination file path.
doctype: DOCTYPE declaration to insert after the XML declaration.
Example: ``'<!DOCTYPE fcpxml>'`` or ``'<!DOCTYPE xmeml>'``.
If empty, no DOCTYPE is inserted.
Returns:
The filepath written to.
"""
xml_str = ET.tostring(root, encoding='unicode')
dom = safe_parse_string(xml_str)
pretty_xml = dom.toprettyxml(indent=" ")
lines = [line for line in pretty_xml.split('\n') if line.strip()]
final_xml = '\n'.join(lines)
if doctype:
final_xml = final_xml.replace(
'<?xml version="1.0" ?>',
f'<?xml version="1.0" encoding="UTF-8"?>\n{doctype}',
)
else:
final_xml = final_xml.replace(
'<?xml version="1.0" ?>',
'<?xml version="1.0" encoding="UTF-8"?>',
)
with open(filepath, 'w', encoding='utf-8') as f:
f.write(final_xml)
return filepath
+387
View File
@@ -0,0 +1,387 @@
"""
Template System — Pre-built timeline structures for common editing patterns.
Templates define slot-based layouts that can be filled with clips to generate
complete FCPXML timelines. Each template has named slots (video, audio, title,
gap) with duration constraints and lane assignments.
"""
import uuid
import xml.etree.ElementTree as ET
from dataclasses import dataclass, field
from datetime import datetime
from typing import Any, Dict, List, Optional
from .models import TimeValue
from .writer import (
_create_asset_element,
_sanitize_xml_value,
write_fcpxml,
)
# ============================================================================
# DATA CLASSES
# ============================================================================
@dataclass
class TemplateSlot:
"""A slot in a template that can be filled with a clip.
Attributes:
name: Unique slot name (e.g. "intro", "main", "music_bed").
slot_type: Type of content: "video", "audio", "title", or "gap".
min_duration: Minimum duration in seconds (0 = no minimum).
max_duration: Maximum duration in seconds (0 = no maximum).
default_duration: Default duration if no clip duration specified.
lane: Lane assignment (0 = primary spine, positive = above, negative = below).
role: Audio/video role (e.g. "music", "dialogue", "titles").
required: Whether this slot must be filled.
"""
name: str
slot_type: str = "video"
min_duration: float = 0.0
max_duration: float = 0.0
default_duration: float = 5.0
lane: int = 0
role: str = ""
required: bool = True
@dataclass
class Template:
"""A timeline template with named slots.
Attributes:
name: Template identifier (e.g. "intro_outro").
description: Human-readable description.
slots: Ordered list of template slots.
"""
name: str
description: str
slots: List[TemplateSlot] = field(default_factory=list)
@dataclass
class ClipSpec:
"""Specification for filling a template slot.
Provide either asset_id (for existing assets) or src (for new media).
Attributes:
asset_id: Reference to an existing asset in the FCPXML.
src: File path for new media.
name: Clip display name.
duration: Override duration in seconds (uses slot default if not set).
"""
asset_id: Optional[str] = None
src: Optional[str] = None
name: str = "Untitled"
duration: Optional[float] = None
# ============================================================================
# BUILTIN TEMPLATES
# ============================================================================
BUILTIN_TEMPLATES: Dict[str, Template] = {
"intro_outro": Template(
name="intro_outro",
description=(
"Title card + main content + end card with optional music bed. "
"Classic YouTube/corporate structure."
),
slots=[
TemplateSlot(
name="intro_card", slot_type="video",
default_duration=5.0, max_duration=15.0,
),
TemplateSlot(
name="main_content", slot_type="video",
default_duration=60.0, min_duration=5.0,
),
TemplateSlot(
name="end_card", slot_type="video",
default_duration=5.0, max_duration=15.0,
),
TemplateSlot(
name="music_bed", slot_type="audio",
default_duration=0.0, lane=-1, role="music",
required=False,
),
],
),
"lower_thirds": Template(
name="lower_thirds",
description=(
"Main content with title overlay positions at lane +1. "
"Useful for interview graphics, name supers."
),
slots=[
TemplateSlot(
name="main_content", slot_type="video",
default_duration=60.0, min_duration=5.0,
),
TemplateSlot(
name="lower_third_1", slot_type="title",
default_duration=4.0, max_duration=10.0,
lane=1, role="titles", required=False,
),
TemplateSlot(
name="lower_third_2", slot_type="title",
default_duration=4.0, max_duration=10.0,
lane=1, role="titles", required=False,
),
TemplateSlot(
name="lower_third_3", slot_type="title",
default_duration=4.0, max_duration=10.0,
lane=1, role="titles", required=False,
),
],
),
"music_video": Template(
name="music_video",
description=(
"A/B roll structure with music bed. "
"Alternating performance and cutaway shots over a music track."
),
slots=[
TemplateSlot(
name="a_roll_1", slot_type="video",
default_duration=8.0,
),
TemplateSlot(
name="b_roll_1", slot_type="video",
default_duration=4.0,
),
TemplateSlot(
name="a_roll_2", slot_type="video",
default_duration=8.0,
),
TemplateSlot(
name="b_roll_2", slot_type="video",
default_duration=4.0,
),
TemplateSlot(
name="a_roll_3", slot_type="video",
default_duration=8.0,
),
TemplateSlot(
name="b_roll_3", slot_type="video",
default_duration=4.0,
),
TemplateSlot(
name="music_bed", slot_type="audio",
default_duration=0.0, lane=-1, role="music",
required=False,
),
],
),
}
# ============================================================================
# PUBLIC API
# ============================================================================
def list_templates() -> List[Dict[str, Any]]:
"""Return all available templates with their slot definitions.
Returns:
List of dicts with template name, description, and slot details.
"""
result = []
for name, tmpl in BUILTIN_TEMPLATES.items():
result.append({
'name': tmpl.name,
'description': tmpl.description,
'slots': [
{
'name': s.name,
'slot_type': s.slot_type,
'default_duration': s.default_duration,
'min_duration': s.min_duration,
'max_duration': s.max_duration,
'lane': s.lane,
'role': s.role,
'required': s.required,
}
for s in tmpl.slots
],
})
return result
def apply_template(
template_name: str,
clips_map: Dict[str, ClipSpec],
output_path: str,
fps: float = 24.0,
) -> str:
"""Fill a template with clips and write the resulting FCPXML.
Args:
template_name: Name of a builtin template (e.g. "intro_outro").
clips_map: Dict mapping slot names to ClipSpec objects.
output_path: Where to write the generated FCPXML.
fps: Frame rate for the output timeline.
Returns:
The output file path.
Raises:
ValueError: If template not found or required slots are missing.
"""
template = BUILTIN_TEMPLATES.get(template_name)
if template is None:
available = ', '.join(BUILTIN_TEMPLATES.keys())
raise ValueError(
f"Template '{template_name}' not found. Available: {available}"
)
# Validate required slots
for slot in template.slots:
if slot.required and slot.name not in clips_map:
raise ValueError(
f"Required slot '{slot.name}' not filled in template '{template_name}'."
)
# Build FCPXML
fps_int = int(fps)
root = ET.Element('fcpxml', version='1.13')
resources = ET.SubElement(root, 'resources')
# Format resource
format_id = 'r1'
ET.SubElement(resources, 'format',
id=format_id,
name=f"FFVideoFormat1080p{fps_int}",
frameDuration=f"1/{fps_int}s",
width="1920", height="1080")
# Create assets for each filled slot
asset_map: Dict[str, str] = {} # slot_name → asset_id
asset_counter = 2
for slot_name, spec in clips_map.items():
if spec.asset_id:
asset_map[slot_name] = spec.asset_id
elif spec.src:
aid = f"r{asset_counter}"
asset_counter += 1
slot = next((s for s in template.slots if s.name == slot_name), None)
dur = spec.duration or (slot.default_duration if slot else 5.0)
dur_tv = TimeValue.from_seconds(dur, fps)
_create_asset_element(
resources, aid, spec.name, spec.src,
duration=dur_tv.to_fcpxml(),
has_video="1" if (not slot or slot.slot_type != "audio") else "0",
has_audio="1",
)
asset_map[slot_name] = aid
# Library/event/project structure
library = ET.SubElement(root, 'library',
location="file:///Users/editor/Movies/Template.fcpbundle/")
event = ET.SubElement(library, 'event',
name=template_name, uid=str(uuid.uuid4()).upper())
project_elem = ET.SubElement(event, 'project',
name=template.name,
uid=str(uuid.uuid4()).upper(),
modDate=datetime.now().strftime("%Y-%m-%d %H:%M:%S -0500"))
# Calculate total spine duration (only lane-0 slots)
total_duration = TimeValue.zero()
for slot in template.slots:
if slot.lane != 0:
continue
spec = clips_map.get(slot.name)
if spec:
dur = spec.duration or slot.default_duration
else:
dur = slot.default_duration
total_duration = total_duration + TimeValue.from_seconds(dur, fps)
sequence = ET.SubElement(project_elem, 'sequence',
format=format_id,
duration=total_duration.to_fcpxml(),
tcStart="0s", tcFormat="NDF",
audioLayout="stereo", audioRate="48k")
spine = ET.SubElement(sequence, 'spine')
# Fill spine slots (lane 0) and track connected slots
current_offset = TimeValue.zero()
spine_clips: Dict[str, ET.Element] = {} # slot_name → spine clip element
for slot in template.slots:
if slot.lane != 0:
continue
spec = clips_map.get(slot.name)
dur = (spec.duration if spec and spec.duration else slot.default_duration)
dur_tv = TimeValue.from_seconds(dur, fps)
aid = asset_map.get(slot.name, format_id)
if spec:
clip_elem = ET.SubElement(spine, 'asset-clip',
ref=aid,
offset=current_offset.to_fcpxml(),
name=_sanitize_xml_value(spec.name, 512),
start="0s",
duration=dur_tv.to_fcpxml(),
format=format_id,
tcFormat="NDF")
else:
# Gap placeholder for unfilled optional slot
clip_elem = ET.SubElement(spine, 'gap',
name=slot.name,
offset=current_offset.to_fcpxml(),
duration=dur_tv.to_fcpxml())
spine_clips[slot.name] = clip_elem
current_offset = current_offset + dur_tv
# Fill connected slots (lane != 0) — attach to first spine clip
first_spine_clip = None
for s in template.slots:
if s.lane == 0 and s.name in spine_clips:
first_spine_clip = spine_clips[s.name]
break
for slot in template.slots:
if slot.lane == 0:
continue
spec = clips_map.get(slot.name)
if not spec:
continue
dur = spec.duration or slot.default_duration
dur_tv = TimeValue.from_seconds(dur, fps)
aid = asset_map.get(slot.name, format_id)
parent = first_spine_clip if first_spine_clip is not None else spine
if slot.slot_type == "audio":
# Audio: spans total if duration is 0
if slot.default_duration == 0.0 and spec.duration is None:
dur_tv = total_duration
connected = ET.SubElement(parent, 'asset-clip',
ref=aid,
lane=str(slot.lane),
offset="0s",
name=_sanitize_xml_value(spec.name, 512),
start="0s",
duration=dur_tv.to_fcpxml(),
audioRole=slot.role or "music")
else:
# Title or video overlay
connected = ET.SubElement(parent, 'asset-clip',
ref=aid,
lane=str(slot.lane),
offset="0s",
name=_sanitize_xml_value(spec.name, 512),
start="0s",
duration=dur_tv.to_fcpxml())
if slot.role:
connected.set('videoRole', slot.role)
write_fcpxml(root, output_path)
return output_path
+826
View File
@@ -0,0 +1,826 @@
"""Text measurement and block layout for kinetic-typography subtitles.
Lays a sentence's words out as a compact block — words packed onto lines,
lines stacked and centered — so each word can be emitted as its own positioned
``<title>`` without ever overlapping a sibling.
CALIBRATION
-----------
Every constant here is derived from a real Final Cut export the user built by
hand and sent back ("Exemplo Letra.fcpxmld", project 2160x3840, template
"Essencial - Título"). The five hand-placed words of its first sentence:
word position fontSize kerning
Toda -251.828 -109 170 2.72
a 2.236 -101.135 128 —
minha 267.319 -95 151 2.416
vida, 71.0395 -233.65 128 2.048
assim, 210.88 -100 128 2.048
Three facts fall out of those numbers, and they are why this module can place
words at all:
1. **Position uses the same unit as fontSize.** Center-to-center distances on
line 1 are 254.06 (Toda→a) and 265.08 (a→minha). Half-width sums from the
Helvetica metrics below at those sizes give 245.1 and 265.0 — matching to
within a few units, which is the slop of dragging by hand. A different unit
would have shown up as a constant ratio; there is none.
2. **The canvas is 1080x1920 points** — half of the 2160x3840 frame, because
Final Cut positions in points over 2x media. Line 1 spans -460.3 to +495.7,
which fills that width with small side margins, exactly as the reference
frame looks.
3. **y grows upward.** "vida," (line 2) sits at -233.65 against line 1's ~-101.
Widths are still ESTIMATES — Final Cut renders the real glyphs — so they are
computed generously. Overestimating costs a little empty space; underestimating
makes two words collide, which is the one failure this module exists to
prevent.
"""
from dataclasses import dataclass, field
from typing import Dict, List, Optional, Sequence
from .font_metrics import METRICS, VERTICAL_METRICS
# Fallback advance widths (standard Helvetica AFM), used only for a font the
# embedded metrics do not cover. Real per-variant metrics live in
# font_metrics.METRICS and are preferred — see measure_text.
_HELVETICA_WIDTHS = {
' ': 0.278, '!': 0.278, '"': 0.355, '#': 0.556, '$': 0.556, '%': 0.889,
'&': 0.667, "'": 0.191, '(': 0.333, ')': 0.333, '*': 0.389, '+': 0.584,
',': 0.278, '-': 0.333, '.': 0.278, '/': 0.278, ':': 0.278, ';': 0.278,
'<': 0.584, '=': 0.584, '>': 0.584, '?': 0.556, '@': 1.015, '[': 0.278,
'\\': 0.278, ']': 0.278, '^': 0.469, '_': 0.556, '`': 0.333, '{': 0.334,
'|': 0.260, '}': 0.334, '~': 0.584,
'A': 0.667, 'B': 0.667, 'C': 0.722, 'D': 0.722, 'E': 0.667, 'F': 0.611,
'G': 0.778, 'H': 0.722, 'I': 0.278, 'J': 0.500, 'K': 0.667, 'L': 0.556,
'M': 0.833, 'N': 0.722, 'O': 0.778, 'P': 0.667, 'Q': 0.778, 'R': 0.722,
'S': 0.667, 'T': 0.611, 'U': 0.722, 'V': 0.667, 'W': 0.944, 'X': 0.667,
'Y': 0.667, 'Z': 0.611,
'a': 0.556, 'b': 0.556, 'c': 0.500, 'd': 0.556, 'e': 0.556, 'f': 0.278,
'g': 0.556, 'h': 0.556, 'i': 0.222, 'j': 0.222, 'k': 0.500, 'l': 0.222,
'm': 0.833, 'n': 0.556, 'o': 0.556, 'p': 0.556, 'q': 0.556, 'r': 0.333,
's': 0.500, 't': 0.278, 'u': 0.556, 'v': 0.500, 'w': 0.722, 'x': 0.500,
'y': 0.500, 'z': 0.500,
}
_FALLBACK_WIDTH = 0.556
# Accented Portuguese letters advance like their base letter.
_ACCENT_BASE = str.maketrans(
'áàâãäéèêëíìîïóòôõöúùûüçñÁÀÂÃÄÉÈÊËÍÌÎÏÓÒÔÕÖÚÙÛÜÇÑ',
'aaaaaeeeeiiiiooooouuuucnAAAAAEEEEIIIIOOOOOUUUUCN',
)
# Bold thickens every stem. Only applied on the fallback path — the embedded
# metrics already carry the bold variant's own advances.
_BOLD_FACTOR = 1.06
# Cushion over the computed advance. Small, because the embedded metrics are
# exact: it covers the renderer's own rounding and any glyph outside the table,
# nothing more.
_SAFETY_MARGIN = 1.02
# The frame is 2160x3840 but Final Cut positions in points over 2x media, so
# the coordinate canvas is half that. See the calibration note above.
POINT_SCALE = 0.5
# The canvas the reference sizes were chosen against: 3840px tall at 2x. Other
# formats scale off this, so a 170pt word keeps the same share of frame height
# (8.9%) on a landscape timeline as it has on the user's vertical one.
REFERENCE_CANVAS_HEIGHT = 3840.0 * POINT_SCALE
# Reference values read off the calibration export.
REFERENCE_FONT_SIZE_LARGE = 170
REFERENCE_FONT_SIZE_MEDIUM = 128
REFERENCE_FONT_SIZE_ALT = 151
REFERENCE_KERNING = 2.048
# Center of the hand-placed block: line 1 at y≈-101, line 2 at y≈-233.65.
REFERENCE_BLOCK_CENTER_Y = -167.0
# Gap between words on a line, as a fraction of the larger neighbour's font
# size. The reference export's own gaps work out to ~10-12 points at 170pt
# (0.06), but that leaves only a few points of slack once glyph-metric error is
# accounted for — one bad estimate and two words touch. This is deliberately
# roomier: still a tight typographic block, with margin that survives the
# estimate being off.
REFERENCE_WORD_GAP_RATIO = 0.14
# Text occupies roughly cap-height, not the full em box, so a line's visual
# height is well under its font size. The reference export puts line 1 (max
# 170pt) and line 2 (128pt) 132.65 apart; with this ratio their half-heights
# sum to 111.75, leaving the ~20pt of breathing room below. Using the full em
# box instead would space the lines ~40% further apart than the user did.
_CAP_HEIGHT_RATIO = 0.75
REFERENCE_LINE_GAP = 20.0
# Progressive composition sets its lines much tighter than the word-by-word
# block: in the reference the small grotesque lines almost touch the display
# italic between them. Small but never negative — overlapping boxes is the one
# failure this module exists to prevent.
REFERENCE_BLOCK_LINE_GAP = 8.0
# How far a body line slides toward its side of the emphasis line, as a
# fraction of the slack between the two widths. 1.0 would flush it against the
# emphasis line's edge; the reference leaves a little air.
REFERENCE_STAGGER_RATIO = 0.8
def metrics_for(font: Optional[str], face: Optional[str] = None) -> Optional[Dict]:
"""The embedded advance table for *font*/*face*, or None if uncovered.
Falls back from the exact "family-face" key to the bare family, so an
unknown face still measures against the right family rather than a
generic table.
"""
if not font:
return None
family = font.strip().lower()
if face:
exact = METRICS.get(f"{family}-{face.strip().lower()}")
if exact:
return exact
return METRICS.get(family)
def char_width_ratio(ch: str, table: Optional[Dict] = None) -> float:
"""Return *ch*'s advance width as a fraction of the font size."""
if table is not None and ch in table:
return table[ch]
base = ch.translate(_ACCENT_BASE)
if table is not None and base in table:
return table[base]
return _HELVETICA_WIDTHS.get(base, _FALLBACK_WIDTH)
def measure_text(
text: str,
font_size: float,
*,
bold: bool = False,
kerning: float = REFERENCE_KERNING,
font: Optional[str] = None,
face: Optional[str] = None,
) -> float:
"""Width of *text* rendered at *font_size*, in canvas points.
Measured against the real advance widths of the macOS font when *font*
names one this module carries metrics for, which is the case for every
font the subtitle rhythm uses. Otherwise falls back to generic Helvetica
advances, which is an estimate.
``kerning`` is Final Cut's per-character tracking, in the same unit as
font size — the calibration export carries 2.048 to 2.72.
"""
if not text:
return 0.0
table = metrics_for(font, face)
width = sum(char_width_ratio(ch, table) for ch in text) * font_size
width += kerning * len(text)
if bold and table is None:
# The embedded tables already carry the bold variant's own advances;
# only the generic fallback needs a correction factor.
width *= _BOLD_FACTOR
return width * _SAFETY_MARGIN
# Glyph classes for the vertical ink extent of a line. A line's real top and
# bottom depend on WHICH characters it contains: "sua legenda" reaches the
# x-height and dips to the descender of its g; "que vão" adds the tilde above.
# Measuring the class actually present keeps the stack as tight as the
# reference without ever letting two lines touch.
_ACCENTED_UPPER = set('ÁÀÂÃÄÉÈÊËÍÌÎÏÓÒÔÕÖÚÙÛÜÑÇ')
_ACCENTED_LOWER = set('áàâãäéèêëíìîïóòôõöúùûüñ')
_ASCENDERS = set('bdfhklt')
_DESCENDERS = set('gjpqyçÇ')
_CAPS = set('ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789')
# Used when a font carries no measured vertical metrics: Helvetica's, which
# are typical for a grotesque and tighter than a display serif's, plus a
# cushion so an unmeasured display face still clears its neighbour.
_FALLBACK_VERTICAL = {
'ascent': 0.98, 'descent': -0.25, 'cap': 0.72, 'x_height': 0.53,
'ascender': 0.73, 'accent_upper': 0.95, 'accent_lower': 0.80,
'descender': -0.22,
}
_UNMEASURED_VERTICAL_CUSHION = 1.08
def vertical_metrics_for(font: Optional[str], face: Optional[str] = None) -> tuple:
"""``(metrics, measured)`` for *font*/*face* — the same lookup as metrics_for."""
family = (font or '').strip().lower()
if face:
exact = VERTICAL_METRICS.get(f"{family}-{face.strip().lower()}")
if exact:
return exact, True
table = VERTICAL_METRICS.get(family)
if table:
return table, True
return _FALLBACK_VERTICAL, False
def ink_extent(
text: str,
font_size: float,
*,
font: Optional[str] = None,
face: Optional[str] = None,
) -> tuple:
"""``(top, bottom)`` of the rendered ink, in points around the title's y.
Final Cut centres the LINE BOX — ascent to descent — on the title's
Position when vertical alignment is Middle, so the ink sits off-centre by
however asymmetric the font is. Both values are returned relative to that
centre: positive up, negative down.
"""
metrics, measured = vertical_metrics_for(font, face)
baseline = -(metrics['ascent'] + metrics['descent']) / 2
top = metrics['x_height']
bottom = 0.0
for ch in text:
if ch in _ACCENTED_UPPER:
top = max(top, metrics['accent_upper'])
elif ch in _ACCENTED_LOWER:
top = max(top, metrics['accent_lower'])
elif ch in _CAPS:
top = max(top, metrics['cap'])
elif ch in _ASCENDERS:
top = max(top, metrics['ascender'])
if ch in _DESCENDERS:
bottom = min(bottom, metrics['descender'])
if not measured:
top *= _UNMEASURED_VERTICAL_CUSHION
bottom *= _UNMEASURED_VERTICAL_CUSHION
return (baseline + top) * font_size, (baseline + bottom) * font_size
@dataclass
class PlacedWord:
"""One word positioned inside a laid-out block.
``x``/``y`` are the word's CENTER in canvas points, origin at frame center,
y growing upward — the convention Final Cut's position param uses, and the
value that goes straight into the title's "Posição" param.
"""
text: str
font_size: float
italic: bool
x: float
y: float
width: float
height: float
line_index: int
source: Optional[Dict] = None
look: Optional[object] = None # the WordLook this word was set with
kerning: float = REFERENCE_KERNING # scaled to the frame, as font_size is
@property
def left(self) -> float:
return self.x - self.width / 2
@property
def right(self) -> float:
return self.x + self.width / 2
@property
def bottom(self) -> float:
return self.y - self.height / 2
@property
def top(self) -> float:
return self.y + self.height / 2
def overlaps(self, other: 'PlacedWord') -> bool:
"""True if this word's box intersects *other*'s."""
return (
self.left < other.right
and other.left < self.right
and self.bottom < other.top
and other.bottom < self.top
)
def position_param(self) -> str:
"""The value for the title's "Posição" param, as FCP writes it."""
return f"{self.x:g} {self.y:g}"
# The rest of this block is the interface a placed unit shares with
# PlacedBlock, so the writer emits titles from either without caring
# whether the composition is per word or per phrase.
@property
def words(self) -> List[Dict]:
return [self.source] if self.source else []
@property
def start(self) -> float:
return float(self.source.get('start', 0.0)) if self.source else 0.0
@property
def end(self) -> float:
return float(self.source.get('end', 0.0)) if self.source else 0.0
@property
def font(self) -> str:
return getattr(self.look, 'font', None) or 'Helvetica Neue'
@property
def face(self) -> Optional[str]:
return getattr(self.look, 'face', None)
@property
def color(self) -> str:
return getattr(self.look, 'color', '1 1 1 1')
@dataclass
class BlockLayout:
"""A laid-out group of words, plus whatever did not fit."""
placed: List[PlacedWord] = field(default_factory=list)
overflow: List[Dict] = field(default_factory=list)
@property
def fitted_count(self) -> int:
return len(self.placed)
@dataclass
class LayoutBox:
"""The usable area words may occupy, in canvas points.
Defaults reproduce the calibration export: a block centered slightly below
frame center, spanning most of the width of a 2160x3840 vertical frame.
"""
width: float = 1080.0 * 0.92
height: float = 1920.0 * 0.22
center_y: float = REFERENCE_BLOCK_CENTER_Y
font_scale: float = 1.0
@classmethod
def for_frame(
cls,
frame_width: float,
frame_height: float,
*,
side_margin: float = 0.04,
band_height: float = 0.22,
center_y: Optional[float] = None,
) -> 'LayoutBox':
"""Build a box for a frame of *frame_width* x *frame_height* pixels.
Pixels are converted to canvas points via ``POINT_SCALE``. Type sizes
and the block's height scale with the frame, so the same rhythm reads
proportionally on any format rather than overflowing a shorter one.
``center_y`` defaults to the calibration export's height, scaled.
"""
w = frame_width * POINT_SCALE
h = frame_height * POINT_SCALE
scale = h / REFERENCE_CANVAS_HEIGHT
return cls(
width=w * (1.0 - 2 * side_margin),
height=h * band_height,
center_y=(
REFERENCE_BLOCK_CENTER_Y * scale if center_y is None else center_y
),
font_scale=scale,
)
def look_for(index: int, style):
"""The look for the word at *index* within its sentence.
Prefers ``style.look_for`` (WordStyle's rhythm of WordLook entries, which
carries size, colour, font and face together). Falls back to the older
parallel-pattern attributes so a bare stand-in style still lays out.
"""
resolver = getattr(style, 'look_for', None)
if callable(resolver):
return resolver(index)
class _Fallback:
pass
look = _Fallback()
sizes = getattr(style, 'size_pattern', None)
look.font_size = (
float(sizes[index % len(sizes)]) if sizes
else float(getattr(style, 'font_size', REFERENCE_FONT_SIZE_MEDIUM))
)
italics = getattr(style, 'italic_pattern', None)
look.italic = bool(italics[index % len(italics)]) if italics else False
look.font = getattr(style, 'font', 'Helvetica Neue')
look.face = None
look.color = getattr(style, 'active_color', '1 1 1 1')
look.kerning = float(getattr(style, 'kerning', REFERENCE_KERNING))
return look
def rhythm_font_size(index: int, style) -> float:
"""Font size for the word at *index* within its sentence."""
return float(look_for(index, style).font_size)
def rhythm_italic(index: int, style) -> bool:
"""Whether the word at *index* within its sentence is italic."""
return bool(look_for(index, style).italic)
def layout_sentence(
words: Sequence[Dict],
style,
box: Optional[LayoutBox] = None,
*,
line_gap: float = REFERENCE_LINE_GAP,
word_gap_ratio: float = REFERENCE_WORD_GAP_RATIO,
) -> BlockLayout:
"""Lay *words* out as a centered, line-wrapped block inside *box*.
Words are packed left to right until the line no longer fits ``box.width``,
then a new line opens. Lines are stacked, the stack centered on
``box.center_y``, and every line centered horizontally — a compact block
with no word ever overlapping another.
Words that would push the block past ``box.height`` come back in
``BlockLayout.overflow`` instead of being placed. The caller starts a fresh
block with them, which is what keeps a long sentence from spilling off
screen.
``word_gap_ratio`` sizes the gap between neighbours off the larger of the
two font sizes, so a 170pt word is not separated by the same sliver as a
128pt one.
"""
if box is None:
box = LayoutBox()
bold = bool(getattr(style, 'bold', False))
default_kerning = float(getattr(style, 'kerning', REFERENCE_KERNING))
# Type, spacing and gaps all scale together, or a shorter frame would get
# reference-sized words that never fit.
scale = float(getattr(box, 'font_scale', 1.0)) or 1.0
line_gap *= scale
def gap_between(left: Dict, right: Dict) -> float:
"""Space between two neighbouring words, off the larger of the two."""
return max(left['font_size'], right['font_size']) * word_gap_ratio
tokens = []
for w in words:
text = str(w.get('word', '')).strip()
if not text:
continue
look = look_for(len(tokens), style)
size = float(look.font_size) * scale
kerning = float(getattr(look, 'kerning', default_kerning)) * scale
tokens.append({
'source': w,
'text': text,
'look': look,
'kerning': kerning,
'font_size': size,
'italic': bool(look.italic),
'width': measure_text(
text, size, bold=bold, kerning=kerning,
font=getattr(look, 'font', None) or getattr(style, 'font', None),
face=getattr(look, 'face', None),
),
'height': size * _CAP_HEIGHT_RATIO,
})
if not tokens:
return BlockLayout()
# Pack into lines. A word wider than the whole box still gets its own line
# rather than being dropped — losing a spoken word is worse than one line
# running wide.
lines: List[List[Dict]] = []
current: List[Dict] = []
current_width = 0.0
for tok in tokens:
gap = gap_between(current[-1], tok) if current else 0.0
projected = current_width + gap + tok['width']
if current and projected > box.width:
lines.append(current)
current = [tok]
current_width = tok['width']
else:
current.append(tok)
current_width = projected
if current:
lines.append(current)
# Keep the leading lines that fit the band; the rest overflow into a new
# block. Consecutive lines are half-height + gap + half-height apart, so a
# tall word only costs what it actually occupies.
line_heights = [max(t['height'] for t in line) for line in lines]
kept = 0
total_height = 0.0
for i, h in enumerate(line_heights):
advance = h if not kept else (line_heights[i - 1] + h) / 2 + line_gap
if kept and total_height + advance > box.height:
break
total_height += advance
kept += 1
kept = max(kept, 1) # always place one line, or the caller never advances
result = BlockLayout()
for line in lines[kept:]:
result.overflow.extend(tok['source'] for tok in line)
# Center the stack: the first line's center sits half the total span above
# box.center_y, measuring the span between line CENTERS.
span = sum(
(line_heights[i - 1] + line_heights[i]) / 2 + line_gap
for i in range(1, kept)
)
cursor_y = box.center_y + span / 2
for line_index, line in enumerate(lines[:kept]):
if line_index:
cursor_y -= (
(line_heights[line_index - 1] + line_heights[line_index]) / 2
+ line_gap
)
gaps = [gap_between(a, b) for a, b in zip(line, line[1:])]
line_width = sum(tok['width'] for tok in line) + sum(gaps)
cursor_x = -line_width / 2
for position, tok in enumerate(line):
if position:
cursor_x += gaps[position - 1]
result.placed.append(PlacedWord(
text=tok['text'],
font_size=tok['font_size'],
italic=tok['italic'],
x=cursor_x + tok['width'] / 2,
y=cursor_y,
width=tok['width'],
height=tok['height'],
line_index=line_index,
source=tok['source'],
look=tok['look'],
kerning=tok['kerning'],
))
cursor_x += tok['width']
return result
# ---------------------------------------------------------------------------
# PROGRESSIVE COMPOSITION (block-per-phrase)
# ---------------------------------------------------------------------------
# The look the user asked for (reference: fernandoluz.d reel, 2026-08-17):
#
# [ que vão ]
# [ melhorar ]
# [ sua legenda ]
#
# One title per BLOCK, not per word. Supporting words are set small in a
# grotesque; the sentence's key word is set large in a display italic, on its
# own line. Blocks appear as their first word is spoken and stay on screen, so
# the sentence assembles itself; they all clear together.
# Function words are never the emphasis — "que", "de", "uma" set 2.5x larger
# than the rest reads as a mistake, not as a design.
STOPWORDS_PT = frozenset("""
a as o os um uma uns umas de do da dos das em no na nos nas por para pra pro
com sem sob sobre e ou mas que se ao aos à às pelo pela pelos pelas num numa
eu tu ele ela nos vos eles elas me te lhe nos vos lhes meu minha seu sua teu
tua nosso nossa este esta esse essa aquele aquela isso isto aquilo já não sim
muito mais menos tão como quando onde quem qual quais é foi ser estar tem ter
vai vou vão era são está estão dos aqui ali lá então porque assim
""".split())
def pick_emphasis_index(texts: Sequence[str]) -> int:
"""Index of the word to set as the block's emphasis.
The longest content word, since length is the best proxy available for
"the word this sentence is about" without a language model. Ties break
toward the middle of the sentence, which is where a designer puts the
hero word. A sentence of nothing but function words emphasises its
longest word anyway rather than emphasising nothing.
"""
if not texts:
return 0
cleaned = [t.strip(".,!?;:…\"'()").lower() for t in texts]
middle = (len(texts) - 1) / 2
content = [i for i, t in enumerate(cleaned) if t and t not in STOPWORDS_PT]
pool = content or list(range(len(texts)))
return max(pool, key=lambda i: (len(cleaned[i]), -abs(i - middle)))
@dataclass
class PlacedBlock:
"""One title's worth of text, positioned as a line of the composition."""
text: str
words: List[Dict]
font: str
face: Optional[str]
font_size: float
color: str
kerning: float
x: float
y: float
width: float
height: float
line_index: int
emphasis: bool = False
# Where the rendered ink actually reaches, relative to y (see ink_extent).
ink_top: float = 0.0
ink_bottom: float = 0.0
@property
def left(self) -> float:
return self.x - self.width / 2
@property
def right(self) -> float:
return self.x + self.width / 2
@property
def bottom(self) -> float:
return self.y + self.ink_bottom
@property
def top(self) -> float:
return self.y + self.ink_top
@property
def start(self) -> float:
"""When this block is spoken — its first word's start, in seconds."""
return min(float(w.get('start', 0.0)) for w in self.words)
@property
def end(self) -> float:
"""When this block finishes being spoken, in seconds."""
return max(float(w.get('end', 0.0)) for w in self.words)
def position_param(self) -> str:
"""The value for the title's "Posição" param, as FCP writes it."""
return f"{self.x:g} {self.y:g}"
def overlaps(self, other: 'PlacedBlock') -> bool:
return (
self.left < other.right
and other.left < self.right
and self.bottom < other.top
and other.bottom < self.top
)
@dataclass
class Composition:
"""A laid-out group of blocks, plus whatever did not fit."""
blocks: List[PlacedBlock] = field(default_factory=list)
overflow: List[Dict] = field(default_factory=list)
def compose_sentence(
words: Sequence[Dict],
style,
box: Optional[LayoutBox] = None,
*,
line_gap: float = REFERENCE_BLOCK_LINE_GAP,
stagger_ratio: float = REFERENCE_STAGGER_RATIO,
) -> Composition:
"""Lay a sentence out as stacked blocks, one title per line.
The emphasis word takes a line of its own, set in the display face; the
words before and after it fill the lines above and below, wrapped at
``box.width`` and set in the body face. Body lines are staggered — pushed
toward opposite edges of the emphasis line — which is what makes the
composition read as diagrammed rather than as a centred caption.
Lines that would push the stack past ``box.height`` come back in
``Composition.overflow`` for the caller to place as the next composition.
"""
if box is None:
box = LayoutBox()
scale = float(getattr(box, 'font_scale', 1.0)) or 1.0
entries = [
(w, str(w.get('word', '')).strip())
for w in words
if str(w.get('word', '')).strip()
]
if not entries:
return Composition()
texts = [t for _, t in entries]
emphasis_index = pick_emphasis_index(texts)
emphasis_look = style.look_for_emphasis()
body_look = style.look_for_body()
def measure(text: str, look) -> tuple:
size = float(look.font_size) * scale
kerning = float(getattr(look, 'kerning', REFERENCE_KERNING)) * scale
width = measure_text(
text, size,
bold=bool(getattr(style, 'bold', False)),
kerning=kerning,
font=getattr(look, 'font', None),
face=getattr(look, 'face', None),
)
return size, kerning, width
# Split into lines: everything before the emphasis, the emphasis alone,
# everything after. Body runs wrap at the box width so a long lead-in
# becomes two lines instead of running off frame.
def body_lines(run: List[tuple]) -> List[List[tuple]]:
out: List[List[tuple]] = []
current: List[tuple] = []
for item in run:
trial = current + [item]
text = ' '.join(t for _, t in trial)
if current and measure(text, body_look)[2] > box.width:
out.append(current)
current = [item]
else:
current = trial
if current:
out.append(current)
return out
lines: List[tuple] = [] # (run, look, is_emphasis)
for run in body_lines(entries[:emphasis_index]):
lines.append((run, body_look, False))
lines.append(([entries[emphasis_index]], emphasis_look, True))
for run in body_lines(entries[emphasis_index + 1:]):
lines.append((run, body_look, False))
measured = []
for run, look, is_emphasis in lines:
text = ' '.join(t for _, t in run)
size, kerning, width = measure(text, look)
# Stack on the real ink each line contains, not on a nominal
# cap-height: the display italic's accents and descenders run well
# past it, and a nominal box lets them collide with the neighbour.
top, bottom = ink_extent(
text, size,
font=getattr(look, 'font', None), face=getattr(look, 'face', None),
)
measured.append({
'run': run, 'look': look, 'emphasis': is_emphasis, 'text': text,
'font_size': size, 'kerning': kerning, 'width': width,
'ink_top': top, 'ink_bottom': bottom, 'height': top - bottom,
})
# Keep the leading lines that fit the band; the rest become the next
# composition. The emphasis line must survive — a block of only body text
# loses the whole point of the look — so if it does not fit, everything
# from the emphasis on overflows together.
gap = line_gap * scale
kept = 0
total = 0.0
for line in measured:
advance = line['height'] if not kept else line['height'] + gap
if kept and total + advance > box.height:
break
total += advance
kept += 1
kept = max(kept, 1)
if not any(line['emphasis'] for line in measured[:kept]):
kept = min(kept, next(
i for i, line in enumerate(measured) if line['emphasis']
)) or 1
result = Composition()
for line in measured[kept:]:
result.overflow.extend(w for w, _ in line['run'])
visible = measured[:kept]
# Stack the ink boxes edge to edge with exactly *gap* between them, then
# centre the whole stack on the band. Because the boxes are the real ink,
# "no overlap" is a property of the arithmetic, not of a safety factor.
stack_height = (
sum(line['height'] for line in visible) + gap * (len(visible) - 1)
)
edge = box.center_y + stack_height / 2
# Body lines hang off the emphasis line's edges, alternating sides in
# reading order — the first body line to the left, the next to the right.
anchor = max(line['width'] for line in visible)
side = -1
for index, line in enumerate(visible):
if index:
edge -= gap
cursor_y = edge - line['ink_top']
edge = cursor_y + line['ink_bottom']
if line['emphasis']:
x = 0.0
else:
x = side * (anchor - line['width']) / 2 * stagger_ratio
side = -side
look = line['look']
result.blocks.append(PlacedBlock(
text=line['text'],
words=[w for w, _ in line['run']],
font=getattr(look, 'font', 'Helvetica Neue'),
face=getattr(look, 'face', None),
font_size=line['font_size'],
color=getattr(look, 'color', '1 1 1 1'),
kerning=line['kerning'],
x=x,
y=cursor_y,
width=line['width'],
height=line['height'],
ink_top=line['ink_top'],
ink_bottom=line['ink_bottom'],
line_index=index,
emphasis=line['emphasis'],
))
return result
+288
View File
@@ -0,0 +1,288 @@
"""Transcript intelligence — local Whisper transcription + text-driven editing.
v0.13 slice 1: transcript-based editing. Transcription runs locally via
faster-whisper (optional ``[transcribe]`` extra) with word-level timestamps;
without it, or when media is missing/unreadable, ``transcribe`` returns
``None`` so callers degrade to an install hint instead of crashing — the
same contract as ``media_intel.detect_beats``.
The matching helpers below are pure functions over word lists so they are
fully testable without any model installed, and so tools can accept
pre-computed transcripts (from a previous ``transcribe_media`` run) instead
of re-transcribing.
"""
import logging
import os
import re
from pathlib import Path
from typing import List, Optional, Sequence, Tuple
logger = logging.getLogger(__name__)
# Model names are used to resolve (and download) model weights, so they are
# validated against an allowlist, not trusted.
ALLOWED_MODELS = (
"tiny", "tiny.en", "base", "base.en", "small", "small.en",
"medium", "medium.en", "large-v2", "large-v3", "distil-large-v3",
)
# Conservative by default: interjections that are near-universally filler.
# "like" / "so" / "actually" are speech, not noise, unless the user opts in.
DEFAULT_FILLERS = ("um", "uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm")
_NORM_RE = re.compile(r"[^\w']+")
def normalize_word(word: str) -> str:
"""Lowercase a word and strip punctuation so matching survives Whisper's
tokenization quirks (leading spaces, trailing commas, case)."""
return _NORM_RE.sub("", word.lower())
def find_phrase_spans(
words: Sequence[dict], phrase: str
) -> List[Tuple[float, float]]:
"""Find every occurrence of ``phrase`` in a word-level transcript.
``words`` is a sequence of ``{"word", "start", "end"}`` dicts in source
seconds. Matching is case- and punctuation-insensitive. Returns
``(start, end)`` source-second ranges spanning first to last matched word.
"""
target = [normalize_word(w) for w in phrase.split()]
target = [t for t in target if t]
if not target:
return []
normed = [normalize_word(w.get("word", "")) for w in words]
spans: List[Tuple[float, float]] = []
i = 0
n, m = len(normed), len(target)
while i <= n - m:
if normed[i:i + m] == target:
spans.append((float(words[i]["start"]), float(words[i + m - 1]["end"])))
i += m
else:
i += 1
return spans
def find_filler_spans(
words: Sequence[dict], fillers: Sequence[str] = DEFAULT_FILLERS
) -> List[Tuple[float, float]]:
"""Find filler-word occurrences (single- or multi-word fillers)."""
spans: List[Tuple[float, float]] = []
for filler in fillers:
spans.extend(find_phrase_spans(words, filler))
spans.sort()
return spans
def merge_ranges(
ranges: Sequence[Tuple[float, float]], min_gap: float = 0.0
) -> List[Tuple[float, float]]:
"""Merge overlapping (or nearly touching, within ``min_gap``) ranges."""
if not ranges:
return []
ordered = sorted(ranges)
merged = [list(ordered[0])]
for start, end in ordered[1:]:
if start <= merged[-1][1] + min_gap:
merged[-1][1] = max(merged[-1][1], end)
else:
merged.append([start, end])
return [(s, e) for s, e in merged]
def invert_ranges(
ranges: Sequence[Tuple[float, float]], window_start: float, window_end: float
) -> List[Tuple[float, float]]:
"""Complement of ``ranges`` within ``[window_start, window_end]`` —
turns keep-ranges into cut-ranges for keep_only mode."""
if window_end <= window_start:
return []
kept = merge_ranges(
[(max(s, window_start), min(e, window_end)) for s, e in ranges
if min(e, window_end) > max(s, window_start)]
)
if not kept:
return [(window_start, window_end)]
out: List[Tuple[float, float]] = []
cursor = window_start
for start, end in kept:
if start > cursor:
out.append((cursor, start))
cursor = max(cursor, end)
if cursor < window_end:
out.append((cursor, window_end))
return out
def transcribe(
path: str, model_size: str = "base", language: Optional[str] = None
) -> Optional[dict]:
"""Transcribe an audio/video file locally with word-level timestamps.
Requires the optional ``[transcribe]`` extra (faster-whisper). Returns
``None`` when the model is unavailable or the file is missing/unreadable.
The model weights are resolved from the configured models directory (see
``model_manager.get_models_dir``), so a model selected/downloaded through
the app is found without an implicit download to the default HF cache.
Returns:
``{"language": str, "duration": float, "text": str,
"segments": [{"text", "start", "end", "start_fmt", "end_fmt"}, ...],
"words": [{"word", "start", "end", "confidence"}, ...]}``
"""
if model_size not in ALLOWED_MODELS:
raise ValueError(
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
)
file_path = Path(path)
if not file_path.is_file():
return None
try:
# Resolve the configured models root and point HF at it *before* the
# first faster_whisper import, so downloads/loads go to our folder.
from .model_manager import get_models_dir
models_dir = get_models_dir()
hf_home = models_dir / "hf_home"
hf_home.mkdir(parents=True, exist_ok=True)
os.environ["HF_HOME"] = str(hf_home)
os.environ["HUGGINGFACE_HUB_CACHE"] = str(hf_home / "hub")
from faster_whisper import WhisperModel
except ImportError:
logger.info("faster-whisper not installed; transcription unavailable")
return None
try:
model = WhisperModel(
model_size,
compute_type="int8",
download_root=str(models_dir),
)
segments_iter, info = model.transcribe(
str(file_path),
language=language,
word_timestamps=True,
vad_filter=True,
)
segments: List[dict] = []
words: List[dict] = []
for seg in segments_iter:
start = float(seg.start)
end = float(seg.end)
segments.append(
{
"text": seg.text.strip(),
"start": start,
"end": end,
"start_fmt": format_timestamp(start),
"end_fmt": format_timestamp(end),
}
)
for w in seg.words or []:
ws = float(w.start)
we = float(w.end)
words.append(
{
"word": w.word.strip(),
"start": ws,
"end": we,
"confidence": float(w.probability),
}
)
except Exception:
logger.warning("whisper transcription failed for %s", file_path)
return None
return {
"language": info.language,
"duration": float(info.duration),
"text": " ".join(s["text"] for s in segments),
"segments": segments,
"words": words,
}
def format_timestamp(seconds: float) -> str:
"""Format float seconds as ``HH:MM:SS.mmm`` (e.g. ``00:01:23.450``)."""
total_ms = max(0, round((seconds or 0.0) * 1000))
hours, rem_ms = divmod(total_ms, 3_600_000)
minutes, rem_ms = divmod(rem_ms, 60_000)
secs, ms = divmod(rem_ms, 1000)
return f"{hours:02d}:{minutes:02d}:{secs:02d}.{ms:03d}"
def group_words_by_segment(
words: Sequence[dict],
segments: Sequence[dict],
) -> List[List[dict]]:
"""Group flat *words* into sentences using *segments*' time windows.
``transcribe()`` returns ``segments`` and ``words`` as sibling flat lists —
the words are flattened out of the segments and the association is lost in
serialisation, leaving only the time ranges to rejoin them by. A word
belongs to the segment whose ``[start, end)`` contains its ``start``.
Words falling in no segment (rounding at a boundary, or a gap between
segments) attach to the group being built rather than being dropped —
losing a spoken word would silently drop it from the subtitles.
With no usable *segments*, every word comes back as a single group, which
the caller can still split on its own terms.
"""
kept = [
s for s in (segments or [])
if s.get('start') is not None and s.get('end') is not None
]
ordered = sorted(kept, key=lambda s: float(s['start']))
if not ordered:
return [list(words)] if words else []
groups: List[List[dict]] = []
current: List[dict] = []
current_index: Optional[int] = None
cursor = 0
for w in words:
start = float(w.get('start', 0.0))
# Segments and words are both chronological, so the search only ever
# moves forward.
while (
cursor + 1 < len(ordered)
and start >= float(ordered[cursor]['end'])
and start >= float(ordered[cursor + 1]['start'])
):
cursor += 1
seg = ordered[cursor]
inside = float(seg['start']) <= start < float(seg['end'])
index = cursor if inside else current_index
if current and index != current_index and inside:
groups.append(current)
current = []
if inside or current_index is None:
current_index = index if index is not None else cursor
current.append(w)
if current:
groups.append(current)
return groups
def segments_to_srt(segments: Sequence[dict]) -> str:
"""Render transcript segments as an SRT string (for captions import)."""
def stamp(seconds: float) -> str:
ms = int(round(seconds * 1000))
h, rem = divmod(ms, 3600000)
m, rem = divmod(rem, 60000)
s, ms = divmod(rem, 1000)
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
blocks = []
for i, seg in enumerate(segments, 1):
blocks.append(f"{i}\n{stamp(seg['start'])} --> {stamp(seg['end'])}\n{seg['text']}\n")
return "\n".join(blocks)
+3822
View File
File diff suppressed because it is too large Load Diff