chore: adiciona .gitignore e commit.command
This commit is contained in:
Executable
+139
@@ -0,0 +1,139 @@
|
||||
"""
|
||||
FCPXML - Python library for reading, writing, and modifying Final Cut Pro XML files.
|
||||
|
||||
This package provides tools to:
|
||||
- Parse FCPXML files into Python objects
|
||||
- Modify existing FCPXML files (add markers, trim clips, reorder, etc.)
|
||||
- Generate new FCPXML files from scratch
|
||||
- Create AI-powered rough cuts from source material
|
||||
- Compare timelines and export to other NLE formats
|
||||
"""
|
||||
|
||||
from .diff import ClipDiff, MarkerDiff, TimelineDiff, compare_timelines
|
||||
from .export import DaVinciExporter
|
||||
from .models import (
|
||||
MARKER_XML_TAGS,
|
||||
AudioClip,
|
||||
Clip,
|
||||
CompoundClip,
|
||||
ConnectedClip,
|
||||
# Core models
|
||||
Keyword,
|
||||
Marker,
|
||||
MarkerColor,
|
||||
# Enums
|
||||
MarkerType,
|
||||
PacingConfig,
|
||||
PacingStyle,
|
||||
Project,
|
||||
RoughCutResult,
|
||||
# Rough cut models
|
||||
SegmentSpec,
|
||||
SilenceCandidate,
|
||||
Timecode,
|
||||
Timeline,
|
||||
# Time handling
|
||||
TimeValue,
|
||||
Transition,
|
||||
TransitionType,
|
||||
VideoClip,
|
||||
)
|
||||
from .parser import FCPXMLParser, parse_fcpxml
|
||||
from .rough_cut import (
|
||||
RoughCutGenerator,
|
||||
generate_rough_cut,
|
||||
generate_segmented_rough_cut,
|
||||
)
|
||||
from .templates import (
|
||||
BUILTIN_TEMPLATES,
|
||||
ClipSpec,
|
||||
Template,
|
||||
TemplateSlot,
|
||||
apply_template,
|
||||
list_templates,
|
||||
)
|
||||
from .writer import (
|
||||
FCP_EFFECTS,
|
||||
FCPXMLModifier,
|
||||
FCPXMLWriter,
|
||||
add_marker_to_file,
|
||||
build_marker_element,
|
||||
list_effects,
|
||||
modify_fcpxml,
|
||||
trim_clip_in_file,
|
||||
validate_fcpxml,
|
||||
write_fcpxml,
|
||||
)
|
||||
|
||||
__version__ = "0.6.35"
|
||||
__author__ = "DareDev256"
|
||||
|
||||
__all__ = [
|
||||
# Version
|
||||
"__version__",
|
||||
|
||||
# Enums & constants
|
||||
"MarkerType",
|
||||
"MarkerColor",
|
||||
"MARKER_XML_TAGS",
|
||||
"TransitionType",
|
||||
"PacingStyle",
|
||||
|
||||
# Time
|
||||
"TimeValue",
|
||||
"Timecode",
|
||||
|
||||
# Models
|
||||
"Keyword",
|
||||
"Marker",
|
||||
"Clip",
|
||||
"AudioClip",
|
||||
"VideoClip",
|
||||
"ConnectedClip",
|
||||
"CompoundClip",
|
||||
"SilenceCandidate",
|
||||
"Transition",
|
||||
"Timeline",
|
||||
"Project",
|
||||
"SegmentSpec",
|
||||
"PacingConfig",
|
||||
"RoughCutResult",
|
||||
|
||||
# Parser
|
||||
"FCPXMLParser",
|
||||
"parse_fcpxml",
|
||||
|
||||
# Writer
|
||||
"FCPXMLWriter",
|
||||
"FCPXMLModifier",
|
||||
"modify_fcpxml",
|
||||
"add_marker_to_file",
|
||||
"trim_clip_in_file",
|
||||
"write_fcpxml",
|
||||
"build_marker_element",
|
||||
"validate_fcpxml",
|
||||
"FCP_EFFECTS",
|
||||
"list_effects",
|
||||
|
||||
# Templates (v0.6.0)
|
||||
"Template",
|
||||
"TemplateSlot",
|
||||
"ClipSpec",
|
||||
"BUILTIN_TEMPLATES",
|
||||
"list_templates",
|
||||
"apply_template",
|
||||
|
||||
# Rough Cut
|
||||
"RoughCutGenerator",
|
||||
"generate_rough_cut",
|
||||
"generate_segmented_rough_cut",
|
||||
|
||||
# Diff
|
||||
"compare_timelines",
|
||||
"TimelineDiff",
|
||||
"ClipDiff",
|
||||
"MarkerDiff",
|
||||
|
||||
# Export
|
||||
"DaVinciExporter",
|
||||
]
|
||||
@@ -0,0 +1,150 @@
|
||||
"""Speaker diarization for local transcripts — WHISPERX-inspired, lazy + graceful.
|
||||
|
||||
pyannote.audio (with an HF token granted access to the gated diarization
|
||||
models) is optional: when it is absent or the token is missing, ``diarize``
|
||||
returns ``None`` and callers fall back to ``SPEAKER_00`` for every word —
|
||||
never crashing. The speaker-assignment helpers are pure functions over
|
||||
``(start, end)`` tracks, so they are fully testable without any model.
|
||||
|
||||
This mirrors the reference WHISPERX app (``app._assign_speakers_to_words``),
|
||||
adapted to our flat ``segments`` + ``words`` transcript shape.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from typing import List, Optional, Sequence, Tuple
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_SPEAKER = "SPEAKER_00"
|
||||
|
||||
# Raw speaker id -> canonical ``SPEAKER_NN`` label.
|
||||
_DEFAULT_LABELS = ("SPEAKER_00", "SPEAKER_01", "SPEAKER_02", "SPEAKER_03")
|
||||
|
||||
|
||||
def diarization_capability(token: Optional[str]) -> Tuple[bool, str]:
|
||||
"""Whether speaker identification is available.
|
||||
|
||||
Returns ``(ok, message)``. ``ok`` is only ``True`` when pyannote.audio is
|
||||
installed *and* a token is configured (both are needed for the gated
|
||||
models).
|
||||
"""
|
||||
try:
|
||||
import pyannote.audio # noqa: F401
|
||||
except Exception:
|
||||
return False, "Diarização indisponível: componente pyannote.audio ausente."
|
||||
if not (token or "").strip():
|
||||
return False, "Diarização indisponível: nenhum token HuggingFace configurado."
|
||||
return True, "Identificação de participantes disponível."
|
||||
|
||||
|
||||
def diarize(
|
||||
path: str,
|
||||
token: Optional[str],
|
||||
num_speakers: str = "",
|
||||
) -> Optional[List[Tuple[float, float, str]]]:
|
||||
"""Run speaker diarization on ``path``.
|
||||
|
||||
Returns a list of ``(start, end, raw_speaker_id)`` turns, or ``None`` when
|
||||
pyannote is unavailable, the token is missing/invalid, or analysis fails —
|
||||
the graceful-degradation contract shared with ``transcribe``.
|
||||
"""
|
||||
if not diarization_capability(token)[0]:
|
||||
return None
|
||||
try:
|
||||
from pyannote.audio import Pipeline
|
||||
|
||||
pipe = None
|
||||
try:
|
||||
pipe = Pipeline.from_pretrained("pyannote/speaker-diarization-3.1", token=token)
|
||||
except TypeError:
|
||||
# pyannote.audio < 4.0 used use_auth_token instead of token.
|
||||
pipe = Pipeline.from_pretrained(
|
||||
"pyannote/speaker-diarization-3.1", use_auth_token=token
|
||||
)
|
||||
if pipe is None:
|
||||
return None
|
||||
kwargs = {}
|
||||
n = str(num_speakers or "").strip()
|
||||
if n.isdigit() and int(n) > 0:
|
||||
kwargs["num_speakers"] = int(n)
|
||||
result = pipe(path, **kwargs)
|
||||
# pyannote.audio >= 4.0 wraps the annotation; normalize to the raw one.
|
||||
if hasattr(result, "exclusive_speaker_diarization"):
|
||||
result = result.exclusive_speaker_diarization
|
||||
elif hasattr(result, "speaker_diarization"):
|
||||
result = result.speaker_diarization
|
||||
tracks: List[Tuple[float, float, str]] = []
|
||||
for turn, _, speaker in result.itertracks(yield_label=True):
|
||||
tracks.append((float(turn.start), float(turn.end), str(speaker)))
|
||||
return tracks or None
|
||||
except Exception:
|
||||
logger.warning("diarization failed for %s", path)
|
||||
return None
|
||||
|
||||
|
||||
def _overlap_speaker(
|
||||
start: float, end: float, tracks: Sequence[Tuple[float, float, str]]
|
||||
) -> Optional[str]:
|
||||
"""Speaker with the largest summed time-overlap with ``[start, end]``."""
|
||||
overlap: dict[str, float] = {}
|
||||
for t_start, t_end, sid in tracks:
|
||||
o = min(end, t_end) - max(start, t_start)
|
||||
if o > 0:
|
||||
overlap[sid] = overlap.get(sid, 0.0) + o
|
||||
if not overlap:
|
||||
return None
|
||||
return max(overlap.items(), key=lambda kv: kv[1])[0]
|
||||
|
||||
|
||||
def assign_speakers(
|
||||
segments: Sequence[dict],
|
||||
words: Sequence[dict],
|
||||
tracks: Optional[Sequence[Tuple[float, float, str]]],
|
||||
default: str = DEFAULT_SPEAKER,
|
||||
) -> Tuple[List[dict], List[dict]]:
|
||||
"""Attach ``speaker_id`` to every segment and word.
|
||||
|
||||
``tracks`` is the output of :func:`diarize` (may be ``None``). Raw speaker
|
||||
ids are mapped to stable ``SPEAKER_NN`` labels in first-seen order. Without
|
||||
tracks everything is assigned ``default``.
|
||||
"""
|
||||
norm_tracks: List[Tuple[float, float, str]] = []
|
||||
if tracks:
|
||||
label_map: dict[str, str] = {}
|
||||
counter = 0
|
||||
for t_start, t_end, raw in tracks:
|
||||
if raw not in label_map:
|
||||
label = _DEFAULT_LABELS[counter] if counter < len(_DEFAULT_LABELS) else f"SPEAKER_{counter:02d}"
|
||||
label_map[raw] = label
|
||||
counter += 1
|
||||
norm_tracks.append((t_start, t_end, label_map[raw]))
|
||||
|
||||
out_segments: List[dict] = []
|
||||
for seg in segments:
|
||||
s = dict(seg)
|
||||
s["speaker_id"] = (
|
||||
_overlap_speaker(seg.get("start", 0.0), seg.get("end", 0.0), norm_tracks) or default
|
||||
)
|
||||
out_segments.append(s)
|
||||
|
||||
out_words: List[dict] = []
|
||||
for w in words:
|
||||
ww = dict(w)
|
||||
ww["speaker_id"] = (
|
||||
_overlap_speaker(w.get("start", 0.0), w.get("end", 0.0), norm_tracks) or default
|
||||
)
|
||||
out_words.append(ww)
|
||||
|
||||
return out_segments, out_words
|
||||
|
||||
|
||||
def build_speakers(segments: Sequence[dict]) -> List[dict]:
|
||||
"""Ordered ``[{"id", "name"}, ...]`` from the speakers present in segments."""
|
||||
ids: List[str] = []
|
||||
seen = set()
|
||||
for seg in segments:
|
||||
sid = seg.get("speaker_id", DEFAULT_SPEAKER)
|
||||
if sid not in seen:
|
||||
seen.add(sid)
|
||||
ids.append(sid)
|
||||
return [{"id": sid, "name": f"Speaker {i + 1}"} for i, sid in enumerate(ids)]
|
||||
Executable
+269
@@ -0,0 +1,269 @@
|
||||
"""Timeline comparison engine for FCPXML files.
|
||||
|
||||
Compares two FCPXML timelines and reports differences in clips,
|
||||
markers, transitions, and format settings.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
from .parser import FCPXMLParser
|
||||
|
||||
|
||||
@dataclass
|
||||
class ClipDiff:
|
||||
"""Difference record for a single clip."""
|
||||
action: str # "added", "removed", "moved", "trimmed", "unchanged"
|
||||
clip_name: str
|
||||
details: str = ""
|
||||
old_start: Optional[float] = None
|
||||
new_start: Optional[float] = None
|
||||
old_duration: Optional[float] = None
|
||||
new_duration: Optional[float] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class MarkerDiff:
|
||||
"""Difference record for a marker."""
|
||||
action: str # "added", "removed", "moved"
|
||||
marker_name: str
|
||||
details: str = ""
|
||||
old_position: Optional[float] = None
|
||||
new_position: Optional[float] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class TimelineDiff:
|
||||
"""Complete diff between two timelines."""
|
||||
timeline_a_name: str
|
||||
timeline_b_name: str
|
||||
clip_diffs: List[ClipDiff] = field(default_factory=list)
|
||||
marker_diffs: List[MarkerDiff] = field(default_factory=list)
|
||||
transition_diffs: List[str] = field(default_factory=list)
|
||||
format_changes: List[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def total_changes(self) -> int:
|
||||
changes = [d for d in self.clip_diffs if d.action != "unchanged"]
|
||||
return len(changes) + len(self.marker_diffs) + len(self.transition_diffs) + len(self.format_changes)
|
||||
|
||||
@property
|
||||
def has_changes(self) -> bool:
|
||||
return self.total_changes > 0
|
||||
|
||||
|
||||
def _clip_identity(clip) -> Tuple[str, float]:
|
||||
"""Build identity key for a clip: (name, source_start rounded to 0.01s)."""
|
||||
source_start = round(clip.source_start.seconds, 2) if clip.source_start else 0.0
|
||||
return (clip.name, source_start)
|
||||
|
||||
|
||||
def compare_timelines(filepath_a: str, filepath_b: str) -> TimelineDiff:
|
||||
"""Compare two FCPXML files and return structured diff.
|
||||
|
||||
Uses clip identity (name + source in-point) to match clips between
|
||||
timelines, then detects moved, trimmed, added, and removed clips.
|
||||
|
||||
Args:
|
||||
filepath_a: Path to baseline FCPXML
|
||||
filepath_b: Path to comparison FCPXML
|
||||
|
||||
Returns:
|
||||
TimelineDiff with all detected changes
|
||||
"""
|
||||
parser_a = FCPXMLParser()
|
||||
parser_b = FCPXMLParser()
|
||||
project_a = parser_a.parse_file(filepath_a)
|
||||
project_b = parser_b.parse_file(filepath_b)
|
||||
|
||||
tl_a = project_a.primary_timeline
|
||||
tl_b = project_b.primary_timeline
|
||||
|
||||
if tl_a is None or tl_b is None:
|
||||
return TimelineDiff(
|
||||
timeline_a_name=tl_a.name if tl_a else "No timeline",
|
||||
timeline_b_name=tl_b.name if tl_b else "No timeline",
|
||||
)
|
||||
|
||||
diff = TimelineDiff(
|
||||
timeline_a_name=tl_a.name,
|
||||
timeline_b_name=tl_b.name,
|
||||
)
|
||||
|
||||
# Compare clips
|
||||
_compare_clips(tl_a.clips, tl_b.clips, diff)
|
||||
|
||||
# Compare markers
|
||||
_compare_markers(tl_a, tl_b, diff)
|
||||
|
||||
# Compare transitions
|
||||
_compare_transitions(tl_a.transitions, tl_b.transitions, diff)
|
||||
|
||||
# Compare format
|
||||
_compare_format(tl_a, tl_b, diff)
|
||||
|
||||
return diff
|
||||
|
||||
|
||||
def _compare_clips(clips_a, clips_b, diff: TimelineDiff):
|
||||
"""Compare clip sequences between two timelines."""
|
||||
# Build identity maps
|
||||
map_a: Dict[Tuple[str, float], list] = {}
|
||||
for clip in clips_a:
|
||||
key = _clip_identity(clip)
|
||||
map_a.setdefault(key, []).append(clip)
|
||||
|
||||
map_b: Dict[Tuple[str, float], list] = {}
|
||||
for clip in clips_b:
|
||||
key = _clip_identity(clip)
|
||||
map_b.setdefault(key, []).append(clip)
|
||||
|
||||
all_keys = set(map_a.keys()) | set(map_b.keys())
|
||||
|
||||
for key in sorted(all_keys, key=lambda k: k[1]):
|
||||
a_clips = map_a.get(key, [])
|
||||
b_clips = map_b.get(key, [])
|
||||
|
||||
if not a_clips and b_clips:
|
||||
for clip in b_clips:
|
||||
diff.clip_diffs.append(ClipDiff(
|
||||
action="added",
|
||||
clip_name=clip.name,
|
||||
details=f"Added at {clip.start.to_smpte()}",
|
||||
new_start=clip.start.seconds,
|
||||
new_duration=clip.duration_seconds,
|
||||
))
|
||||
elif a_clips and not b_clips:
|
||||
for clip in a_clips:
|
||||
diff.clip_diffs.append(ClipDiff(
|
||||
action="removed",
|
||||
clip_name=clip.name,
|
||||
details=f"Removed from {clip.start.to_smpte()}",
|
||||
old_start=clip.start.seconds,
|
||||
old_duration=clip.duration_seconds,
|
||||
))
|
||||
else:
|
||||
# Match clips pairwise
|
||||
for i in range(max(len(a_clips), len(b_clips))):
|
||||
if i >= len(a_clips):
|
||||
diff.clip_diffs.append(ClipDiff(
|
||||
action="added", clip_name=b_clips[i].name,
|
||||
new_start=b_clips[i].start.seconds,
|
||||
new_duration=b_clips[i].duration_seconds,
|
||||
))
|
||||
elif i >= len(b_clips):
|
||||
diff.clip_diffs.append(ClipDiff(
|
||||
action="removed", clip_name=a_clips[i].name,
|
||||
old_start=a_clips[i].start.seconds,
|
||||
old_duration=a_clips[i].duration_seconds,
|
||||
))
|
||||
else:
|
||||
a, b = a_clips[i], b_clips[i]
|
||||
moved = abs(a.start.seconds - b.start.seconds) > 0.04
|
||||
trimmed = abs(a.duration_seconds - b.duration_seconds) > 0.04
|
||||
|
||||
if moved and trimmed:
|
||||
diff.clip_diffs.append(ClipDiff(
|
||||
action="moved",
|
||||
clip_name=a.name,
|
||||
details=(
|
||||
f"Moved {a.start.to_smpte()} -> {b.start.to_smpte()}, "
|
||||
f"trimmed {a.duration_seconds:.2f}s -> {b.duration_seconds:.2f}s"
|
||||
),
|
||||
old_start=a.start.seconds,
|
||||
new_start=b.start.seconds,
|
||||
old_duration=a.duration_seconds,
|
||||
new_duration=b.duration_seconds,
|
||||
))
|
||||
elif moved:
|
||||
diff.clip_diffs.append(ClipDiff(
|
||||
action="moved",
|
||||
clip_name=a.name,
|
||||
details=f"Moved {a.start.to_smpte()} -> {b.start.to_smpte()}",
|
||||
old_start=a.start.seconds,
|
||||
new_start=b.start.seconds,
|
||||
old_duration=a.duration_seconds,
|
||||
new_duration=b.duration_seconds,
|
||||
))
|
||||
elif trimmed:
|
||||
diff.clip_diffs.append(ClipDiff(
|
||||
action="trimmed",
|
||||
clip_name=a.name,
|
||||
details=f"Duration {a.duration_seconds:.2f}s -> {b.duration_seconds:.2f}s",
|
||||
old_start=a.start.seconds,
|
||||
new_start=b.start.seconds,
|
||||
old_duration=a.duration_seconds,
|
||||
new_duration=b.duration_seconds,
|
||||
))
|
||||
|
||||
|
||||
def _compare_markers(tl_a, tl_b, diff: TimelineDiff):
|
||||
"""Compare markers between two timelines."""
|
||||
# Collect all markers including clip-level markers
|
||||
markers_a = {}
|
||||
for m in tl_a.markers:
|
||||
markers_a[m.name] = m.start.seconds
|
||||
for clip in tl_a.clips:
|
||||
for m in clip.markers:
|
||||
markers_a[m.name] = clip.start.seconds + m.start.seconds
|
||||
|
||||
markers_b = {}
|
||||
for m in tl_b.markers:
|
||||
markers_b[m.name] = m.start.seconds
|
||||
for clip in tl_b.clips:
|
||||
for m in clip.markers:
|
||||
markers_b[m.name] = clip.start.seconds + m.start.seconds
|
||||
|
||||
all_names = set(markers_a.keys()) | set(markers_b.keys())
|
||||
|
||||
for name in sorted(all_names):
|
||||
if name in markers_a and name not in markers_b:
|
||||
diff.marker_diffs.append(MarkerDiff(
|
||||
action="removed", marker_name=name,
|
||||
old_position=markers_a[name],
|
||||
))
|
||||
elif name not in markers_a and name in markers_b:
|
||||
diff.marker_diffs.append(MarkerDiff(
|
||||
action="added", marker_name=name,
|
||||
new_position=markers_b[name],
|
||||
))
|
||||
elif abs(markers_a[name] - markers_b[name]) > 1.0:
|
||||
diff.marker_diffs.append(MarkerDiff(
|
||||
action="moved", marker_name=name,
|
||||
details=f"Position {markers_a[name]:.2f}s -> {markers_b[name]:.2f}s",
|
||||
old_position=markers_a[name],
|
||||
new_position=markers_b[name],
|
||||
))
|
||||
|
||||
|
||||
def _compare_transitions(trans_a, trans_b, diff: TimelineDiff):
|
||||
"""Compare transitions between timelines."""
|
||||
count_a = len(trans_a)
|
||||
count_b = len(trans_b)
|
||||
if count_a != count_b:
|
||||
diff.transition_diffs.append(
|
||||
f"Transition count changed: {count_a} -> {count_b}"
|
||||
)
|
||||
|
||||
for i in range(min(count_a, count_b)):
|
||||
a, b = trans_a[i], trans_b[i]
|
||||
if a.name != b.name:
|
||||
diff.transition_diffs.append(
|
||||
f"Transition {i + 1}: {a.name} -> {b.name}"
|
||||
)
|
||||
if abs(a.duration.seconds - b.duration.seconds) > 0.04:
|
||||
diff.transition_diffs.append(
|
||||
f"Transition {i + 1} duration: {a.duration.seconds:.2f}s -> {b.duration.seconds:.2f}s"
|
||||
)
|
||||
|
||||
|
||||
def _compare_format(tl_a, tl_b, diff: TimelineDiff):
|
||||
"""Compare format settings (resolution, frame rate)."""
|
||||
if tl_a.width != tl_b.width or tl_a.height != tl_b.height:
|
||||
diff.format_changes.append(
|
||||
f"Resolution: {tl_a.width}x{tl_a.height} -> {tl_b.width}x{tl_b.height}"
|
||||
)
|
||||
if abs(tl_a.frame_rate - tl_b.frame_rate) > 0.01:
|
||||
diff.format_changes.append(
|
||||
f"Frame rate: {tl_a.frame_rate:.3f} -> {tl_b.frame_rate:.3f}"
|
||||
)
|
||||
Executable
+112
@@ -0,0 +1,112 @@
|
||||
"""Validate FCPXML against Apple's official DTDs.
|
||||
|
||||
Apple stopped publishing FCPXML DTDs online after version 1.10 — the only
|
||||
authoritative spec is the set of DTD files shipped inside the Final Cut
|
||||
Pro app bundle (FCPXMLv1_0.dtd through FCPXMLv1_14.dtd as of FCP 12.x).
|
||||
This module locates those DTDs on the local machine and validates
|
||||
generated FCPXML against them with ``xmllint``.
|
||||
|
||||
Validation is strictly opt-in/best-effort: on machines without Final Cut
|
||||
Pro (CI, Linux, servers) every helper degrades to "DTDs unavailable"
|
||||
rather than failing, so the server never *requires* FCP to be installed.
|
||||
|
||||
Set ``FCPXML_DTD_DIR`` to point at a directory of ``FCPXMLv*_*.dtd``
|
||||
files to validate without an FCP install (e.g. DTDs copied to a CI
|
||||
runner — note Apple's DTDs are Apple-copyrighted, so they are located
|
||||
at runtime rather than redistributed with this project).
|
||||
"""
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Optional, Tuple
|
||||
|
||||
# Where FCP 10.6+ keeps its DTDs (verified against FCP 12.2)
|
||||
_INTERCHANGE_RESOURCES = Path(
|
||||
"/Applications/Final Cut Pro.app/Contents/Frameworks/"
|
||||
"Interchange.framework/Versions/A/Resources"
|
||||
)
|
||||
|
||||
_XMLLINT_TIMEOUT_SECONDS = 30
|
||||
|
||||
|
||||
def dtd_search_dir() -> Path:
|
||||
"""Directory searched for Apple FCPXML DTDs (env-overridable)."""
|
||||
return Path(os.environ.get("FCPXML_DTD_DIR", str(_INTERCHANGE_RESOURCES)))
|
||||
|
||||
|
||||
def find_apple_dtd(version: str) -> Optional[Path]:
|
||||
"""Locate the Apple DTD for an FCPXML *version* (e.g. ``"1.13"``).
|
||||
|
||||
Returns None when the DTD (or the whole directory) is absent —
|
||||
typically a machine without Final Cut Pro installed.
|
||||
"""
|
||||
safe = version.strip()
|
||||
if not safe.replace(".", "").isdigit():
|
||||
return None
|
||||
candidate = dtd_search_dir() / f"FCPXMLv{safe.replace('.', '_')}.dtd"
|
||||
return candidate if candidate.is_file() else None
|
||||
|
||||
|
||||
def available_dtd_versions() -> list[str]:
|
||||
"""List FCPXML versions with a locally available Apple DTD."""
|
||||
directory = dtd_search_dir()
|
||||
if not directory.is_dir():
|
||||
return []
|
||||
versions = []
|
||||
for dtd in directory.glob("FCPXMLv*_*.dtd"):
|
||||
versions.append(dtd.stem.removeprefix("FCPXMLv").replace("_", "."))
|
||||
return sorted(versions, key=lambda v: [int(p) for p in v.split(".")])
|
||||
|
||||
|
||||
def validate_against_dtd(
|
||||
fcpxml_path: str,
|
||||
version: Optional[str] = None,
|
||||
) -> Tuple[Optional[bool], str]:
|
||||
"""Validate an FCPXML file against Apple's DTD via ``xmllint``.
|
||||
|
||||
Args:
|
||||
fcpxml_path: Path to a ``.fcpxml`` file or ``.fcpxmld`` bundle.
|
||||
version: FCPXML version to validate against. Defaults to the
|
||||
file's own ``version`` attribute.
|
||||
|
||||
Returns:
|
||||
``(ok, detail)`` where *ok* is True (valid), False (invalid), or
|
||||
None (validation unavailable: no xmllint, no DTD on this
|
||||
machine, or unreadable input — *detail* says which).
|
||||
"""
|
||||
path = Path(fcpxml_path)
|
||||
if path.suffix.lower() == ".fcpxmld":
|
||||
path = path / "Info.fcpxml"
|
||||
if not path.is_file():
|
||||
return None, f"File not found: {path}"
|
||||
|
||||
if shutil.which("xmllint") is None:
|
||||
return None, "xmllint not available on PATH"
|
||||
|
||||
if version is None:
|
||||
# Cheap version sniff without a full parse
|
||||
from .safe_xml import safe_parse
|
||||
|
||||
version = safe_parse(str(path)).getroot().get("version", "1.13")
|
||||
|
||||
dtd = find_apple_dtd(version)
|
||||
if dtd is None:
|
||||
return None, (
|
||||
f"No Apple DTD for FCPXML {version} found in {dtd_search_dir()} "
|
||||
f"(Final Cut Pro not installed? Set FCPXML_DTD_DIR to override)"
|
||||
)
|
||||
|
||||
# xmllint treats the --dtdvalid argument as a URI: raw paths containing
|
||||
# spaces (e.g. ".../Final Cut Pro.app/...") fail entity resolution with
|
||||
# "xmlSAX2ResolveEntity". Path.as_uri() percent-encodes them.
|
||||
proc = subprocess.run(
|
||||
["xmllint", "--noout", "--dtdvalid", dtd.resolve().as_uri(), str(path)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=_XMLLINT_TIMEOUT_SECONDS,
|
||||
)
|
||||
if proc.returncode == 0:
|
||||
return True, f"Valid against FCPXML {version} DTD ({dtd.name})"
|
||||
return False, proc.stderr.strip() or f"xmllint exit code {proc.returncode}"
|
||||
Executable
+226
@@ -0,0 +1,226 @@
|
||||
"""Export FCPXML to other NLE formats.
|
||||
|
||||
Supports:
|
||||
- Simplified FCPXML v1.9 for DaVinci Resolve compatibility
|
||||
- FCP7 XML (XMEML) for Premiere Pro / Resolve / Avid compatibility
|
||||
"""
|
||||
|
||||
import copy
|
||||
import xml.etree.ElementTree as ET
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from .parser import FCPXMLParser
|
||||
from .safe_xml import serialize_xml
|
||||
from .writer import _sanitize_xml_value
|
||||
|
||||
|
||||
class DaVinciExporter:
|
||||
"""Export FCPXML in formats compatible with other NLEs."""
|
||||
|
||||
def __init__(self, source_path: str):
|
||||
"""Load the source FCPXML file.
|
||||
|
||||
Args:
|
||||
source_path: Path to the source FCPXML file
|
||||
"""
|
||||
self.source_path = source_path
|
||||
from .safe_xml import safe_parse
|
||||
self.tree = safe_parse(source_path)
|
||||
self.root = self.tree.getroot()
|
||||
self.parser = FCPXMLParser()
|
||||
self.project = self.parser.parse_file(source_path)
|
||||
|
||||
def export_simplified_fcpxml(self, output_path: str,
|
||||
flatten_compounds: bool = True) -> str:
|
||||
"""Generate simplified FCPXML v1.9 for DaVinci Resolve.
|
||||
|
||||
Strips features that cause Resolve import issues:
|
||||
- Downgrades version to 1.9
|
||||
- Optionally flattens compound clips
|
||||
- Removes unsupported attributes
|
||||
|
||||
Args:
|
||||
output_path: Destination file path
|
||||
flatten_compounds: Whether to flatten compound clips
|
||||
|
||||
Returns:
|
||||
Path to the generated file
|
||||
"""
|
||||
root = copy.deepcopy(self.root)
|
||||
root.set('version', '1.9')
|
||||
|
||||
# Strip attributes that cause Resolve issues
|
||||
unsupported_attrs = ['mcClipAngle', 'modDate', 'colorProcessing']
|
||||
for elem in root.iter():
|
||||
for attr in unsupported_attrs:
|
||||
if attr in elem.attrib:
|
||||
del elem.attrib[attr]
|
||||
|
||||
# Flatten compound clips (ref-clips pointing to media sequences)
|
||||
if flatten_compounds:
|
||||
for spine in root.findall('.//spine'):
|
||||
for ref_clip in spine.findall('ref-clip'):
|
||||
# Convert ref-clip to simple asset-clip
|
||||
ref_clip.tag = 'asset-clip'
|
||||
# Keep core attributes, strip compound-specific ones
|
||||
for attr in list(ref_clip.attrib.keys()):
|
||||
if attr not in ('ref', 'offset', 'duration', 'start', 'name', 'format'):
|
||||
del ref_clip.attrib[attr]
|
||||
|
||||
return serialize_xml(root, output_path, '<!DOCTYPE fcpxml>')
|
||||
|
||||
def export_xmeml(self, output_path: str) -> str:
|
||||
"""Generate FCP7 XML (XMEML) for maximum NLE compatibility.
|
||||
|
||||
Converts FCPXML's spine-based model to XMEML's track-based model:
|
||||
- Primary storyline -> Video Track 1
|
||||
- Connected clips lane +N -> Video Track N+1
|
||||
- Connected clips lane -N -> Audio Track N+1
|
||||
|
||||
Args:
|
||||
output_path: Destination file path
|
||||
|
||||
Returns:
|
||||
Path to the generated file
|
||||
"""
|
||||
tl = self.project.primary_timeline
|
||||
if tl is None:
|
||||
raise ValueError("No timeline found in source FCPXML")
|
||||
|
||||
# Build track model from spine
|
||||
video_tracks, audio_tracks = self._spine_to_tracks()
|
||||
|
||||
# Build XMEML document
|
||||
xmeml = ET.Element('xmeml')
|
||||
xmeml.set('version', '5')
|
||||
|
||||
sequence = ET.SubElement(xmeml, 'sequence')
|
||||
ET.SubElement(sequence, 'name').text = _sanitize_xml_value(tl.name, 512)
|
||||
|
||||
total_frames = int(tl.duration.seconds * tl.frame_rate)
|
||||
ET.SubElement(sequence, 'duration').text = str(total_frames)
|
||||
|
||||
rate = ET.SubElement(sequence, 'rate')
|
||||
ET.SubElement(rate, 'timebase').text = str(int(tl.frame_rate))
|
||||
is_ntsc = tl.frame_rate in (29.97, 23.976, 59.94)
|
||||
ET.SubElement(rate, 'ntsc').text = 'TRUE' if is_ntsc else 'FALSE'
|
||||
|
||||
# Timecode
|
||||
tc = ET.SubElement(sequence, 'timecode')
|
||||
tc_rate = ET.SubElement(tc, 'rate')
|
||||
ET.SubElement(tc_rate, 'timebase').text = str(int(tl.frame_rate))
|
||||
ET.SubElement(tc_rate, 'ntsc').text = 'TRUE' if is_ntsc else 'FALSE'
|
||||
ET.SubElement(tc, 'string').text = '00:00:00:00'
|
||||
ET.SubElement(tc, 'frame').text = '0'
|
||||
|
||||
media = ET.SubElement(sequence, 'media')
|
||||
|
||||
# Video tracks
|
||||
video = ET.SubElement(media, 'video')
|
||||
format_elem = ET.SubElement(video, 'format')
|
||||
sample = ET.SubElement(format_elem, 'samplecharacteristics')
|
||||
ET.SubElement(sample, 'width').text = str(tl.width)
|
||||
ET.SubElement(sample, 'height').text = str(tl.height)
|
||||
|
||||
for track_num in sorted(video_tracks.keys()):
|
||||
track = ET.SubElement(video, 'track')
|
||||
for clip_data in video_tracks[track_num]:
|
||||
self._add_xmeml_clipitem(track, clip_data, tl.frame_rate)
|
||||
|
||||
# Audio tracks
|
||||
audio = ET.SubElement(media, 'audio')
|
||||
for track_num in sorted(audio_tracks.keys()):
|
||||
track = ET.SubElement(audio, 'track')
|
||||
for clip_data in audio_tracks[track_num]:
|
||||
self._add_xmeml_clipitem(track, clip_data, tl.frame_rate)
|
||||
|
||||
# If no explicit audio tracks, create one from primary storyline
|
||||
if not audio_tracks and video_tracks.get(0):
|
||||
track = ET.SubElement(audio, 'track')
|
||||
for clip_data in video_tracks[0]:
|
||||
if clip_data.get('has_audio', True):
|
||||
self._add_xmeml_clipitem(track, clip_data, tl.frame_rate)
|
||||
|
||||
return serialize_xml(xmeml, output_path, '<!DOCTYPE xmeml>')
|
||||
|
||||
def _spine_to_tracks(self) -> tuple:
|
||||
"""Convert spine + connected clips to track-based model.
|
||||
|
||||
Returns:
|
||||
Tuple of (video_tracks, audio_tracks) where each is a
|
||||
dict mapping track number to list of clip data dicts.
|
||||
"""
|
||||
tl = self.project.primary_timeline
|
||||
video_tracks: Dict[int, List[Dict[str, Any]]] = {0: []}
|
||||
audio_tracks: Dict[int, List[Dict[str, Any]]] = {}
|
||||
|
||||
# Primary storyline -> track 0
|
||||
for clip in tl.clips:
|
||||
video_tracks[0].append({
|
||||
'name': clip.name,
|
||||
'start_seconds': clip.start.seconds,
|
||||
'duration_seconds': clip.duration_seconds,
|
||||
'source_start_seconds': clip.source_start.seconds if clip.source_start else 0,
|
||||
'media_path': clip.media_path,
|
||||
'has_audio': True,
|
||||
})
|
||||
|
||||
# Connected clips -> higher tracks
|
||||
for cc in tl.connected_clips:
|
||||
if cc.lane > 0:
|
||||
track_num = cc.lane
|
||||
if track_num not in video_tracks:
|
||||
video_tracks[track_num] = []
|
||||
video_tracks[track_num].append({
|
||||
'name': cc.name,
|
||||
'start_seconds': cc.offset.seconds if cc.offset else 0,
|
||||
'duration_seconds': cc.duration_seconds,
|
||||
'source_start_seconds': cc.source_start.seconds if cc.source_start else 0,
|
||||
'media_path': cc.media_path,
|
||||
'has_audio': cc.clip_type not in ('title', 'video'),
|
||||
})
|
||||
elif cc.lane < 0:
|
||||
track_num = abs(cc.lane) - 1
|
||||
if track_num not in audio_tracks:
|
||||
audio_tracks[track_num] = []
|
||||
audio_tracks[track_num].append({
|
||||
'name': cc.name,
|
||||
'start_seconds': cc.offset.seconds if cc.offset else 0,
|
||||
'duration_seconds': cc.duration_seconds,
|
||||
'source_start_seconds': cc.source_start.seconds if cc.source_start else 0,
|
||||
'media_path': cc.media_path,
|
||||
'has_audio': True,
|
||||
})
|
||||
|
||||
return video_tracks, audio_tracks
|
||||
|
||||
def _add_xmeml_clipitem(self, track: ET.Element,
|
||||
clip_data: Dict[str, Any],
|
||||
frame_rate: float):
|
||||
"""Add a clipitem element to an XMEML track."""
|
||||
clipitem = ET.SubElement(track, 'clipitem')
|
||||
ET.SubElement(clipitem, 'name').text = _sanitize_xml_value(
|
||||
clip_data['name'], 512
|
||||
)
|
||||
|
||||
duration_frames = int(clip_data['duration_seconds'] * frame_rate)
|
||||
ET.SubElement(clipitem, 'duration').text = str(duration_frames)
|
||||
|
||||
rate = ET.SubElement(clipitem, 'rate')
|
||||
ET.SubElement(rate, 'timebase').text = str(int(frame_rate))
|
||||
is_ntsc = frame_rate in (29.97, 23.976, 59.94)
|
||||
ET.SubElement(rate, 'ntsc').text = 'TRUE' if is_ntsc else 'FALSE'
|
||||
|
||||
start_frame = int(clip_data['start_seconds'] * frame_rate)
|
||||
source_in = int(clip_data['source_start_seconds'] * frame_rate)
|
||||
source_out = source_in + duration_frames
|
||||
|
||||
ET.SubElement(clipitem, 'start').text = str(start_frame)
|
||||
ET.SubElement(clipitem, 'end').text = str(start_frame + duration_frames)
|
||||
ET.SubElement(clipitem, 'in').text = str(source_in)
|
||||
ET.SubElement(clipitem, 'out').text = str(source_out)
|
||||
|
||||
if clip_data.get('media_path'):
|
||||
file_elem = ET.SubElement(clipitem, 'file')
|
||||
pathurl = ET.SubElement(file_elem, 'pathurl')
|
||||
pathurl.text = _sanitize_xml_value(clip_data['media_path'], 2048)
|
||||
@@ -0,0 +1,445 @@
|
||||
"""Glyph advance widths for the fonts the subtitle rhythm uses.
|
||||
|
||||
Generated from the actual font files on the editing machine — the macOS
|
||||
system fonts (/System/Library/Fonts/Helvetica.ttc, HelveticaNeue.ttc) and the
|
||||
installed Playfair Display family (~/Library/Fonts) the editorial caption look
|
||||
uses — as a fraction of the em size. Embedded rather than read at runtime so layout needs no font
|
||||
library and stays identical on any machine.
|
||||
|
||||
Estimating these from a generic table was not good enough: the variants differ
|
||||
by up to 21% of the em on some glyphs, and an estimate that runs low on even
|
||||
one word puts two words on top of each other — the failure this table exists to
|
||||
rule out. Measured error against these tables is zero for the covered glyphs.
|
||||
|
||||
Keys are "<family>" or "<family>-<face>", lowercased.
|
||||
"""
|
||||
|
||||
METRICS = {
|
||||
'helvetica': {
|
||||
' ': 0.2778, '!': 0.2778, '"': 0.355, '#': 0.5562, '$': 0.5562, '%': 0.8892, '&':
|
||||
0.667, "'": 0.1909, '(': 0.333, ')': 0.333, '*': 0.3892, '+': 0.584, ',': 0.2778,
|
||||
'-': 0.333, '.': 0.2778, '/': 0.2778, '0': 0.5562, '1': 0.5562, '2': 0.5562, '3':
|
||||
0.5562, '4': 0.5562, '5': 0.5562, '6': 0.5562, '7': 0.5562, '8': 0.5562, '9':
|
||||
0.5562, ':': 0.2778, ';': 0.2778, '<': 0.584, '=': 0.584, '>': 0.584, '?': 0.5562,
|
||||
'@': 1.0151, 'A': 0.667, 'B': 0.667, 'C': 0.7222, 'D': 0.7222, 'E': 0.667, 'F':
|
||||
0.6108, 'G': 0.7778, 'H': 0.7222, 'I': 0.2778, 'J': 0.5, 'K': 0.667, 'L': 0.5562,
|
||||
'M': 0.833, 'N': 0.7222, 'O': 0.7778, 'P': 0.667, 'Q': 0.7778, 'R': 0.7222, 'S':
|
||||
0.667, 'T': 0.6108, 'U': 0.7222, 'V': 0.667, 'W': 0.9438, 'X': 0.667, 'Y': 0.667,
|
||||
'Z': 0.6108, '[': 0.2778, '\\': 0.2778, ']': 0.2778, '_': 0.5562, 'a': 0.5562, 'b':
|
||||
0.5562, 'c': 0.5, 'd': 0.5562, 'e': 0.5562, 'f': 0.2778, 'g': 0.5562, 'h': 0.5562,
|
||||
'i': 0.2222, 'j': 0.2222, 'k': 0.5, 'l': 0.2222, 'm': 0.833, 'n': 0.5562, 'o':
|
||||
0.5562, 'p': 0.5562, 'q': 0.5562, 'r': 0.333, 's': 0.5, 't': 0.2778, 'u': 0.5562,
|
||||
'v': 0.5, 'w': 0.7222, 'x': 0.5, 'y': 0.5, 'z': 0.5, '{': 0.334, '|': 0.2598, '}':
|
||||
0.334, '~': 0.584, 'ª': 0.3701, 'º': 0.3652, 'À': 0.667, 'Á': 0.667, 'Â': 0.667,
|
||||
'Ã': 0.667, 'Ä': 0.667, 'Ç': 0.7222, 'È': 0.667, 'É': 0.667, 'Ê': 0.667, 'Ë': 0.667,
|
||||
'Ì': 0.2778, 'Í': 0.2778, 'Î': 0.2778, 'Ï': 0.2778, 'Ñ': 0.7222, 'Ò': 0.7778, 'Ó':
|
||||
0.7778, 'Ô': 0.7778, 'Õ': 0.7778, 'Ö': 0.7778, 'Ù': 0.7222, 'Ú': 0.7222, 'Û':
|
||||
0.7222, 'Ü': 0.7222, 'à': 0.5562, 'á': 0.5562, 'â': 0.5562, 'ã': 0.5562, 'ä':
|
||||
0.5562, 'ç': 0.5, 'è': 0.5562, 'é': 0.5562, 'ê': 0.5562, 'ë': 0.5562, 'ì': 0.2778,
|
||||
'í': 0.2778, 'î': 0.2778, 'ï': 0.2778, 'ñ': 0.5562, 'ò': 0.5562, 'ó': 0.5562, 'ô':
|
||||
0.5562, 'õ': 0.5562, 'ö': 0.5562, 'ù': 0.5562, 'ú': 0.5562, 'û': 0.5562, 'ü': 0.5562
|
||||
},
|
||||
'helvetica neue': {
|
||||
' ': 0.278, '!': 0.259, '"': 0.426, '#': 0.556, '$': 0.556, '%': 1.0, '&': 0.63,
|
||||
"'": 0.278, '(': 0.259, ')': 0.259, '*': 0.352, '+': 0.6, ',': 0.278, '-': 0.389,
|
||||
'.': 0.278, '/': 0.333, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
|
||||
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
|
||||
'<': 0.6, '=': 0.6, '>': 0.6, '?': 0.556, '@': 0.8, 'A': 0.648, 'B': 0.685, 'C':
|
||||
0.722, 'D': 0.704, 'E': 0.611, 'F': 0.574, 'G': 0.759, 'H': 0.722, 'I': 0.259, 'J':
|
||||
0.519, 'K': 0.667, 'L': 0.556, 'M': 0.871, 'N': 0.722, 'O': 0.76, 'P': 0.648, 'Q':
|
||||
0.76, 'R': 0.685, 'S': 0.648, 'T': 0.574, 'U': 0.722, 'V': 0.611, 'W': 0.926, 'X':
|
||||
0.611, 'Y': 0.648, 'Z': 0.611, '[': 0.259, '\\': 0.333, ']': 0.259, '_': 0.5, 'a':
|
||||
0.537, 'b': 0.593, 'c': 0.537, 'd': 0.593, 'e': 0.537, 'f': 0.296, 'g': 0.574, 'h':
|
||||
0.556, 'i': 0.222, 'j': 0.222, 'k': 0.519, 'l': 0.222, 'm': 0.853, 'n': 0.556, 'o':
|
||||
0.574, 'p': 0.593, 'q': 0.593, 'r': 0.333, 's': 0.5, 't': 0.315, 'u': 0.556, 'v':
|
||||
0.5, 'w': 0.758, 'x': 0.518, 'y': 0.5, 'z': 0.48, '{': 0.333, '|': 0.222, '}':
|
||||
0.333, '~': 0.6, 'ª': 0.378, 'º': 0.384, 'À': 0.648, 'Á': 0.648, 'Â': 0.648, 'Ã':
|
||||
0.648, 'Ä': 0.648, 'Ç': 0.722, 'È': 0.611, 'É': 0.611, 'Ê': 0.611, 'Ë': 0.611, 'Ì':
|
||||
0.259, 'Í': 0.259, 'Î': 0.259, 'Ï': 0.259, 'Ñ': 0.722, 'Ò': 0.76, 'Ó': 0.76, 'Ô':
|
||||
0.76, 'Õ': 0.76, 'Ö': 0.76, 'Ù': 0.722, 'Ú': 0.722, 'Û': 0.722, 'Ü': 0.722, 'à':
|
||||
0.537, 'á': 0.537, 'â': 0.537, 'ã': 0.537, 'ä': 0.537, 'ç': 0.537, 'è': 0.537, 'é':
|
||||
0.537, 'ê': 0.537, 'ë': 0.537, 'ì': 0.222, 'í': 0.222, 'î': 0.222, 'ï': 0.222, 'ñ':
|
||||
0.556, 'ò': 0.574, 'ó': 0.574, 'ô': 0.574, 'õ': 0.574, 'ö': 0.574, 'ù': 0.556, 'ú':
|
||||
0.556, 'û': 0.556, 'ü': 0.556
|
||||
},
|
||||
'helvetica neue-bold': {
|
||||
' ': 0.278, '!': 0.278, '"': 0.463, '#': 0.556, '$': 0.556, '%': 1.0, '&': 0.685,
|
||||
"'": 0.278, '(': 0.296, ')': 0.296, '*': 0.407, '+': 0.6, ',': 0.278, '-': 0.407,
|
||||
'.': 0.278, '/': 0.371, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
|
||||
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
|
||||
'<': 0.6, '=': 0.6, '>': 0.6, '?': 0.556, '@': 0.8, 'A': 0.685, 'B': 0.704, 'C':
|
||||
0.741, 'D': 0.741, 'E': 0.648, 'F': 0.593, 'G': 0.759, 'H': 0.741, 'I': 0.295, 'J':
|
||||
0.556, 'K': 0.722, 'L': 0.593, 'M': 0.907, 'N': 0.741, 'O': 0.778, 'P': 0.667, 'Q':
|
||||
0.778, 'R': 0.722, 'S': 0.649, 'T': 0.611, 'U': 0.741, 'V': 0.63, 'W': 0.944, 'X':
|
||||
0.667, 'Y': 0.667, 'Z': 0.648, '[': 0.333, '\\': 0.371, ']': 0.333, '_': 0.5, 'a':
|
||||
0.574, 'b': 0.611, 'c': 0.574, 'd': 0.611, 'e': 0.574, 'f': 0.333, 'g': 0.611, 'h':
|
||||
0.593, 'i': 0.258, 'j': 0.278, 'k': 0.574, 'l': 0.258, 'm': 0.906, 'n': 0.593, 'o':
|
||||
0.611, 'p': 0.611, 'q': 0.611, 'r': 0.389, 's': 0.537, 't': 0.352, 'u': 0.593, 'v':
|
||||
0.52, 'w': 0.814, 'x': 0.537, 'y': 0.519, 'z': 0.519, '{': 0.333, '|': 0.223, '}':
|
||||
0.333, '~': 0.6, 'ª': 0.344, 'º': 0.367, 'À': 0.685, 'Á': 0.685, 'Â': 0.685, 'Ã':
|
||||
0.685, 'Ä': 0.685, 'Ç': 0.741, 'È': 0.648, 'É': 0.648, 'Ê': 0.648, 'Ë': 0.648, 'Ì':
|
||||
0.295, 'Í': 0.295, 'Î': 0.295, 'Ï': 0.295, 'Ñ': 0.741, 'Ò': 0.778, 'Ó': 0.778, 'Ô':
|
||||
0.778, 'Õ': 0.778, 'Ö': 0.778, 'Ù': 0.741, 'Ú': 0.741, 'Û': 0.741, 'Ü': 0.741, 'à':
|
||||
0.574, 'á': 0.574, 'â': 0.574, 'ã': 0.574, 'ä': 0.574, 'ç': 0.574, 'è': 0.574, 'é':
|
||||
0.574, 'ê': 0.574, 'ë': 0.574, 'ì': 0.258, 'í': 0.258, 'î': 0.258, 'ï': 0.258, 'ñ':
|
||||
0.593, 'ò': 0.611, 'ó': 0.611, 'ô': 0.611, 'õ': 0.611, 'ö': 0.611, 'ù': 0.593, 'ú':
|
||||
0.593, 'û': 0.593, 'ü': 0.593
|
||||
},
|
||||
'helvetica neue-italic': {
|
||||
' ': 0.278, '!': 0.259, '"': 0.426, '#': 0.556, '$': 0.556, '%': 0.926, '&': 0.63,
|
||||
"'": 0.278, '(': 0.259, ')': 0.259, '*': 0.352, '+': 0.6, ',': 0.278, '-': 0.389,
|
||||
'.': 0.278, '/': 0.333, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
|
||||
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
|
||||
'<': 0.6, '=': 0.6, '>': 0.6, '?': 0.556, '@': 0.8, 'A': 0.667, 'B': 0.685, 'C':
|
||||
0.722, 'D': 0.704, 'E': 0.611, 'F': 0.574, 'G': 0.759, 'H': 0.722, 'I': 0.259, 'J':
|
||||
0.519, 'K': 0.667, 'L': 0.556, 'M': 0.87, 'N': 0.722, 'O': 0.759, 'P': 0.648, 'Q':
|
||||
0.759, 'R': 0.685, 'S': 0.648, 'T': 0.574, 'U': 0.722, 'V': 0.611, 'W': 0.926, 'X':
|
||||
0.611, 'Y': 0.611, 'Z': 0.611, '[': 0.259, '\\': 0.333, ']': 0.259, '_': 0.5, 'a':
|
||||
0.519, 'b': 0.593, 'c': 0.537, 'd': 0.593, 'e': 0.537, 'f': 0.296, 'g': 0.574, 'h':
|
||||
0.556, 'i': 0.222, 'j': 0.222, 'k': 0.481, 'l': 0.222, 'm': 0.852, 'n': 0.556, 'o':
|
||||
0.574, 'p': 0.593, 'q': 0.593, 'r': 0.333, 's': 0.481, 't': 0.315, 'u': 0.556, 'v':
|
||||
0.481, 'w': 0.759, 'x': 0.481, 'y': 0.481, 'z': 0.444, '{': 0.333, '|': 0.222, '}':
|
||||
0.333, '~': 0.6, 'ª': 0.311, 'º': 0.344, 'À': 0.667, 'Á': 0.667, 'Â': 0.667, 'Ã':
|
||||
0.667, 'Ä': 0.667, 'Ç': 0.722, 'È': 0.611, 'É': 0.611, 'Ê': 0.611, 'Ë': 0.611, 'Ì':
|
||||
0.259, 'Í': 0.259, 'Î': 0.259, 'Ï': 0.259, 'Ñ': 0.722, 'Ò': 0.759, 'Ó': 0.759, 'Ô':
|
||||
0.759, 'Õ': 0.759, 'Ö': 0.759, 'Ù': 0.722, 'Ú': 0.722, 'Û': 0.722, 'Ü': 0.722, 'à':
|
||||
0.519, 'á': 0.519, 'â': 0.519, 'ã': 0.519, 'ä': 0.519, 'ç': 0.537, 'è': 0.537, 'é':
|
||||
0.537, 'ê': 0.537, 'ë': 0.537, 'ì': 0.222, 'í': 0.222, 'î': 0.222, 'ï': 0.222, 'ñ':
|
||||
0.556, 'ò': 0.574, 'ó': 0.574, 'ô': 0.574, 'õ': 0.574, 'ö': 0.574, 'ù': 0.556, 'ú':
|
||||
0.556, 'û': 0.556, 'ü': 0.556
|
||||
},
|
||||
'helvetica neue-light': {
|
||||
' ': 0.278, '!': 0.241, '"': 0.37, '#': 0.556, '$': 0.556, '%': 0.889, '&': 0.611,
|
||||
"'": 0.278, '(': 0.241, ')': 0.241, '*': 0.352, '+': 0.6, ',': 0.278, '-': 0.37,
|
||||
'.': 0.278, '/': 0.333, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
|
||||
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
|
||||
'<': 0.6, '=': 0.6, '>': 0.6, '?': 0.537, '@': 0.8, 'A': 0.63, 'B': 0.667, 'C':
|
||||
0.704, 'D': 0.685, 'E': 0.593, 'F': 0.537, 'G': 0.741, 'H': 0.704, 'I': 0.222, 'J':
|
||||
0.5, 'K': 0.648, 'L': 0.537, 'M': 0.821, 'N': 0.704, 'O': 0.741, 'P': 0.63, 'Q':
|
||||
0.741, 'R': 0.667, 'S': 0.63, 'T': 0.556, 'U': 0.685, 'V': 0.593, 'W': 0.907, 'X':
|
||||
0.574, 'Y': 0.611, 'Z': 0.574, '[': 0.241, '\\': 0.333, ']': 0.241, '_': 0.5, 'a':
|
||||
0.519, 'b': 0.574, 'c': 0.519, 'd': 0.574, 'e': 0.519, 'f': 0.259, 'g': 0.556, 'h':
|
||||
0.537, 'i': 0.185, 'j': 0.185, 'k': 0.5, 'l': 0.185, 'm': 0.833, 'n': 0.537, 'o':
|
||||
0.556, 'p': 0.574, 'q': 0.574, 'r': 0.315, 's': 0.481, 't': 0.296, 'u': 0.537, 'v':
|
||||
0.463, 'w': 0.741, 'x': 0.481, 'y': 0.463, 'z': 0.463, '{': 0.333, '|': 0.222, '}':
|
||||
0.333, '~': 0.6, 'ª': 0.311, 'º': 0.334, 'À': 0.63, 'Á': 0.63, 'Â': 0.63, 'Ã': 0.63,
|
||||
'Ä': 0.63, 'Ç': 0.704, 'È': 0.593, 'É': 0.593, 'Ê': 0.593, 'Ë': 0.593, 'Ì': 0.222,
|
||||
'Í': 0.222, 'Î': 0.222, 'Ï': 0.222, 'Ñ': 0.704, 'Ò': 0.741, 'Ó': 0.741, 'Ô': 0.741,
|
||||
'Õ': 0.741, 'Ö': 0.741, 'Ù': 0.685, 'Ú': 0.685, 'Û': 0.685, 'Ü': 0.685, 'à': 0.519,
|
||||
'á': 0.519, 'â': 0.519, 'ã': 0.519, 'ä': 0.519, 'ç': 0.519, 'è': 0.519, 'é': 0.519,
|
||||
'ê': 0.519, 'ë': 0.519, 'ì': 0.185, 'í': 0.185, 'î': 0.185, 'ï': 0.185, 'ñ': 0.537,
|
||||
'ò': 0.556, 'ó': 0.556, 'ô': 0.556, 'õ': 0.556, 'ö': 0.556, 'ù': 0.537, 'ú': 0.537,
|
||||
'û': 0.537, 'ü': 0.537
|
||||
},
|
||||
'helvetica neue-light italic': {
|
||||
' ': 0.278, '!': 0.278, '"': 0.37, '#': 0.556, '$': 0.556, '%': 0.833, '&': 0.611,
|
||||
"'": 0.278, '(': 0.259, ')': 0.259, '*': 0.352, '+': 0.6, ',': 0.278, '-': 0.37,
|
||||
'.': 0.278, '/': 0.333, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
|
||||
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
|
||||
'<': 0.6, '=': 0.6, '>': 0.6, '?': 0.537, '@': 0.8, 'A': 0.63, 'B': 0.667, 'C':
|
||||
0.704, 'D': 0.685, 'E': 0.574, 'F': 0.537, 'G': 0.741, 'H': 0.704, 'I': 0.222, 'J':
|
||||
0.5, 'K': 0.648, 'L': 0.537, 'M': 0.852, 'N': 0.704, 'O': 0.741, 'P': 0.63, 'Q':
|
||||
0.741, 'R': 0.648, 'S': 0.63, 'T': 0.556, 'U': 0.704, 'V': 0.593, 'W': 0.907, 'X':
|
||||
0.574, 'Y': 0.574, 'Z': 0.574, '[': 0.241, '\\': 0.333, ']': 0.241, '_': 0.5, 'a':
|
||||
0.519, 'b': 0.574, 'c': 0.519, 'd': 0.574, 'e': 0.519, 'f': 0.259, 'g': 0.556, 'h':
|
||||
0.537, 'i': 0.185, 'j': 0.185, 'k': 0.463, 'l': 0.185, 'm': 0.833, 'n': 0.537, 'o':
|
||||
0.556, 'p': 0.574, 'q': 0.574, 'r': 0.315, 's': 0.481, 't': 0.296, 'u': 0.537, 'v':
|
||||
0.463, 'w': 0.741, 'x': 0.463, 'y': 0.463, 'z': 0.426, '{': 0.333, '|': 0.222, '}':
|
||||
0.333, '~': 0.6, 'ª': 0.311, 'º': 0.334, 'À': 0.63, 'Á': 0.63, 'Â': 0.63, 'Ã': 0.63,
|
||||
'Ä': 0.63, 'Ç': 0.704, 'È': 0.574, 'É': 0.574, 'Ê': 0.574, 'Ë': 0.574, 'Ì': 0.222,
|
||||
'Í': 0.222, 'Î': 0.222, 'Ï': 0.222, 'Ñ': 0.704, 'Ò': 0.741, 'Ó': 0.741, 'Ô': 0.741,
|
||||
'Õ': 0.741, 'Ö': 0.741, 'Ù': 0.704, 'Ú': 0.704, 'Û': 0.704, 'Ü': 0.704, 'à': 0.519,
|
||||
'á': 0.519, 'â': 0.519, 'ã': 0.519, 'ä': 0.519, 'ç': 0.519, 'è': 0.519, 'é': 0.519,
|
||||
'ê': 0.519, 'ë': 0.519, 'ì': 0.185, 'í': 0.185, 'î': 0.185, 'ï': 0.185, 'ñ': 0.537,
|
||||
'ò': 0.556, 'ó': 0.556, 'ô': 0.556, 'õ': 0.556, 'ö': 0.556, 'ù': 0.537, 'ú': 0.537,
|
||||
'û': 0.537, 'ü': 0.537
|
||||
},
|
||||
'helvetica-bold': {
|
||||
' ': 0.2778, '!': 0.333, '"': 0.4741, '#': 0.5562, '$': 0.5562, '%': 0.8892, '&':
|
||||
0.7222, "'": 0.2378, '(': 0.333, ')': 0.333, '*': 0.3892, '+': 0.584, ',': 0.2778,
|
||||
'-': 0.333, '.': 0.2778, '/': 0.2778, '0': 0.5562, '1': 0.5562, '2': 0.5562, '3':
|
||||
0.5562, '4': 0.5562, '5': 0.5562, '6': 0.5562, '7': 0.5562, '8': 0.5562, '9':
|
||||
0.5562, ':': 0.333, ';': 0.333, '<': 0.584, '=': 0.584, '>': 0.584, '?': 0.6108,
|
||||
'@': 0.9751, 'A': 0.7222, 'B': 0.7222, 'C': 0.7222, 'D': 0.7222, 'E': 0.667, 'F':
|
||||
0.6108, 'G': 0.7778, 'H': 0.7222, 'I': 0.2778, 'J': 0.5562, 'K': 0.7222, 'L':
|
||||
0.6108, 'M': 0.833, 'N': 0.7222, 'O': 0.7778, 'P': 0.667, 'Q': 0.7778, 'R': 0.7222,
|
||||
'S': 0.667, 'T': 0.6108, 'U': 0.7222, 'V': 0.667, 'W': 0.9438, 'X': 0.667, 'Y':
|
||||
0.667, 'Z': 0.6108, '[': 0.333, '\\': 0.2778, ']': 0.333, '_': 0.5562, 'a': 0.5562,
|
||||
'b': 0.6108, 'c': 0.5562, 'd': 0.6108, 'e': 0.5562, 'f': 0.333, 'g': 0.6108, 'h':
|
||||
0.6108, 'i': 0.2778, 'j': 0.2778, 'k': 0.5562, 'l': 0.2778, 'm': 0.8892, 'n':
|
||||
0.6108, 'o': 0.6108, 'p': 0.6108, 'q': 0.6108, 'r': 0.3892, 's': 0.5562, 't': 0.333,
|
||||
'u': 0.6108, 'v': 0.5562, 'w': 0.7778, 'x': 0.5562, 'y': 0.5562, 'z': 0.5, '{':
|
||||
0.3892, '|': 0.2798, '}': 0.3892, '~': 0.584, 'ª': 0.3701, 'º': 0.3652, 'À': 0.7222,
|
||||
'Á': 0.7222, 'Â': 0.7222, 'Ã': 0.7222, 'Ä': 0.7222, 'Ç': 0.7222, 'È': 0.667, 'É':
|
||||
0.667, 'Ê': 0.667, 'Ë': 0.667, 'Ì': 0.2778, 'Í': 0.2778, 'Î': 0.2778, 'Ï': 0.2778,
|
||||
'Ñ': 0.7222, 'Ò': 0.7778, 'Ó': 0.7778, 'Ô': 0.7778, 'Õ': 0.7778, 'Ö': 0.7778, 'Ù':
|
||||
0.7222, 'Ú': 0.7222, 'Û': 0.7222, 'Ü': 0.7222, 'à': 0.5562, 'á': 0.5562, 'â':
|
||||
0.5562, 'ã': 0.5562, 'ä': 0.5562, 'ç': 0.5562, 'è': 0.5562, 'é': 0.5562, 'ê':
|
||||
0.5562, 'ë': 0.5562, 'ì': 0.2778, 'í': 0.2778, 'î': 0.2778, 'ï': 0.2778, 'ñ':
|
||||
0.6108, 'ò': 0.6108, 'ó': 0.6108, 'ô': 0.6108, 'õ': 0.6108, 'ö': 0.6108, 'ù':
|
||||
0.6108, 'ú': 0.6108, 'û': 0.6108, 'ü': 0.6108
|
||||
},
|
||||
'helvetica-light': {
|
||||
' ': 0.278, '!': 0.333, '"': 0.278, '#': 0.556, '$': 0.556, '%': 0.889, '&': 0.667,
|
||||
"'": 0.222, '(': 0.333, ')': 0.333, '*': 0.389, '+': 0.66, ',': 0.278, '-': 0.333,
|
||||
'.': 0.278, '/': 0.278, '0': 0.556, '1': 0.556, '2': 0.556, '3': 0.556, '4': 0.556,
|
||||
'5': 0.556, '6': 0.556, '7': 0.556, '8': 0.556, '9': 0.556, ':': 0.278, ';': 0.278,
|
||||
'<': 0.66, '=': 0.66, '>': 0.66, '?': 0.5, '@': 0.8, 'A': 0.667, 'B': 0.667, 'C':
|
||||
0.722, 'D': 0.722, 'E': 0.611, 'F': 0.556, 'G': 0.778, 'H': 0.722, 'I': 0.278, 'J':
|
||||
0.5, 'K': 0.667, 'L': 0.556, 'M': 0.833, 'N': 0.722, 'O': 0.778, 'P': 0.611, 'Q':
|
||||
0.778, 'R': 0.667, 'S': 0.611, 'T': 0.556, 'U': 0.722, 'V': 0.611, 'W': 0.889, 'X':
|
||||
0.611, 'Y': 0.611, 'Z': 0.611, '[': 0.333, '\\': 0.278, ']': 0.333, '_': 0.5, 'a':
|
||||
0.556, 'b': 0.611, 'c': 0.556, 'd': 0.611, 'e': 0.556, 'f': 0.278, 'g': 0.611, 'h':
|
||||
0.556, 'i': 0.222, 'j': 0.222, 'k': 0.5, 'l': 0.222, 'm': 0.833, 'n': 0.556, 'o':
|
||||
0.556, 'p': 0.611, 'q': 0.611, 'r': 0.333, 's': 0.5, 't': 0.278, 'u': 0.556, 'v':
|
||||
0.5, 'w': 0.722, 'x': 0.5, 'y': 0.5, 'z': 0.5, '{': 0.333, '|': 0.222, '}': 0.333,
|
||||
'~': 0.66, 'ª': 0.334, 'º': 0.334, 'À': 0.667, 'Á': 0.667, 'Â': 0.667, 'Ã': 0.667,
|
||||
'Ä': 0.667, 'Ç': 0.722, 'È': 0.611, 'É': 0.611, 'Ê': 0.611, 'Ë': 0.611, 'Ì': 0.278,
|
||||
'Í': 0.278, 'Î': 0.278, 'Ï': 0.278, 'Ñ': 0.722, 'Ò': 0.778, 'Ó': 0.778, 'Ô': 0.778,
|
||||
'Õ': 0.778, 'Ö': 0.778, 'Ù': 0.722, 'Ú': 0.722, 'Û': 0.722, 'Ü': 0.722, 'à': 0.556,
|
||||
'á': 0.556, 'â': 0.556, 'ã': 0.556, 'ä': 0.556, 'ç': 0.556, 'è': 0.556, 'é': 0.556,
|
||||
'ê': 0.556, 'ë': 0.556, 'ì': 0.222, 'í': 0.222, 'î': 0.222, 'ï': 0.222, 'ñ': 0.556,
|
||||
'ò': 0.556, 'ó': 0.556, 'ô': 0.556, 'õ': 0.556, 'ö': 0.556, 'ù': 0.556, 'ú': 0.556,
|
||||
'û': 0.556, 'ü': 0.556
|
||||
},
|
||||
'helvetica-oblique': {
|
||||
' ': 0.2778, '!': 0.2778, '"': 0.355, '#': 0.5562, '$': 0.5562, '%': 0.8892, '&':
|
||||
0.667, "'": 0.1909, '(': 0.333, ')': 0.333, '*': 0.3892, '+': 0.584, ',': 0.2778,
|
||||
'-': 0.333, '.': 0.2778, '/': 0.2778, '0': 0.5562, '1': 0.5562, '2': 0.5562, '3':
|
||||
0.5562, '4': 0.5562, '5': 0.5562, '6': 0.5562, '7': 0.5562, '8': 0.5562, '9':
|
||||
0.5562, ':': 0.2778, ';': 0.2778, '<': 0.584, '=': 0.584, '>': 0.584, '?': 0.5562,
|
||||
'@': 1.0151, 'A': 0.667, 'B': 0.667, 'C': 0.7222, 'D': 0.7222, 'E': 0.667, 'F':
|
||||
0.6108, 'G': 0.7778, 'H': 0.7222, 'I': 0.2778, 'J': 0.5, 'K': 0.667, 'L': 0.5562,
|
||||
'M': 0.833, 'N': 0.7222, 'O': 0.7778, 'P': 0.667, 'Q': 0.7778, 'R': 0.7222, 'S':
|
||||
0.667, 'T': 0.6108, 'U': 0.7222, 'V': 0.667, 'W': 0.9438, 'X': 0.667, 'Y': 0.667,
|
||||
'Z': 0.6108, '[': 0.2778, '\\': 0.2778, ']': 0.2778, '_': 0.5562, 'a': 0.5562, 'b':
|
||||
0.5562, 'c': 0.5, 'd': 0.5562, 'e': 0.5562, 'f': 0.2778, 'g': 0.5562, 'h': 0.5562,
|
||||
'i': 0.2222, 'j': 0.2222, 'k': 0.5, 'l': 0.2222, 'm': 0.833, 'n': 0.5562, 'o':
|
||||
0.5562, 'p': 0.5562, 'q': 0.5562, 'r': 0.333, 's': 0.5, 't': 0.2778, 'u': 0.5562,
|
||||
'v': 0.5, 'w': 0.7222, 'x': 0.5, 'y': 0.5, 'z': 0.5, '{': 0.334, '|': 0.2598, '}':
|
||||
0.334, '~': 0.584, 'ª': 0.3701, 'º': 0.3652, 'À': 0.667, 'Á': 0.667, 'Â': 0.667,
|
||||
'Ã': 0.667, 'Ä': 0.667, 'Ç': 0.7222, 'È': 0.667, 'É': 0.667, 'Ê': 0.667, 'Ë': 0.667,
|
||||
'Ì': 0.2778, 'Í': 0.2778, 'Î': 0.2778, 'Ï': 0.2778, 'Ñ': 0.7222, 'Ò': 0.7778, 'Ó':
|
||||
0.7778, 'Ô': 0.7778, 'Õ': 0.7778, 'Ö': 0.7778, 'Ù': 0.7222, 'Ú': 0.7222, 'Û':
|
||||
0.7222, 'Ü': 0.7222, 'à': 0.5562, 'á': 0.5562, 'â': 0.5562, 'ã': 0.5562, 'ä':
|
||||
0.5562, 'ç': 0.5, 'è': 0.5562, 'é': 0.5562, 'ê': 0.5562, 'ë': 0.5562, 'ì': 0.2778,
|
||||
'í': 0.2778, 'î': 0.2778, 'ï': 0.2778, 'ñ': 0.5562, 'ò': 0.5562, 'ó': 0.5562, 'ô':
|
||||
0.5562, 'õ': 0.5562, 'ö': 0.5562, 'ù': 0.5562, 'ú': 0.5562, 'û': 0.5562, 'ü': 0.5562
|
||||
},
|
||||
'playfair display': {
|
||||
' ': 0.249, '!': 0.262, '"': 0.298, '#': 0.637, '$': 0.523, '%': 0.742, '&':
|
||||
0.837, "'": 0.189, '(': 0.294, ')': 0.294, '*': 0.492, '+': 0.587, ',': 0.261,
|
||||
'-': 0.499, '.': 0.252, '/': 0.372, '0': 0.6, '1': 0.37, '2': 0.479, '3': 0.451,
|
||||
'4': 0.494, '5': 0.415, '6': 0.517, '7': 0.406, '8': 0.52, '9': 0.509, ':':
|
||||
0.266, ';': 0.285, '<': 0.608, '=': 0.647, '>': 0.608, '?': 0.459, '@': 0.878,
|
||||
'A': 0.63, 'B': 0.623, 'C': 0.688, 'D': 0.723, 'E': 0.609, 'F': 0.569, 'G':
|
||||
0.703, 'H': 0.757, 'I': 0.339, 'J': 0.326, 'K': 0.655, 'L': 0.588, 'M': 0.871,
|
||||
'N': 0.706, 'O': 0.741, 'P': 0.585, 'Q': 0.741, 'R': 0.648, 'S': 0.54, 'T':
|
||||
0.621, 'U': 0.682, 'V': 0.63, 'W': 0.905, 'X': 0.634, 'Y': 0.591, 'Z': 0.594,
|
||||
'[': 0.296, '\\': 0.372, ']': 0.296, '_': 0.591, 'a': 0.498, 'b': 0.563, 'c':
|
||||
0.486, 'd': 0.581, 'e': 0.506, 'f': 0.332, 'g': 0.528, 'h': 0.588, 'i': 0.293,
|
||||
'j': 0.267, 'k': 0.55, 'l': 0.286, 'm': 0.888, 'n': 0.597, 'o': 0.548, 'p':
|
||||
0.581, 'q': 0.563, 'r': 0.445, 's': 0.447, 't': 0.354, 'u': 0.581, 'v': 0.489,
|
||||
'w': 0.77, 'x': 0.519, 'y': 0.504, 'z': 0.476, '{': 0.304, '|': 0.222, '}':
|
||||
0.304, '~': 0.649, 'ª': 0.463, 'º': 0.454, 'À': 0.63, 'Á': 0.63, 'Â': 0.63, 'Ã':
|
||||
0.63, 'Ä': 0.63, 'Ç': 0.688, 'È': 0.609, 'É': 0.609, 'Ê': 0.609, 'Ë': 0.609,
|
||||
'Ì': 0.339, 'Í': 0.339, 'Î': 0.339, 'Ï': 0.339, 'Ñ': 0.706, 'Ò': 0.741, 'Ó':
|
||||
0.741, 'Ô': 0.741, 'Õ': 0.741, 'Ö': 0.741, 'Ù': 0.682, 'Ú': 0.682, 'Û': 0.682,
|
||||
'Ü': 0.682, 'à': 0.498, 'á': 0.498, 'â': 0.498, 'ã': 0.498, 'ä': 0.498, 'ç':
|
||||
0.486, 'è': 0.506, 'é': 0.506, 'ê': 0.506, 'ë': 0.506, 'ì': 0.293, 'í': 0.293,
|
||||
'î': 0.293, 'ï': 0.293, 'ñ': 0.597, 'ò': 0.548, 'ó': 0.548, 'ô': 0.548, 'õ':
|
||||
0.548, 'ö': 0.548, 'ù': 0.581, 'ú': 0.581, 'û': 0.581, 'ü': 0.581
|
||||
},
|
||||
'playfair display-italic': {
|
||||
' ': 0.253, '!': 0.263, '"': 0.296, '#': 0.639, '$': 0.551, '%': 0.766, '&':
|
||||
0.817, "'": 0.187, '(': 0.361, ')': 0.361, '*': 0.495, '+': 0.585, ',': 0.252,
|
||||
'-': 0.504, '.': 0.244, '/': 0.468, '0': 0.565, '1': 0.342, '2': 0.503, '3':
|
||||
0.485, '4': 0.515, '5': 0.459, '6': 0.527, '7': 0.487, '8': 0.544, '9': 0.513,
|
||||
':': 0.278, ';': 0.287, '<': 0.602, '=': 0.663, '>': 0.602, '?': 0.438, '@':
|
||||
0.848, 'A': 0.629, 'B': 0.632, 'C': 0.67, 'D': 0.725, 'E': 0.61, 'F': 0.566,
|
||||
'G': 0.705, 'H': 0.755, 'I': 0.341, 'J': 0.636, 'K': 0.675, 'L': 0.586, 'M':
|
||||
0.868, 'N': 0.77, 'O': 0.729, 'P': 0.589, 'Q': 0.781, 'R': 0.664, 'S': 0.554,
|
||||
'T': 0.659, 'U': 0.682, 'V': 0.632, 'W': 0.902, 'X': 0.699, 'Y': 0.67, 'Z':
|
||||
0.652, '[': 0.3, '\\': 0.492, ']': 0.3, '_': 0.744, 'a': 0.523, 'b': 0.504, 'c':
|
||||
0.432, 'd': 0.542, 'e': 0.429, 'f': 0.321, 'g': 0.449, 'h': 0.536, 'i': 0.302,
|
||||
'j': 0.276, 'k': 0.515, 'l': 0.277, 'm': 0.809, 'n': 0.58, 'o': 0.488, 'p':
|
||||
0.551, 'q': 0.507, 'r': 0.426, 's': 0.381, 't': 0.336, 'u': 0.567, 'v': 0.533,
|
||||
'w': 0.685, 'x': 0.526, 'y': 0.481, 'z': 0.445, '{': 0.306, '|': 0.318, '}':
|
||||
0.306, '~': 0.649, 'ª': 0.454, 'º': 0.411, 'À': 0.629, 'Á': 0.629, 'Â': 0.629,
|
||||
'Ã': 0.629, 'Ä': 0.629, 'Ç': 0.67, 'È': 0.61, 'É': 0.61, 'Ê': 0.61, 'Ë': 0.61,
|
||||
'Ì': 0.341, 'Í': 0.341, 'Î': 0.341, 'Ï': 0.341, 'Ñ': 0.77, 'Ò': 0.729, 'Ó':
|
||||
0.729, 'Ô': 0.729, 'Õ': 0.729, 'Ö': 0.729, 'Ù': 0.682, 'Ú': 0.682, 'Û': 0.682,
|
||||
'Ü': 0.682, 'à': 0.523, 'á': 0.523, 'â': 0.523, 'ã': 0.523, 'ä': 0.523, 'ç':
|
||||
0.432, 'è': 0.429, 'é': 0.429, 'ê': 0.429, 'ë': 0.429, 'ì': 0.302, 'í': 0.302,
|
||||
'î': 0.302, 'ï': 0.302, 'ñ': 0.58, 'ò': 0.488, 'ó': 0.488, 'ô': 0.488, 'õ':
|
||||
0.488, 'ö': 0.488, 'ù': 0.567, 'ú': 0.567, 'û': 0.567, 'ü': 0.567
|
||||
},
|
||||
'playfair display-medium italic': {
|
||||
' ': 0.254, '!': 0.28, '"': 0.328, '#': 0.634, '$': 0.571, '%': 0.776, '&':
|
||||
0.832, "'": 0.201, '(': 0.373, ')': 0.373, '*': 0.491, '+': 0.582, ',': 0.264,
|
||||
'-': 0.505, '.': 0.256, '/': 0.466, '0': 0.584, '1': 0.352, '2': 0.514, '3':
|
||||
0.496, '4': 0.522, '5': 0.463, '6': 0.54, '7': 0.492, '8': 0.554, '9': 0.527,
|
||||
':': 0.289, ';': 0.297, '<': 0.599, '=': 0.659, '>': 0.6, '?': 0.454, '@':
|
||||
0.853, 'A': 0.637, 'B': 0.648, 'C': 0.678, 'D': 0.74, 'E': 0.615, 'F': 0.573,
|
||||
'G': 0.716, 'H': 0.773, 'I': 0.357, 'J': 0.652, 'K': 0.697, 'L': 0.592, 'M':
|
||||
0.884, 'N': 0.788, 'O': 0.745, 'P': 0.607, 'Q': 0.798, 'R': 0.68, 'S': 0.565,
|
||||
'T': 0.666, 'U': 0.686, 'V': 0.642, 'W': 0.929, 'X': 0.713, 'Y': 0.691, 'Z':
|
||||
0.654, '[': 0.321, '\\': 0.492, ']': 0.321, '_': 0.744, 'a': 0.535, 'b': 0.518,
|
||||
'c': 0.442, 'd': 0.554, 'e': 0.442, 'f': 0.328, 'g': 0.457, 'h': 0.547, 'i':
|
||||
0.309, 'j': 0.284, 'k': 0.527, 'l': 0.285, 'm': 0.825, 'n': 0.588, 'o': 0.502,
|
||||
'p': 0.562, 'q': 0.522, 'r': 0.441, 's': 0.394, 't': 0.342, 'u': 0.577, 'v':
|
||||
0.53, 'w': 0.683, 'x': 0.539, 'y': 0.481, 'z': 0.451, '{': 0.324, '|': 0.335,
|
||||
'}': 0.324, '~': 0.648, 'ª': 0.48, 'º': 0.438, 'À': 0.637, 'Á': 0.637, 'Â':
|
||||
0.637, 'Ã': 0.637, 'Ä': 0.637, 'Ç': 0.678, 'È': 0.615, 'É': 0.615, 'Ê': 0.615,
|
||||
'Ë': 0.615, 'Ì': 0.357, 'Í': 0.357, 'Î': 0.357, 'Ï': 0.357, 'Ñ': 0.788, 'Ò':
|
||||
0.745, 'Ó': 0.745, 'Ô': 0.745, 'Õ': 0.745, 'Ö': 0.745, 'Ù': 0.686, 'Ú': 0.686,
|
||||
'Û': 0.686, 'Ü': 0.686, 'à': 0.535, 'á': 0.535, 'â': 0.535, 'ã': 0.535, 'ä':
|
||||
0.535, 'ç': 0.442, 'è': 0.442, 'é': 0.442, 'ê': 0.442, 'ë': 0.442, 'ì': 0.309,
|
||||
'í': 0.309, 'î': 0.309, 'ï': 0.309, 'ñ': 0.588, 'ò': 0.502, 'ó': 0.502, 'ô':
|
||||
0.502, 'õ': 0.502, 'ö': 0.502, 'ù': 0.577, 'ú': 0.577, 'û': 0.577, 'ü': 0.577
|
||||
},
|
||||
'playfair display-semibold italic': {
|
||||
' ': 0.255, '!': 0.297, '"': 0.359, '#': 0.63, '$': 0.59, '%': 0.786, '&':
|
||||
0.848, "'": 0.215, '(': 0.385, ')': 0.385, '*': 0.488, '+': 0.579, ',': 0.275,
|
||||
'-': 0.505, '.': 0.267, '/': 0.463, '0': 0.603, '1': 0.362, '2': 0.524, '3':
|
||||
0.507, '4': 0.529, '5': 0.467, '6': 0.553, '7': 0.498, '8': 0.564, '9': 0.541,
|
||||
':': 0.3, ';': 0.307, '<': 0.596, '=': 0.654, '>': 0.598, '?': 0.47, '@': 0.858,
|
||||
'A': 0.645, 'B': 0.663, 'C': 0.685, 'D': 0.756, 'E': 0.621, 'F': 0.581, 'G':
|
||||
0.727, 'H': 0.791, 'I': 0.373, 'J': 0.668, 'K': 0.718, 'L': 0.598, 'M': 0.9,
|
||||
'N': 0.806, 'O': 0.761, 'P': 0.625, 'Q': 0.815, 'R': 0.695, 'S': 0.576, 'T':
|
||||
0.673, 'U': 0.69, 'V': 0.651, 'W': 0.955, 'X': 0.727, 'Y': 0.712, 'Z': 0.656,
|
||||
'[': 0.342, '\\': 0.491, ']': 0.342, '_': 0.744, 'a': 0.547, 'b': 0.532, 'c':
|
||||
0.453, 'd': 0.566, 'e': 0.456, 'f': 0.335, 'g': 0.464, 'h': 0.558, 'i': 0.316,
|
||||
'j': 0.293, 'k': 0.54, 'l': 0.293, 'm': 0.84, 'n': 0.596, 'o': 0.516, 'p':
|
||||
0.573, 'q': 0.537, 'r': 0.455, 's': 0.407, 't': 0.348, 'u': 0.587, 'v': 0.527,
|
||||
'w': 0.681, 'x': 0.552, 'y': 0.482, 'z': 0.457, '{': 0.343, '|': 0.352, '}':
|
||||
0.342, '~': 0.647, 'ª': 0.506, 'º': 0.465, 'À': 0.645, 'Á': 0.645, 'Â': 0.645,
|
||||
'Ã': 0.645, 'Ä': 0.645, 'Ç': 0.685, 'È': 0.621, 'É': 0.621, 'Ê': 0.621, 'Ë':
|
||||
0.621, 'Ì': 0.373, 'Í': 0.373, 'Î': 0.373, 'Ï': 0.373, 'Ñ': 0.806, 'Ò': 0.761,
|
||||
'Ó': 0.761, 'Ô': 0.761, 'Õ': 0.761, 'Ö': 0.761, 'Ù': 0.69, 'Ú': 0.69, 'Û': 0.69,
|
||||
'Ü': 0.69, 'à': 0.547, 'á': 0.547, 'â': 0.547, 'ã': 0.547, 'ä': 0.547, 'ç':
|
||||
0.453, 'è': 0.456, 'é': 0.456, 'ê': 0.456, 'ë': 0.456, 'ì': 0.315, 'í': 0.315,
|
||||
'î': 0.315, 'ï': 0.315, 'ñ': 0.596, 'ò': 0.516, 'ó': 0.516, 'ô': 0.516, 'õ':
|
||||
0.516, 'ö': 0.516, 'ù': 0.587, 'ú': 0.587, 'û': 0.587, 'ü': 0.587
|
||||
},
|
||||
'playfair display-bold italic': {
|
||||
' ': 0.255, '!': 0.314, '"': 0.391, '#': 0.625, '$': 0.61, '%': 0.796, '&':
|
||||
0.863, "'": 0.228, '(': 0.396, ')': 0.397, '*': 0.484, '+': 0.575, ',': 0.287,
|
||||
'-': 0.506, '.': 0.279, '/': 0.461, '0': 0.621, '1': 0.371, '2': 0.535, '3':
|
||||
0.517, '4': 0.535, '5': 0.472, '6': 0.567, '7': 0.503, '8': 0.575, '9': 0.554,
|
||||
':': 0.311, ';': 0.318, '<': 0.594, '=': 0.65, '>': 0.595, '?': 0.486, '@':
|
||||
0.863, 'A': 0.653, 'B': 0.679, 'C': 0.693, 'D': 0.771, 'E': 0.626, 'F': 0.588,
|
||||
'G': 0.739, 'H': 0.81, 'I': 0.39, 'J': 0.683, 'K': 0.74, 'L': 0.605, 'M': 0.917,
|
||||
'N': 0.825, 'O': 0.777, 'P': 0.643, 'Q': 0.831, 'R': 0.711, 'S': 0.587, 'T':
|
||||
0.681, 'U': 0.694, 'V': 0.661, 'W': 0.982, 'X': 0.741, 'Y': 0.734, 'Z': 0.658,
|
||||
'[': 0.362, '\\': 0.491, ']': 0.363, '_': 0.744, 'a': 0.558, 'b': 0.547, 'c':
|
||||
0.463, 'd': 0.579, 'e': 0.469, 'f': 0.341, 'g': 0.472, 'h': 0.57, 'i': 0.322,
|
||||
'j': 0.301, 'k': 0.552, 'l': 0.3, 'm': 0.856, 'n': 0.603, 'o': 0.529, 'p':
|
||||
0.583, 'q': 0.551, 'r': 0.47, 's': 0.42, 't': 0.353, 'u': 0.597, 'v': 0.525,
|
||||
'w': 0.68, 'x': 0.565, 'y': 0.482, 'z': 0.464, '{': 0.361, '|': 0.369, '}':
|
||||
0.361, '~': 0.645, 'ª': 0.532, 'º': 0.491, 'À': 0.653, 'Á': 0.653, 'Â': 0.653,
|
||||
'Ã': 0.653, 'Ä': 0.653, 'Ç': 0.693, 'È': 0.626, 'É': 0.626, 'Ê': 0.626, 'Ë':
|
||||
0.626, 'Ì': 0.39, 'Í': 0.39, 'Î': 0.39, 'Ï': 0.39, 'Ñ': 0.825, 'Ò': 0.777, 'Ó':
|
||||
0.777, 'Ô': 0.777, 'Õ': 0.777, 'Ö': 0.777, 'Ù': 0.694, 'Ú': 0.694, 'Û': 0.694,
|
||||
'Ü': 0.694, 'à': 0.558, 'á': 0.558, 'â': 0.558, 'ã': 0.558, 'ä': 0.558, 'ç':
|
||||
0.463, 'è': 0.469, 'é': 0.469, 'ê': 0.469, 'ë': 0.469, 'ì': 0.322, 'í': 0.322,
|
||||
'î': 0.322, 'ï': 0.322, 'ñ': 0.603, 'ò': 0.529, 'ó': 0.529, 'ô': 0.529, 'õ':
|
||||
0.529, 'ö': 0.529, 'ù': 0.597, 'ú': 0.597, 'û': 0.597, 'ü': 0.597
|
||||
},
|
||||
'playfair display-bold': {
|
||||
' ': 0.233, '!': 0.288, '"': 0.369, '#': 0.62, '$': 0.574, '%': 0.728, '&':
|
||||
0.898, "'": 0.209, '(': 0.318, ')': 0.318, '*': 0.481, '+': 0.574, ',': 0.271,
|
||||
'-': 0.481, '.': 0.27, '/': 0.368, '0': 0.645, '1': 0.383, '2': 0.523, '3':
|
||||
0.488, '4': 0.528, '5': 0.459, '6': 0.561, '7': 0.455, '8': 0.561, '9': 0.553,
|
||||
':': 0.29, ';': 0.296, '<': 0.602, '=': 0.635, '>': 0.602, '?': 0.508, '@':
|
||||
0.884, 'A': 0.667, 'B': 0.672, 'C': 0.705, 'D': 0.776, 'E': 0.634, 'F': 0.592,
|
||||
'G': 0.735, 'H': 0.8, 'I': 0.376, 'J': 0.364, 'K': 0.714, 'L': 0.611, 'M':
|
||||
0.927, 'N': 0.722, 'O': 0.793, 'P': 0.641, 'Q': 0.793, 'R': 0.692, 'S': 0.584,
|
||||
'T': 0.668, 'U': 0.689, 'V': 0.668, 'W': 0.968, 'X': 0.691, 'Y': 0.623, 'Z':
|
||||
0.602, '[': 0.322, '\\': 0.368, ']': 0.322, '_': 0.591, 'a': 0.518, 'b': 0.586,
|
||||
'c': 0.496, 'd': 0.598, 'e': 0.504, 'f': 0.352, 'g': 0.553, 'h': 0.608, 'i':
|
||||
0.315, 'j': 0.297, 'k': 0.585, 'l': 0.311, 'm': 0.905, 'n': 0.614, 'o': 0.565,
|
||||
'p': 0.6, 'q': 0.585, 'r': 0.467, 's': 0.462, 't': 0.352, 'u': 0.606, 'v':
|
||||
0.508, 'w': 0.787, 'x': 0.536, 'y': 0.519, 'z': 0.474, '{': 0.333, '|': 0.253,
|
||||
'}': 0.333, '~': 0.668, 'ª': 0.482, 'º': 0.468, 'À': 0.667, 'Á': 0.667, 'Â':
|
||||
0.667, 'Ã': 0.667, 'Ä': 0.667, 'Ç': 0.705, 'È': 0.634, 'É': 0.634, 'Ê': 0.634,
|
||||
'Ë': 0.634, 'Ì': 0.376, 'Í': 0.376, 'Î': 0.376, 'Ï': 0.376, 'Ñ': 0.722, 'Ò':
|
||||
0.793, 'Ó': 0.793, 'Ô': 0.793, 'Õ': 0.793, 'Ö': 0.793, 'Ù': 0.689, 'Ú': 0.689,
|
||||
'Û': 0.689, 'Ü': 0.689, 'à': 0.518, 'á': 0.518, 'â': 0.518, 'ã': 0.518, 'ä':
|
||||
0.518, 'ç': 0.496, 'è': 0.504, 'é': 0.504, 'ê': 0.504, 'ë': 0.504, 'ì': 0.315,
|
||||
'í': 0.315, 'î': 0.315, 'ï': 0.315, 'ñ': 0.614, 'ò': 0.565, 'ó': 0.565, 'ô':
|
||||
0.565, 'õ': 0.565, 'ö': 0.565, 'ù': 0.606, 'ú': 0.606, 'û': 0.606, 'ü': 0.606
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
# Vertical extents, also measured off the font files, as a fraction of the em:
|
||||
# ``ascent``/``descent`` are the line box Final Cut centres on the title's
|
||||
# Position when vertical alignment is Middle; the rest are the real INK
|
||||
# extremes of the glyph classes a line can contain. Horizontal metrics keep
|
||||
# two words from colliding side by side; these keep two LINES from colliding,
|
||||
# which cap-height alone could not — a display italic's accents and descenders
|
||||
# reach far past it (Playfair's ç bottoms out at -0.241 em, its à tops out at
|
||||
# 1.007), and that is exactly the pair that touched in the reference layout.
|
||||
VERTICAL_METRICS = {
|
||||
'helvetica neue': {
|
||||
'ascent': 0.9520, 'descent': -0.2130,
|
||||
'cap': 0.7140, 'x_height': 0.5290, 'ascender': 0.7220,
|
||||
'accent_upper': 0.9470, 'accent_lower': 0.7700, 'descender': -0.2090,
|
||||
},
|
||||
'helvetica neue-bold': {
|
||||
'ascent': 0.9750, 'descent': -0.2170,
|
||||
'cap': 0.7140, 'x_height': 0.5310, 'ascender': 0.7140,
|
||||
'accent_upper': 0.9550, 'accent_lower': 0.7850, 'descender': -0.2170,
|
||||
},
|
||||
'helvetica neue-italic': {
|
||||
'ascent': 0.9570, 'descent': -0.2130,
|
||||
'cap': 0.7140, 'x_height': 0.5290, 'ascender': 0.7220,
|
||||
'accent_upper': 0.8980, 'accent_lower': 0.7310, 'descender': -0.2090,
|
||||
},
|
||||
'helvetica neue-light': {
|
||||
'ascent': 0.9310, 'descent': -0.2130,
|
||||
'cap': 0.7140, 'x_height': 0.5261, 'ascender': 0.7140,
|
||||
'accent_upper': 0.9090, 'accent_lower': 0.7090, 'descender': -0.1880,
|
||||
},
|
||||
'helvetica': {
|
||||
'ascent': 0.7700, 'descent': -0.2300,
|
||||
'cap': 0.7173, 'x_height': 0.5381, 'ascender': 0.7275,
|
||||
'accent_upper': 0.9346, 'accent_lower': 0.7334, 'descender': -0.2251,
|
||||
},
|
||||
'helvetica-bold': {
|
||||
'ascent': 0.7700, 'descent': -0.2300,
|
||||
'cap': 0.7197, 'x_height': 0.5493, 'ascender': 0.7271,
|
||||
'accent_upper': 0.9453, 'accent_lower': 0.7500, 'descender': -0.2300,
|
||||
},
|
||||
'playfair display': {
|
||||
'ascent': 1.0820, 'descent': -0.2510,
|
||||
'cap': 0.7080, 'x_height': 0.5290, 'ascender': 0.7840,
|
||||
'accent_upper': 0.9362, 'accent_lower': 0.8204, 'descender': -0.1880,
|
||||
},
|
||||
'playfair display-bold': {
|
||||
'ascent': 1.0820, 'descent': -0.2510,
|
||||
'cap': 0.7090, 'x_height': 0.5310, 'ascender': 0.7830,
|
||||
'accent_upper': 0.9422, 'accent_lower': 0.8301, 'descender': -0.1880,
|
||||
},
|
||||
'playfair display-italic': {
|
||||
'ascent': 1.0820, 'descent': -0.2510,
|
||||
'cap': 0.8170, 'x_height': 0.5280, 'ascender': 0.7950,
|
||||
'accent_upper': 0.9365, 'accent_lower': 0.8204, 'descender': -0.1950,
|
||||
},
|
||||
'playfair display-medium italic': {
|
||||
'ascent': 1.0820, 'descent': -0.2510,
|
||||
'cap': 0.8310, 'x_height': 0.5290, 'ascender': 0.7950,
|
||||
'accent_upper': 0.9433, 'accent_lower': 0.8276, 'descender': -0.1940,
|
||||
},
|
||||
'playfair display-semibold italic': {
|
||||
'ascent': 1.0820, 'descent': -0.2510,
|
||||
'cap': 0.8450, 'x_height': 0.5300, 'ascender': 0.7950,
|
||||
'accent_upper': 0.9492, 'accent_lower': 0.8348, 'descender': -0.1920,
|
||||
},
|
||||
'playfair display-bold italic': {
|
||||
'ascent': 1.0820, 'descent': -0.2510,
|
||||
'cap': 0.8580, 'x_height': 0.5310, 'ascender': 0.7950,
|
||||
'accent_upper': 0.9563, 'accent_lower': 0.8427, 'descender': -0.1910,
|
||||
},
|
||||
}
|
||||
Executable
+273
@@ -0,0 +1,273 @@
|
||||
"""Live mode v1 — officially-supported control of a running Final Cut Pro.
|
||||
|
||||
This module is the first piece of the dual-mode (XML + Live) architecture
|
||||
(see docs/CAPABILITY-AUDIT-2026-06.md). Everything here rides Apple's
|
||||
sanctioned surfaces only — no injection, no private APIs, no accessibility
|
||||
scripting:
|
||||
|
||||
- **push_to_fcp** — programmatic FCPXML *import* via the Open Document
|
||||
Apple event, Apple's documented zero-click ingestion path. Behavior is
|
||||
steered by an ``<import-options>`` element injected into the document
|
||||
(library location, suppress warnings, copy assets).
|
||||
|
||||
Live-verified findings (FCP 12.2, 2026-06-11) that shape the contract:
|
||||
|
||||
1. **Truly zero-click requires a ``.fcpbundle`` library location.** Given
|
||||
a new ``.fcpbundle`` path, FCP silently creates the library + a dated
|
||||
event and imports. With NO location, or a ``.fcplibrary``/bare path,
|
||||
FCP raises a modal "Open Library" picker — a *required choice* that
|
||||
``suppress warnings`` does not cover — and the Apple event blocks until
|
||||
a human answers. ``inject_import_options`` therefore normalises the
|
||||
location to ``.fcpbundle``.
|
||||
2. **Media-identity collisions** — importing a project whose media already
|
||||
exists in the target library fails with "the media already exists with a
|
||||
unique identifier". Push into a fresh library, or reuse the exact asset
|
||||
IDs FCP already holds.
|
||||
- **list_fcp_libraries** — FCP 12's AppleScript dictionary is read-only
|
||||
library inspection (suite ``com.apple.FinalCut.library.inspection``);
|
||||
we use it to enumerate open libraries → events → projects.
|
||||
|
||||
The asymmetry is structural: import is scriptable, but Apple offers NO
|
||||
programmatic export — reading back the user's current timeline still
|
||||
requires a manual File > Export XML. Live mode therefore *pushes*;
|
||||
round-trips come back through the XML tools.
|
||||
|
||||
macOS notes: ``osascript`` targeting Final Cut Pro requires the host
|
||||
process to hold an Apple Events automation grant (System Settings →
|
||||
Privacy & Security → Automation) — the first call triggers the consent
|
||||
prompt. ``tell application "Final Cut Pro"`` launches FCP if it is not
|
||||
already running; ``list_fcp_libraries`` checks first and declines to
|
||||
launch, while ``push_to_fcp`` launching FCP is the point.
|
||||
"""
|
||||
|
||||
import subprocess
|
||||
import xml.etree.ElementTree as ET
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
_OSASCRIPT_TIMEOUT_SECONDS = 120 # FCP cold-launch + import can be slow
|
||||
_FCP_BUNDLE_ID = "com.apple.FinalCut"
|
||||
|
||||
# Field separators for AppleScript list output — ASCII unit/record
|
||||
# separators cannot appear in user-facing library/project names.
|
||||
_FIELD_SEP = "\x1f"
|
||||
_RECORD_SEP = "\x1e"
|
||||
|
||||
|
||||
def _applescript_quote(value: str) -> str:
|
||||
"""Escape a string for embedding in a double-quoted AppleScript literal."""
|
||||
return value.replace("\\", "\\\\").replace('"', '\\"')
|
||||
|
||||
|
||||
def _run_osascript(script: str) -> subprocess.CompletedProcess:
|
||||
return subprocess.run(
|
||||
["osascript", "-e", script],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=_OSASCRIPT_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
|
||||
def fcp_is_running() -> bool:
|
||||
"""True when a Final Cut Pro process is active (no launch side-effect)."""
|
||||
proc = subprocess.run(
|
||||
["pgrep", "-x", "Final Cut Pro"], capture_output=True, text=True
|
||||
)
|
||||
return proc.returncode == 0
|
||||
|
||||
|
||||
def inject_import_options(
|
||||
fcpxml_path: str,
|
||||
output_path: str,
|
||||
library_location: Optional[str] = None,
|
||||
suppress_warnings: bool = True,
|
||||
copy_assets: Optional[bool] = None,
|
||||
) -> str:
|
||||
"""Write a copy of *fcpxml_path* with an ``<import-options>`` element.
|
||||
|
||||
The DTD requires ``import-options`` as the FIRST child of ``<fcpxml>``
|
||||
(``<!ELEMENT fcpxml (import-options?, resources?, ...)>``). Any
|
||||
existing import-options element is replaced.
|
||||
|
||||
Args:
|
||||
fcpxml_path: Source ``.fcpxml`` document.
|
||||
output_path: Where to write the import-ready copy.
|
||||
library_location: Path or ``file://`` URL of the target library
|
||||
(FCP creates the library if none exists there).
|
||||
suppress_warnings: Suppress non-fatal import warning dialogs.
|
||||
copy_assets: True = copy media into the library, False = link
|
||||
in place, None = let FCP use its default.
|
||||
|
||||
Returns:
|
||||
*output_path*.
|
||||
"""
|
||||
from .safe_xml import safe_parse
|
||||
from .writer import write_fcpxml
|
||||
|
||||
tree = safe_parse(fcpxml_path)
|
||||
root = tree.getroot()
|
||||
|
||||
for stale in root.findall('import-options'):
|
||||
root.remove(stale)
|
||||
|
||||
options = ET.Element('import-options')
|
||||
if library_location:
|
||||
loc = library_location
|
||||
if not loc.startswith('file://'):
|
||||
from urllib.parse import quote
|
||||
resolved = Path(loc).expanduser()
|
||||
# Live-verified on FCP 12.2: FCP auto-creates a library ONLY when
|
||||
# the location carries the .fcpbundle extension. A bare directory
|
||||
# or a .fcplibrary path triggers the modal "Open Library" picker
|
||||
# (a required choice that `suppress warnings` does NOT dismiss),
|
||||
# which blocks the Apple event. Normalise to .fcpbundle.
|
||||
if resolved.suffix.lower() != '.fcpbundle':
|
||||
resolved = resolved.with_suffix('.fcpbundle')
|
||||
loc = 'file://' + quote(str(resolved.resolve()))
|
||||
ET.SubElement(options, 'option', key='library location', value=loc)
|
||||
ET.SubElement(
|
||||
options, 'option',
|
||||
key='suppress warnings', value='1' if suppress_warnings else '0',
|
||||
)
|
||||
if copy_assets is not None:
|
||||
ET.SubElement(
|
||||
options, 'option',
|
||||
key='copy assets', value='1' if copy_assets else '0',
|
||||
)
|
||||
root.insert(0, options)
|
||||
|
||||
write_fcpxml(root, output_path)
|
||||
return output_path
|
||||
|
||||
|
||||
def push_to_fcp(
|
||||
fcpxml_path: str,
|
||||
library_location: Optional[str] = None,
|
||||
suppress_warnings: bool = True,
|
||||
copy_assets: Optional[bool] = None,
|
||||
import_copy_path: Optional[str] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Send an FCPXML document to Final Cut Pro via the Open Document event.
|
||||
|
||||
This is Apple's documented programmatic-import path: FCP ingests the
|
||||
document without any clicks, creating libraries/events as directed by
|
||||
``<import-options>``. Launches FCP when it isn't running.
|
||||
|
||||
Args:
|
||||
fcpxml_path: ``.fcpxml`` file or ``.fcpxmld`` bundle to import.
|
||||
library_location: Target library path/URL (created if absent).
|
||||
suppress_warnings: Suppress non-fatal import warning dialogs.
|
||||
copy_assets: Copy media into the library vs. link in place.
|
||||
import_copy_path: Where to write the options-injected copy for
|
||||
flat files (defaults handled by the caller; required when
|
||||
options are used on a flat file).
|
||||
|
||||
Returns:
|
||||
Dict with ``sent`` (path actually opened), ``launched_fcp``
|
||||
(whether FCP was started by this call), and ``stdout``.
|
||||
|
||||
Raises:
|
||||
RuntimeError: When osascript fails (most commonly a missing
|
||||
Automation permission grant for the host process).
|
||||
"""
|
||||
path = Path(fcpxml_path)
|
||||
send_path = path
|
||||
|
||||
# Bundles: open directly (option injection inside a copied bundle is
|
||||
# a v0.10 refinement); flat files get an import-ready copy so the
|
||||
# user's original is never touched.
|
||||
if path.suffix.lower() != '.fcpxmld' and import_copy_path:
|
||||
send_path = Path(
|
||||
inject_import_options(
|
||||
str(path),
|
||||
import_copy_path,
|
||||
library_location=library_location,
|
||||
suppress_warnings=suppress_warnings,
|
||||
copy_assets=copy_assets,
|
||||
)
|
||||
)
|
||||
|
||||
was_running = fcp_is_running()
|
||||
script = (
|
||||
'tell application "Final Cut Pro"\n'
|
||||
'activate\n'
|
||||
f'open POSIX file "{_applescript_quote(str(send_path.resolve()))}"\n'
|
||||
'end tell'
|
||||
)
|
||||
proc = _run_osascript(script)
|
||||
if proc.returncode != 0:
|
||||
stderr = proc.stderr.strip()
|
||||
hint = ""
|
||||
if "-1743" in stderr or "not allowed" in stderr.lower():
|
||||
hint = (
|
||||
" — grant Automation permission: System Settings → "
|
||||
"Privacy & Security → Automation → allow your terminal/MCP "
|
||||
"host to control Final Cut Pro, then retry"
|
||||
)
|
||||
raise RuntimeError(f"osascript failed: {stderr}{hint}")
|
||||
|
||||
return {
|
||||
"sent": str(send_path),
|
||||
"launched_fcp": not was_running,
|
||||
"stdout": proc.stdout.strip(),
|
||||
}
|
||||
|
||||
|
||||
def list_fcp_libraries(allow_launch: bool = False) -> List[Dict[str, Any]]:
|
||||
"""Enumerate open libraries → events → projects via AppleScript.
|
||||
|
||||
Uses FCP 12's read-only scripting dictionary. By default this
|
||||
refuses to launch FCP (``tell application`` would start it);
|
||||
pass ``allow_launch=True`` to override.
|
||||
|
||||
Returns:
|
||||
List of ``{name, events: [{name, projects: [str, ...]}, ...]}``.
|
||||
|
||||
Raises:
|
||||
RuntimeError: FCP not running (and *allow_launch* False), or
|
||||
osascript failure.
|
||||
"""
|
||||
if not allow_launch and not fcp_is_running():
|
||||
raise RuntimeError(
|
||||
"Final Cut Pro is not running (pass allow_launch=true to start it)"
|
||||
)
|
||||
|
||||
script = (
|
||||
'set fieldSep to (ASCII character 31)\n'
|
||||
'set recSep to (ASCII character 30)\n'
|
||||
'set out to ""\n'
|
||||
'tell application "Final Cut Pro"\n'
|
||||
' repeat with lib in libraries\n'
|
||||
' set libName to name of lib\n'
|
||||
' repeat with evt in (events of lib)\n'
|
||||
' set evtName to name of evt\n'
|
||||
' set projNames to ""\n'
|
||||
' repeat with proj in (projects of evt)\n'
|
||||
' set projNames to projNames & (name of proj) & fieldSep\n'
|
||||
' end repeat\n'
|
||||
' set out to out & libName & fieldSep & evtName & fieldSep '
|
||||
'& projNames & recSep\n'
|
||||
' end repeat\n'
|
||||
' if (count of events of lib) is 0 then\n'
|
||||
' set out to out & libName & fieldSep & recSep\n'
|
||||
' end if\n'
|
||||
' end repeat\n'
|
||||
'end tell\n'
|
||||
'return out'
|
||||
)
|
||||
proc = _run_osascript(script)
|
||||
if proc.returncode != 0:
|
||||
raise RuntimeError(f"osascript failed: {proc.stderr.strip()}")
|
||||
|
||||
libraries: Dict[str, Dict[str, Any]] = {}
|
||||
for record in proc.stdout.strip().split(_RECORD_SEP):
|
||||
record = record.strip('\n')
|
||||
if not record:
|
||||
continue
|
||||
fields = record.split(_FIELD_SEP)
|
||||
lib_name = fields[0]
|
||||
lib = libraries.setdefault(lib_name, {"name": lib_name, "events": []})
|
||||
if len(fields) >= 2 and fields[1]:
|
||||
projects = [p for p in fields[2:] if p]
|
||||
lib["events"].append({"name": fields[1], "projects": projects})
|
||||
return list(libraries.values())
|
||||
Executable
+177
@@ -0,0 +1,177 @@
|
||||
"""Media intelligence — real analysis of source media referenced by timelines.
|
||||
|
||||
v0.10 slice 1: audio silence detection via ffmpeg's silencedetect filter.
|
||||
No new Python dependencies: ffmpeg is invoked as a bounded subprocess
|
||||
(list-form arguments, validated numeric parameters, hard timeout), and
|
||||
detection degrades gracefully — ``detect_silence`` returns ``None`` when
|
||||
ffmpeg is unavailable or the file cannot be analyzed, so callers can fall
|
||||
back or report instead of crashing.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Hard ceiling on a single ffmpeg analysis pass. Decoding audio-only is far
|
||||
# faster than realtime, so this covers multi-hour media while still bounding
|
||||
# an adversarial/corrupt file that makes the decoder hang.
|
||||
PROBE_TIMEOUT_SECONDS = 120
|
||||
|
||||
# silencedetect prints times as plain seconds on stderr; starts can be
|
||||
# slightly negative (encoder priming samples), so allow a leading minus.
|
||||
_SILENCE_START_RE = re.compile(r"silence_start:\s*(-?\d+(?:\.\d+)?)")
|
||||
_SILENCE_END_RE = re.compile(r"silence_end:\s*(-?\d+(?:\.\d+)?)")
|
||||
_DURATION_RE = re.compile(r"Duration:\s*(\d+):(\d\d):(\d\d(?:\.\d+)?)")
|
||||
|
||||
|
||||
def parse_silencedetect_output(
|
||||
stderr: str, total_duration: Optional[float] = None
|
||||
) -> List[Tuple[float, float]]:
|
||||
"""Parse ffmpeg silencedetect stderr into (start, end) ranges in seconds.
|
||||
|
||||
A trailing ``silence_start`` with no matching ``silence_end`` (media that
|
||||
ends silent) is closed at ``total_duration`` when known, otherwise dropped.
|
||||
"""
|
||||
ranges: List[Tuple[float, float]] = []
|
||||
pending: Optional[float] = None
|
||||
for line in stderr.splitlines():
|
||||
start_match = _SILENCE_START_RE.search(line)
|
||||
if start_match:
|
||||
pending = max(0.0, float(start_match.group(1)))
|
||||
continue
|
||||
end_match = _SILENCE_END_RE.search(line)
|
||||
if end_match and pending is not None:
|
||||
ranges.append((pending, float(end_match.group(1))))
|
||||
pending = None
|
||||
if pending is not None and total_duration is not None and total_duration > pending:
|
||||
ranges.append((pending, total_duration))
|
||||
return ranges
|
||||
|
||||
|
||||
def map_silence_to_timeline(
|
||||
silences: List[Tuple[float, float]],
|
||||
source_start: float,
|
||||
clip_duration: float,
|
||||
timeline_offset: float,
|
||||
) -> List[Tuple[float, float]]:
|
||||
"""Map source-time silence ranges onto the timeline.
|
||||
|
||||
A clip uses ``[source_start, source_start + clip_duration)`` of its source
|
||||
media and sits at ``timeline_offset``. Ranges outside the used window are
|
||||
excluded; ranges overlapping its edges are clamped.
|
||||
"""
|
||||
source_end = source_start + clip_duration
|
||||
mapped: List[Tuple[float, float]] = []
|
||||
for start, end in silences:
|
||||
clamped_start = max(start, source_start)
|
||||
clamped_end = min(end, source_end)
|
||||
if clamped_end <= clamped_start:
|
||||
continue
|
||||
mapped.append((
|
||||
timeline_offset + (clamped_start - source_start),
|
||||
timeline_offset + (clamped_end - source_start),
|
||||
))
|
||||
return mapped
|
||||
|
||||
|
||||
def detect_beats(
|
||||
path: str, max_analysis_seconds: float = 1200.0
|
||||
) -> Optional[dict]:
|
||||
"""Detect musical beats in an audio file via librosa's beat tracker.
|
||||
|
||||
librosa is an optional dependency (``pip install 'fcp-mcp-server[intelligence]'``);
|
||||
without it, or when the file is missing/unreadable, this returns ``None``
|
||||
so callers can degrade to a helpful message instead of crashing.
|
||||
|
||||
Returns:
|
||||
``{'bpm': float, 'beats': [seconds, ...]}`` or ``None``.
|
||||
"""
|
||||
file_path = Path(path)
|
||||
if not file_path.is_file():
|
||||
return None
|
||||
try:
|
||||
import librosa
|
||||
except ImportError:
|
||||
logger.info("librosa not installed; beat detection unavailable")
|
||||
return None
|
||||
try:
|
||||
# duration cap bounds memory on adversarially long media
|
||||
y, sr = librosa.load(str(file_path), sr=None, mono=True,
|
||||
duration=max_analysis_seconds)
|
||||
tempo, frames = librosa.beat.beat_track(y=y, sr=sr)
|
||||
beats = librosa.frames_to_time(frames, sr=sr)
|
||||
except Exception:
|
||||
logger.warning("librosa beat analysis failed for %s", file_path)
|
||||
return None
|
||||
bpm = float(tempo[0] if hasattr(tempo, "__len__") else tempo)
|
||||
return {"bpm": bpm, "beats": [float(b) for b in beats]}
|
||||
|
||||
|
||||
def media_src_to_path(src: str) -> str:
|
||||
"""Convert an FCPXML media src (``file://`` URL or plain path) to a filesystem path."""
|
||||
if src.startswith("file://"):
|
||||
from urllib.parse import unquote, urlparse
|
||||
|
||||
return unquote(urlparse(src).path)
|
||||
return src
|
||||
|
||||
|
||||
def _parse_total_duration(stderr: str) -> Optional[float]:
|
||||
match = _DURATION_RE.search(stderr)
|
||||
if not match:
|
||||
return None
|
||||
hours, minutes, seconds = match.groups()
|
||||
return int(hours) * 3600 + int(minutes) * 60 + float(seconds)
|
||||
|
||||
|
||||
def detect_silence(
|
||||
path: str, noise_db: float = -30.0, min_duration: float = 0.5
|
||||
) -> Optional[List[Tuple[float, float]]]:
|
||||
"""Detect silence in an audio/video file's first audio stream.
|
||||
|
||||
Returns (start, end) ranges in source seconds, or ``None`` when the file
|
||||
is missing, ffmpeg is unavailable, or analysis fails. Raises ``ValueError``
|
||||
on out-of-bounds parameters (they end up in a subprocess argument, so they
|
||||
are validated, not trusted).
|
||||
"""
|
||||
if not (-120.0 <= noise_db <= 0.0):
|
||||
raise ValueError(f"noise_db must be between -120 and 0 dB, got {noise_db}")
|
||||
if not (0 < min_duration <= 3600):
|
||||
raise ValueError(f"min_duration must be between 0 and 3600 seconds, got {min_duration}")
|
||||
|
||||
file_path = Path(path)
|
||||
if not file_path.is_file():
|
||||
return None
|
||||
if shutil.which("ffmpeg") is None:
|
||||
logger.info("ffmpeg not found on PATH; silence detection unavailable")
|
||||
return None
|
||||
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[
|
||||
"ffmpeg", "-hide_banner", "-nostdin",
|
||||
"-i", str(file_path),
|
||||
# -vn: silence detection only needs the audio stream. Without it
|
||||
# ffmpeg decodes the full video track into the null muxer, which
|
||||
# blows past PROBE_TIMEOUT_SECONDS on long/high-bitrate files.
|
||||
"-vn",
|
||||
"-af", f"silencedetect=noise={float(noise_db)}dB:d={float(min_duration)}",
|
||||
"-f", "null", "-",
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=PROBE_TIMEOUT_SECONDS,
|
||||
)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
logger.warning("ffmpeg silence analysis failed for %s", file_path)
|
||||
return None
|
||||
if result.returncode != 0:
|
||||
return None
|
||||
return parse_silencedetect_output(
|
||||
result.stderr, total_duration=_parse_total_duration(result.stderr)
|
||||
)
|
||||
@@ -0,0 +1,361 @@
|
||||
"""Explicit transcription model management (Hex-inspired design).
|
||||
|
||||
Adapts the model-management concept from the Hex macOS app to the fcp-mcp-server
|
||||
stack (Python + MCP), keeping our conventions: rational, allowlisted model
|
||||
names, lazy optional imports, graceful degradation, and I/O confined to the
|
||||
model cache directory.
|
||||
|
||||
This module is the skeleton of the manager. The catalog and cache primitives
|
||||
are real; the MCP handlers and wiring into ``transcribe()`` come in a later
|
||||
phase (see docs/TRANSCRIPTION-MODELS.md).
|
||||
|
||||
Model weights are the same Systran/faster-whisper artifacts that
|
||||
``transcribe.py`` already loads, so install status here matches what
|
||||
``WhisperModel(model_size, ...)`` would download on its own.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import threading
|
||||
from pathlib import Path
|
||||
from typing import Callable, List, Optional
|
||||
|
||||
from .transcribe import ALLOWED_MODELS
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# faster-whisper (CTranslate2) caches HF snapshots under ~/.cache/huggingface/hub
|
||||
# as models--Systran--faster-whisper-<size>. This is the on-disk truth for
|
||||
# "is downloaded".
|
||||
_HF_REPO = "Systran/faster-whisper"
|
||||
|
||||
# Default cache root — matches what faster-whisper (HF hub) uses on its own,
|
||||
# so a default install lines up with anything already on disk.
|
||||
_DEFAULT_MODELS_DIR = Path.home() / ".cache" / "huggingface" / "hub"
|
||||
|
||||
# Config file for the persisted model selection + models dir.
|
||||
_CONFIG_DIR = Path.home() / ".fcp-mcp-server"
|
||||
_CONFIG_FILE = _CONFIG_DIR / "config.json"
|
||||
|
||||
# Path-traversal guard: only allow [A-Za-z0-9._-] in an internal model name
|
||||
# (already enforced by ALLOWED_MODELS, but the cache-dir helper is belt + braces).
|
||||
_SAFE_NAME_RE = re.compile(r"^[\w.-]+$")
|
||||
|
||||
# Progress callback signature used across the module.
|
||||
ProgressCallback = Callable[[float], None]
|
||||
|
||||
|
||||
def get_models_dir() -> Path:
|
||||
"""The configured models root (falls back to the default HF hub cache)."""
|
||||
configured = _load_config().get("models_dir")
|
||||
if configured:
|
||||
return Path(configured)
|
||||
return _DEFAULT_MODELS_DIR
|
||||
|
||||
|
||||
def save_models_dir(path: str) -> str:
|
||||
"""Persist the models root directory. Returns the stored value."""
|
||||
if not path.strip():
|
||||
raise ValueError("models_dir cannot be empty")
|
||||
_CONFIG_DIR.mkdir(parents=True, exist_ok=True)
|
||||
data = _load_config()
|
||||
data["models_dir"] = str(Path(path).expanduser())
|
||||
_write_config(data)
|
||||
return data["models_dir"]
|
||||
|
||||
|
||||
def _load_config() -> dict:
|
||||
"""Read the full config JSON (never raises; returns {} on error)."""
|
||||
try:
|
||||
return json.loads(_CONFIG_FILE.read_text(encoding="utf-8"))
|
||||
except (OSError, ValueError):
|
||||
return {}
|
||||
|
||||
|
||||
def _write_config(data: dict) -> None:
|
||||
_CONFIG_DIR.mkdir(parents=True, exist_ok=True)
|
||||
_CONFIG_FILE.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def _hf_snapshot_dir(model_size: str) -> Path:
|
||||
"""Cache folder for a given model size's HF snapshot, under the models root.
|
||||
|
||||
``Systran/faster-whisper`` -> ``<root>/models--Systran--faster-whisper-<size>``.
|
||||
"""
|
||||
name = "-".join(_HF_REPO.replace("/", "--").split("-")) + f"-{model_size}"
|
||||
return get_models_dir() / f"models--{name}"
|
||||
|
||||
|
||||
def model_cache_dir(model_size: str) -> Path:
|
||||
"""The on-disk cache directory for ``model_size`` (as downloaded on disk)."""
|
||||
if not _SAFE_NAME_RE.match(model_size):
|
||||
raise ValueError(f"Unsafe model name: {model_size!r}")
|
||||
return _hf_snapshot_dir(model_size)
|
||||
|
||||
|
||||
def _load_bundled_catalog() -> Optional[list[dict]]:
|
||||
"""Read the curated catalog from the bundled ``models.json`` (lazy, cached)."""
|
||||
global _catalog
|
||||
if _catalog is not None:
|
||||
return _catalog
|
||||
path = Path(__file__).with_name("models.json")
|
||||
try:
|
||||
with path.open(encoding="utf-8") as fh:
|
||||
_catalog = json.load(fh)
|
||||
except (OSError, ValueError):
|
||||
logger.warning("Failed to load bundled models.json (%s)", path)
|
||||
_catalog = []
|
||||
return _catalog
|
||||
|
||||
|
||||
_catalog: Optional[list[dict]] = None
|
||||
|
||||
|
||||
def load_catalog() -> list[dict]:
|
||||
"""All curated models (or ``[]`` if the bundled file is unreadable)."""
|
||||
return list(_load_bundled_catalog() or [])
|
||||
|
||||
|
||||
def get_catalog_model(internal_name: str) -> Optional[dict]:
|
||||
"""The catalog entry whose ``internal_name`` equals ``internal_name``."""
|
||||
for entry in load_catalog():
|
||||
if entry.get("internal_name") == internal_name:
|
||||
return entry
|
||||
return None
|
||||
|
||||
|
||||
def is_model_downloaded(model_size: str) -> bool:
|
||||
"""True when the model's cache snapshot exists and isn't an empty dir.
|
||||
|
||||
Never raises on I/O; reports ``False`` for missing/unreadable dirs so
|
||||
callers can offer a download instead of crashing.
|
||||
"""
|
||||
try:
|
||||
d = model_cache_dir(model_size)
|
||||
except ValueError:
|
||||
return False
|
||||
if not d.is_dir():
|
||||
return False
|
||||
try:
|
||||
return any(d.iterdir())
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
|
||||
def list_installed_models() -> List[str]:
|
||||
"""Model sizes present in the cache, filtered to the known allowlist."""
|
||||
root = get_models_dir()
|
||||
if not root.is_dir():
|
||||
return []
|
||||
installed: List[str] = []
|
||||
for entry in root.iterdir():
|
||||
if not entry.is_dir():
|
||||
continue
|
||||
for size in ALLOWED_MODELS:
|
||||
if entry.name == _hf_snapshot_dir(size).name and is_model_downloaded(size):
|
||||
installed.append(size)
|
||||
break
|
||||
# Stable, de-duped order following the allowlist.
|
||||
return [s for s in ALLOWED_MODELS if s in installed]
|
||||
|
||||
|
||||
def download_model(
|
||||
model_size: str,
|
||||
*,
|
||||
progress_cb: Optional[ProgressCallback] = None,
|
||||
cancel_event: Optional[threading.Event] = None,
|
||||
) -> Optional[Path]:
|
||||
"""Download a model snapshot to the cache.
|
||||
|
||||
Validates ``model_size`` against ``ALLOWED_MODELS`` and returns ``None``
|
||||
(with a logged install hint) when ``huggingface_hub`` is unavailable —
|
||||
the same graceful-degradation contract as ``transcribe()``.
|
||||
|
||||
``cancel_event`` (a ``threading.Event``) aborts the download on the next
|
||||
progress tick if set; the partially-downloaded snapshot is removed.
|
||||
"""
|
||||
if model_size not in ALLOWED_MODELS:
|
||||
raise ValueError(
|
||||
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
|
||||
)
|
||||
try:
|
||||
from huggingface_hub import snapshot_download
|
||||
except ImportError:
|
||||
logger.info(
|
||||
"huggingface_hub not installed; install the [transcribe] extra "
|
||||
"to enable model download"
|
||||
)
|
||||
return None
|
||||
|
||||
target = model_cache_dir(model_size)
|
||||
|
||||
# tqdm hook that both reports progress and honors cancellation.
|
||||
from tqdm import tqdm
|
||||
|
||||
class _CancelableTqdm(tqdm):
|
||||
def update(self, n=1):
|
||||
if cancel_event is not None and cancel_event.is_set():
|
||||
raise _DownloadCancelledError(model_size)
|
||||
super().update(n)
|
||||
|
||||
def _on_progress(current: int, total: int) -> None:
|
||||
if cancel_event is not None and cancel_event.is_set():
|
||||
raise _DownloadCancelledError(model_size)
|
||||
if progress_cb is not None and total > 0:
|
||||
progress_cb(current / total)
|
||||
|
||||
try:
|
||||
snapshot_download(
|
||||
repo_id=_HF_REPO,
|
||||
revision=model_size,
|
||||
# write into the same snapshot folder faster-whisper expects,
|
||||
# under the configured models root.
|
||||
cache_dir=str(get_models_dir()),
|
||||
local_dir=str(target),
|
||||
tqdm_class=_CancelableTqdm,
|
||||
local_dir_use_symlinks=False,
|
||||
)
|
||||
# Local snapshot already landed in `target`; simpler than a temp+move.
|
||||
except _DownloadCancelledError:
|
||||
logger.info("Download cancelled for model %s", model_size)
|
||||
_remove_dir(target)
|
||||
return None
|
||||
except Exception:
|
||||
logger.warning("Failed to download model %s", model_size)
|
||||
_remove_dir(target)
|
||||
return None
|
||||
return target
|
||||
|
||||
|
||||
class _DownloadCancelledError(Exception):
|
||||
"""Internal signal raised to abort a model download."""
|
||||
|
||||
|
||||
def _remove_dir(path: Path) -> None:
|
||||
import shutil
|
||||
|
||||
try:
|
||||
if path.is_dir():
|
||||
shutil.rmtree(path, ignore_errors=True)
|
||||
except OSError:
|
||||
logger.warning("Failed to clean up partial download %s", path)
|
||||
|
||||
|
||||
def delete_model(model_size: str) -> bool:
|
||||
"""Remove a model's cache snapshot from disk. Returns True if something was removed."""
|
||||
try:
|
||||
d = model_cache_dir(model_size)
|
||||
except ValueError:
|
||||
return False
|
||||
if not d.is_dir():
|
||||
return False
|
||||
try:
|
||||
import shutil
|
||||
|
||||
shutil.rmtree(d, ignore_errors=True)
|
||||
except OSError:
|
||||
logger.warning("Failed to delete model cache %s", d)
|
||||
return False
|
||||
return not d.exists()
|
||||
|
||||
|
||||
def _load_selection() -> str:
|
||||
"""The persisted selected model size (or ``""`` when absent/invalid)."""
|
||||
return str(_load_config().get("selected_model", ""))
|
||||
|
||||
|
||||
def save_selected_model(model_size: str) -> str:
|
||||
"""Persist the selected model size. Validates against ``ALLOWED_MODELS``.
|
||||
|
||||
Returns the persisted value so callers can confirm it round-tripped.
|
||||
"""
|
||||
if model_size not in ALLOWED_MODELS:
|
||||
raise ValueError(
|
||||
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
|
||||
)
|
||||
data = _load_config()
|
||||
data["selected_model"] = model_size
|
||||
_write_config(data)
|
||||
return model_size
|
||||
|
||||
|
||||
def load_selected_model() -> str:
|
||||
"""The effective selected model, with a safe fallback.
|
||||
|
||||
- Returns the persisted selection when it's in ``ALLOWED_MODELS``.
|
||||
- If the persisted model isn't on disk but another is installed, returns that
|
||||
installed one (never clears the user's persisted value on a false-negative
|
||||
availability scan, mirroring Hex's rule).
|
||||
- Otherwise returns ``""`` so callers can fall back to ``"base"``.
|
||||
"""
|
||||
selected = _load_selection()
|
||||
if selected and selected in ALLOWED_MODELS and is_model_downloaded(selected):
|
||||
return selected
|
||||
installed = list_installed_models()
|
||||
if installed:
|
||||
return installed[0]
|
||||
return ""
|
||||
|
||||
|
||||
def load_hf_token() -> str:
|
||||
"""The persisted HuggingFace token for diarization (or ``""``)."""
|
||||
return str(_load_config().get("hf_token", ""))
|
||||
|
||||
|
||||
def save_hf_token(token: str) -> str:
|
||||
"""Persist the HuggingFace token used for speaker diarization."""
|
||||
data = _load_config()
|
||||
data["hf_token"] = str(token or "").strip()
|
||||
_write_config(data)
|
||||
return data["hf_token"]
|
||||
|
||||
|
||||
# Transcription languages, matching the codes used across the app UI.
|
||||
# ``""`` / ``"auto"`` means "detect automatically".
|
||||
ALLOWED_LANGUAGES = {
|
||||
"auto",
|
||||
"pt",
|
||||
"en",
|
||||
"es",
|
||||
"fr",
|
||||
"de",
|
||||
"it",
|
||||
"nl",
|
||||
"ja",
|
||||
"ko",
|
||||
"zh",
|
||||
}
|
||||
|
||||
|
||||
def load_transcript_language() -> str:
|
||||
"""The persisted transcription language (``""``/``"auto"`` = auto-detect)."""
|
||||
lang = str(_load_config().get("language", ""))
|
||||
return lang if lang in ALLOWED_LANGUAGES else "auto"
|
||||
|
||||
|
||||
def save_transcript_language(lang: str) -> str:
|
||||
"""Persist the transcription language. Returns the stored value."""
|
||||
val = str(lang or "auto").strip().lower()
|
||||
if val not in ALLOWED_LANGUAGES:
|
||||
raise ValueError(
|
||||
f"language must be one of {', '.join(sorted(ALLOWED_LANGUAGES))}, got {lang!r}"
|
||||
)
|
||||
data = _load_config()
|
||||
data["language"] = val
|
||||
_write_config(data)
|
||||
return val
|
||||
|
||||
|
||||
def load_num_speakers() -> str:
|
||||
"""The persisted expected participant count (``""`` = auto-detect)."""
|
||||
return str(_load_config().get("num_speakers", ""))
|
||||
|
||||
|
||||
def save_num_speakers(num: str) -> str:
|
||||
"""Persist the expected participant count (empty string = auto)."""
|
||||
val = str(num or "").strip()
|
||||
data = _load_config()
|
||||
data["num_speakers"] = val
|
||||
_write_config(data)
|
||||
return val
|
||||
@@ -0,0 +1,66 @@
|
||||
[
|
||||
{
|
||||
"display_name": "Whisper Tiny",
|
||||
"internal_name": "tiny",
|
||||
"language": "Multilingual",
|
||||
"accuracy": 2,
|
||||
"speed": 5,
|
||||
"storage": "76 MB"
|
||||
},
|
||||
{
|
||||
"display_name": "Whisper Tiny EN",
|
||||
"internal_name": "tiny.en",
|
||||
"language": "English",
|
||||
"accuracy": 2,
|
||||
"speed": 5,
|
||||
"storage": "76 MB"
|
||||
},
|
||||
{
|
||||
"display_name": "Whisper Base",
|
||||
"internal_name": "base",
|
||||
"language": "Multilingual",
|
||||
"accuracy": 3,
|
||||
"speed": 4,
|
||||
"storage": "145 MB"
|
||||
},
|
||||
{
|
||||
"display_name": "Whisper Base EN",
|
||||
"internal_name": "base.en",
|
||||
"language": "English",
|
||||
"accuracy": 3,
|
||||
"speed": 4,
|
||||
"storage": "145 MB"
|
||||
},
|
||||
{
|
||||
"display_name": "Whisper Small",
|
||||
"internal_name": "small",
|
||||
"language": "Multilingual",
|
||||
"accuracy": 4,
|
||||
"speed": 3,
|
||||
"storage": "484 MB"
|
||||
},
|
||||
{
|
||||
"display_name": "Whisper Medium",
|
||||
"internal_name": "medium",
|
||||
"language": "Multilingual",
|
||||
"accuracy": 5,
|
||||
"speed": 2,
|
||||
"storage": "1.53 GB"
|
||||
},
|
||||
{
|
||||
"display_name": "Whisper Large v3",
|
||||
"internal_name": "large-v3",
|
||||
"language": "Multilingual",
|
||||
"accuracy": 5,
|
||||
"speed": 1,
|
||||
"storage": "3.08 GB"
|
||||
},
|
||||
{
|
||||
"display_name": "Whisper Distil v3",
|
||||
"internal_name": "distil-large-v3",
|
||||
"language": "English",
|
||||
"accuracy": 5,
|
||||
"speed": 4,
|
||||
"storage": "1.5 GB"
|
||||
}
|
||||
]
|
||||
Executable
+1073
File diff suppressed because it is too large
Load Diff
Executable
+367
@@ -0,0 +1,367 @@
|
||||
"""
|
||||
FCPXML Parser - Reads Final Cut Pro XML files into Python objects.
|
||||
"""
|
||||
|
||||
import xml.etree.ElementTree as ET
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from .models import (
|
||||
MARKER_XML_TAGS,
|
||||
Clip,
|
||||
ConnectedClip,
|
||||
Keyword,
|
||||
Marker,
|
||||
MarkerType,
|
||||
Project,
|
||||
Timecode,
|
||||
Timeline,
|
||||
TimeValue,
|
||||
Transition,
|
||||
)
|
||||
from .safe_xml import safe_fromstring, safe_parse
|
||||
|
||||
# Maximum FCPXML file size (50 MB) — prevents memory exhaustion from crafted files
|
||||
_MAX_FILE_SIZE_BYTES = 50 * 1024 * 1024
|
||||
|
||||
# Tags that represent connected clip elements (includes 'title' for text overlays)
|
||||
_CONNECTED_CLIP_TAGS = ('asset-clip', 'clip', 'video', 'audio', 'title', 'ref-clip')
|
||||
|
||||
|
||||
class FCPXMLParser:
|
||||
"""Parser for Final Cut Pro FCPXML files. Supports versions 1.8 - 1.14.
|
||||
|
||||
Unknown elements introduced by newer FCPXML versions (e.g. 1.13's
|
||||
``adjust-stereo-3D`` / ``hidden-clip-marker``, 1.14's smart-collection
|
||||
search rules) are ignored on read and preserved untouched by the
|
||||
modify path, which operates on the raw ElementTree.
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
self.resources: Dict[str, Dict[str, Any]] = {}
|
||||
self.formats: Dict[str, Dict[str, Any]] = {}
|
||||
self.frame_rate: float = 24.0
|
||||
|
||||
def _tc(self, elem: ET.Element, attr: str, default: str = '0s') -> Timecode:
|
||||
"""Parse a rational time attribute from an XML element.
|
||||
|
||||
Centralises the ``Timecode.from_rational(elem.get(attr), frame_rate)``
|
||||
pattern that repeats across every clip/marker/transition parser.
|
||||
"""
|
||||
return Timecode.from_rational(elem.get(attr, default), self.frame_rate)
|
||||
|
||||
def parse_file(self, filepath: str) -> Project:
|
||||
"""Parse an FCPXML file and return a Project object.
|
||||
|
||||
Enforces a file size limit to prevent memory exhaustion from
|
||||
maliciously large XML files.
|
||||
"""
|
||||
path = Path(filepath)
|
||||
if path.suffix == '.fcpxmld':
|
||||
fcpxml_path = path / 'Info.fcpxml'
|
||||
if not fcpxml_path.exists():
|
||||
raise FileNotFoundError(f"Info.fcpxml not found in bundle: {filepath}")
|
||||
filepath = str(fcpxml_path)
|
||||
path = Path(filepath)
|
||||
file_size = path.stat().st_size
|
||||
if file_size > _MAX_FILE_SIZE_BYTES:
|
||||
raise ValueError(
|
||||
f"FCPXML file exceeds maximum size "
|
||||
f"({file_size / 1024 / 1024:.1f} MB > "
|
||||
f"{_MAX_FILE_SIZE_BYTES / 1024 / 1024:.0f} MB limit)"
|
||||
)
|
||||
tree = safe_parse(filepath)
|
||||
return self._parse_fcpxml(tree.getroot())
|
||||
|
||||
def parse_string(self, xml_string: str) -> Project:
|
||||
"""Parse FCPXML from a string."""
|
||||
return self._parse_fcpxml(safe_fromstring(xml_string))
|
||||
|
||||
def _parse_fcpxml(self, root: ET.Element) -> Project:
|
||||
"""Parse the root fcpxml element."""
|
||||
version = root.get('version', '1.11')
|
||||
resources_elem = root.find('resources')
|
||||
if resources_elem is not None:
|
||||
self._parse_resources(resources_elem)
|
||||
|
||||
timelines = []
|
||||
for library in root.findall('.//library'):
|
||||
for event in library.findall('event'):
|
||||
for project in event.findall('project'):
|
||||
timeline = self._parse_project(project)
|
||||
if timeline:
|
||||
timelines.append(timeline)
|
||||
|
||||
if not timelines:
|
||||
for project in root.findall('.//project'):
|
||||
timeline = self._parse_project(project)
|
||||
if timeline:
|
||||
timelines.append(timeline)
|
||||
|
||||
project_name = timelines[0].name if timelines else "Untitled"
|
||||
return Project(name=project_name, timelines=timelines, fcpxml_version=version)
|
||||
|
||||
def _parse_resources(self, resources: ET.Element):
|
||||
"""Parse the resources section."""
|
||||
for fmt in resources.findall('format'):
|
||||
fmt_id = fmt.get('id', '')
|
||||
self.formats[fmt_id] = {
|
||||
'id': fmt_id, 'name': fmt.get('name', ''),
|
||||
'width': int(fmt.get('width', 1920)),
|
||||
'height': int(fmt.get('height', 1080)),
|
||||
'frameDuration': fmt.get('frameDuration', '1/24s')
|
||||
}
|
||||
frame_dur = fmt.get('frameDuration', '1/24s')
|
||||
if '/' in frame_dur:
|
||||
parts = frame_dur.rstrip('s').split('/', 1)
|
||||
num, denom = int(parts[0]), int(parts[1])
|
||||
if num <= 0:
|
||||
raise ValueError(f"Invalid frameDuration numerator: {frame_dur}")
|
||||
if denom <= 0:
|
||||
raise ValueError(f"Invalid frameDuration denominator: {frame_dur}")
|
||||
self.frame_rate = denom / num
|
||||
|
||||
for asset in resources.findall('asset'):
|
||||
asset_id = asset.get('id', '')
|
||||
self.resources[asset_id] = {
|
||||
'id': asset_id, 'name': asset.get('name', ''),
|
||||
'src': asset.get('src', '') or (media_rep.get('src', '') if (media_rep := asset.find('media-rep')) is not None else ''),
|
||||
'start': asset.get('start', '0s'),
|
||||
'duration': asset.get('duration', '0s'),
|
||||
'hasVideo': asset.get('hasVideo', '1') == '1',
|
||||
'hasAudio': asset.get('hasAudio', '1') == '1',
|
||||
}
|
||||
|
||||
def _parse_project(self, project: ET.Element) -> Optional[Timeline]:
|
||||
"""Parse a project element into a Timeline."""
|
||||
name = project.get('name', 'Untitled')
|
||||
sequence = project.find('sequence')
|
||||
if sequence is None:
|
||||
return None
|
||||
|
||||
format_ref = sequence.get('format', '')
|
||||
fmt = self.formats.get(format_ref, {})
|
||||
|
||||
timeline = Timeline(
|
||||
name=name,
|
||||
duration=self._tc(sequence, 'duration'),
|
||||
frame_rate=self.frame_rate,
|
||||
width=fmt.get('width', 1920),
|
||||
height=fmt.get('height', 1080)
|
||||
)
|
||||
|
||||
spine = sequence.find('spine')
|
||||
if spine is not None:
|
||||
self._parse_spine(spine, timeline)
|
||||
|
||||
timeline.markers.extend(self._collect_markers(sequence))
|
||||
|
||||
return timeline
|
||||
|
||||
def _parse_spine(self, spine: ET.Element, timeline: Timeline):
|
||||
"""Parse the spine (primary storyline) including connected clips."""
|
||||
current_offset = 0
|
||||
for elem in spine:
|
||||
tag = elem.tag
|
||||
if tag in ('asset-clip', 'clip', 'video', 'mc-clip', 'sync-clip', 'ref-clip'):
|
||||
clip = self._parse_clip(elem, current_offset)
|
||||
if clip:
|
||||
timeline.clips.append(clip)
|
||||
self._parse_connected_clips(elem, clip, timeline)
|
||||
current_offset += clip.duration.frames
|
||||
elif tag == 'gap':
|
||||
gap_frames = self._tc(elem, 'duration').frames
|
||||
self._parse_gap_connected_clips(elem, current_offset, timeline)
|
||||
current_offset += gap_frames
|
||||
elif tag == 'transition':
|
||||
transition = self._parse_transition(elem, current_offset)
|
||||
if transition:
|
||||
timeline.transitions.append(transition)
|
||||
|
||||
def _parse_clip(self, elem: ET.Element, offset: int) -> Optional[Clip]:
|
||||
"""Parse a clip element."""
|
||||
name = elem.get('name', 'Untitled Clip')
|
||||
duration = self._tc(elem, 'duration')
|
||||
source_start = self._tc(elem, 'start')
|
||||
ref = elem.get('ref', '')
|
||||
media_path = self.resources.get(ref, {}).get('src', '')
|
||||
|
||||
clip = Clip(
|
||||
name=name,
|
||||
start=Timecode(frames=offset, frame_rate=self.frame_rate),
|
||||
duration=duration,
|
||||
source_start=source_start,
|
||||
media_path=media_path,
|
||||
audio_role=elem.get('audioRole', ''),
|
||||
video_role=elem.get('videoRole', ''),
|
||||
)
|
||||
|
||||
clip.markers.extend(self._collect_markers(elem))
|
||||
|
||||
for keyword_elem in elem.findall('keyword'):
|
||||
keyword = self._parse_keyword(keyword_elem)
|
||||
if keyword:
|
||||
clip.keywords.append(keyword)
|
||||
|
||||
return clip
|
||||
|
||||
def _parse_marker_element(self, elem: ET.Element) -> Optional[Marker]:
|
||||
"""Parse any marker element (<marker> or <chapter-marker>).
|
||||
|
||||
Type detection is delegated to MarkerType.from_xml_element which
|
||||
owns the completed-attribute semantics. This means the parser
|
||||
doesn't need separate methods for each tag.
|
||||
"""
|
||||
return Marker(
|
||||
name=elem.get('value', ''),
|
||||
start=self._tc(elem, 'start'),
|
||||
duration=self._tc(elem, 'duration', '1/24s'),
|
||||
marker_type=MarkerType.from_xml_element(elem),
|
||||
note=elem.get('note', '')
|
||||
)
|
||||
|
||||
def _collect_markers(self, elem: ET.Element) -> list:
|
||||
"""Collect all markers (standard + chapter) from an element in a single pass.
|
||||
|
||||
Iterates children once, selecting recognised marker tags via
|
||||
MARKER_XML_TAGS rather than making a separate findall per tag.
|
||||
"""
|
||||
return [
|
||||
marker
|
||||
for child in elem
|
||||
if child.tag in MARKER_XML_TAGS
|
||||
for marker in [self._parse_marker_element(child)]
|
||||
if marker is not None
|
||||
]
|
||||
|
||||
def _parse_keyword(self, elem: ET.Element) -> Optional[Keyword]:
|
||||
"""Parse a keyword element."""
|
||||
return Keyword(
|
||||
value=elem.get('value', ''),
|
||||
start=self._tc(elem, 'start') if elem.get('start') else None,
|
||||
duration=self._tc(elem, 'duration') if elem.get('duration') else None,
|
||||
)
|
||||
|
||||
def _parse_transition(self, elem: ET.Element, offset: int) -> Optional[Transition]:
|
||||
"""Parse a transition element."""
|
||||
return Transition(
|
||||
name=elem.get('name', 'Cross Dissolve'),
|
||||
duration=self._tc(elem, 'duration', '1s'),
|
||||
start=Timecode(frames=offset, frame_rate=self.frame_rate)
|
||||
)
|
||||
|
||||
def get_library_clips(self, keywords: Optional[list] = None) -> list:
|
||||
"""
|
||||
Get all available clips from the library (assets in resources section).
|
||||
|
||||
Args:
|
||||
keywords: Optional list of keywords to filter by
|
||||
|
||||
Returns:
|
||||
List of dicts with asset metadata: name, asset_id, duration_seconds, src
|
||||
"""
|
||||
result = []
|
||||
for asset_id, asset_data in self.resources.items():
|
||||
# Parse duration to seconds
|
||||
duration_str = asset_data.get('duration', '0s')
|
||||
duration_seconds = self._parse_duration_to_seconds(duration_str)
|
||||
|
||||
clip_info = {
|
||||
'asset_id': asset_id,
|
||||
'name': asset_data.get('name', ''),
|
||||
'duration_seconds': duration_seconds,
|
||||
'src': asset_data.get('src', ''),
|
||||
'has_video': asset_data.get('hasVideo', True),
|
||||
'has_audio': asset_data.get('hasAudio', True),
|
||||
}
|
||||
result.append(clip_info)
|
||||
|
||||
# Filter by keywords if provided
|
||||
if keywords:
|
||||
# For now, assets don't have keywords directly - return empty if filtering
|
||||
# In real FCPXML, keywords are typically on clips in events, not assets
|
||||
return []
|
||||
|
||||
return result
|
||||
|
||||
def _iter_connected_elements(self, parent_elem: ET.Element, parent_name: str):
|
||||
"""Yield ``(element, lane, parent_name)`` tuples for connected clips.
|
||||
|
||||
Shared iteration logic for both spine-clip and gap-attached connected
|
||||
clips — walks direct children with a ``lane`` attribute and
|
||||
``<storyline>`` wrappers, yielding parsed :class:`ConnectedClip`
|
||||
objects without prescribing where they get stored.
|
||||
"""
|
||||
for child in parent_elem:
|
||||
lane = child.get('lane')
|
||||
if lane is not None and child.tag in _CONNECTED_CLIP_TAGS:
|
||||
connected = self._parse_one_connected_clip(
|
||||
child, int(lane), parent_name)
|
||||
if connected:
|
||||
yield connected
|
||||
elif child.tag == 'storyline':
|
||||
lane_val = int(child.get('lane', '1'))
|
||||
for sub_elem in child:
|
||||
if sub_elem.tag in _CONNECTED_CLIP_TAGS:
|
||||
connected = self._parse_one_connected_clip(
|
||||
sub_elem, lane_val, parent_name)
|
||||
if connected:
|
||||
yield connected
|
||||
|
||||
def _parse_connected_clips(self, parent_elem: ET.Element,
|
||||
parent_clip: Clip, timeline: Timeline):
|
||||
"""Parse connected clips attached to a primary storyline clip."""
|
||||
for connected in self._iter_connected_elements(parent_elem, parent_clip.name):
|
||||
parent_clip.connected_clips.append(connected)
|
||||
timeline.connected_clips.append(connected)
|
||||
|
||||
def _parse_gap_connected_clips(self, gap_elem: ET.Element,
|
||||
gap_offset: int, timeline: Timeline):
|
||||
"""Parse connected clips attached to gap elements."""
|
||||
for connected in self._iter_connected_elements(gap_elem, f"gap@{gap_offset}"):
|
||||
timeline.connected_clips.append(connected)
|
||||
|
||||
def _parse_one_connected_clip(self, elem: ET.Element, lane: int,
|
||||
parent_name: str) -> Optional[ConnectedClip]:
|
||||
"""Parse a single connected clip element."""
|
||||
name = elem.get('name', 'Untitled')
|
||||
duration = self._tc(elem, 'duration')
|
||||
start = self._tc(elem, 'start')
|
||||
offset = self._tc(elem, 'offset')
|
||||
ref = elem.get('ref', '')
|
||||
media_path = self.resources.get(ref, {}).get('src', '')
|
||||
role = elem.get('audioRole', '') or elem.get('videoRole', '')
|
||||
|
||||
connected = ConnectedClip(
|
||||
name=name, start=start, duration=duration,
|
||||
lane=lane, offset=offset, source_start=start,
|
||||
media_path=media_path, clip_type=elem.tag, role=role,
|
||||
ref_id=ref, parent_clip_name=parent_name,
|
||||
)
|
||||
|
||||
connected.markers.extend(self._collect_markers(elem))
|
||||
|
||||
for keyword_elem in elem.findall('keyword'):
|
||||
keyword = self._parse_keyword(keyword_elem)
|
||||
if keyword:
|
||||
connected.keywords.append(keyword)
|
||||
|
||||
return connected
|
||||
|
||||
def _parse_duration_to_seconds(self, duration_str: str) -> float:
|
||||
"""Convert FCPXML duration string to seconds.
|
||||
|
||||
Delegates to TimeValue.from_timecode() which handles rational
|
||||
format (``"150/30s"``), plain seconds (``"10s"``), timecode
|
||||
(``HH:MM:SS:FF``), and frame counts (``"15f"``).
|
||||
"""
|
||||
try:
|
||||
return TimeValue.from_timecode(duration_str).to_seconds()
|
||||
except (ValueError, ZeroDivisionError):
|
||||
# Zero-denominator or unparseable → 0.0 (matches prior behaviour)
|
||||
return 0.0
|
||||
|
||||
|
||||
def parse_fcpxml(filepath: str) -> Project:
|
||||
"""Convenience function to parse an FCPXML file."""
|
||||
return FCPXMLParser().parse_file(filepath)
|
||||
Executable
+798
@@ -0,0 +1,798 @@
|
||||
"""
|
||||
Auto Rough Cut - AI-powered timeline assembly from source clips.
|
||||
|
||||
The flagship feature: select clips by keywords, set a target duration
|
||||
and pacing style, and get a complete rough cut assembled automatically.
|
||||
"""
|
||||
|
||||
import random
|
||||
import uuid
|
||||
import xml.etree.ElementTree as ET
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from .models import (
|
||||
MontageConfig,
|
||||
PacingConfig,
|
||||
PacingCurve,
|
||||
RoughCutResult,
|
||||
SegmentSpec,
|
||||
TimeValue,
|
||||
)
|
||||
from .writer import _create_asset_element, write_fcpxml
|
||||
|
||||
|
||||
class RoughCutGenerator:
|
||||
"""
|
||||
Generates rough cuts from source FCPXML files.
|
||||
|
||||
Usage:
|
||||
generator = RoughCutGenerator("source.fcpxml")
|
||||
result = generator.generate(
|
||||
output_path="rough_cut.fcpxml",
|
||||
target_duration="00:03:00:00",
|
||||
pacing="medium",
|
||||
segments=[
|
||||
SegmentSpec(name="Intro", keywords=["intro"], duration_seconds=30),
|
||||
SegmentSpec(name="Interview", keywords=["interview", "talking"]),
|
||||
SegmentSpec(name="Outro", keywords=["outro"], duration_seconds=15),
|
||||
]
|
||||
)
|
||||
"""
|
||||
|
||||
def __init__(self, source_fcpxml: str):
|
||||
"""Load source FCPXML for clip selection."""
|
||||
self.source_path = Path(source_fcpxml)
|
||||
from .safe_xml import safe_parse
|
||||
self.tree = safe_parse(source_fcpxml)
|
||||
self.root = self.tree.getroot()
|
||||
self.fps = self._detect_fps()
|
||||
self._index_clips()
|
||||
self._index_resources()
|
||||
|
||||
def _detect_fps(self) -> float:
|
||||
"""Extract frame rate from format resource."""
|
||||
for fmt in self.root.findall('.//format'):
|
||||
frame_dur = fmt.get('frameDuration', '1/30s')
|
||||
if '/' in frame_dur:
|
||||
num, denom = frame_dur.replace('s', '').split('/')
|
||||
return int(denom) / int(num)
|
||||
return 30.0
|
||||
|
||||
def _index_clips(self) -> None:
|
||||
"""Index all clips with their metadata."""
|
||||
self.clips: List[Dict[str, Any]] = []
|
||||
|
||||
for clip in self.root.findall('.//asset-clip'):
|
||||
clip_data = self._extract_clip_data(clip, 'asset-clip')
|
||||
if clip_data:
|
||||
self.clips.append(clip_data)
|
||||
|
||||
for clip in self.root.findall('.//clip'):
|
||||
clip_data = self._extract_clip_data(clip, 'clip')
|
||||
if clip_data:
|
||||
self.clips.append(clip_data)
|
||||
|
||||
for video in self.root.findall('.//video'):
|
||||
clip_data = self._extract_clip_data(video, 'video')
|
||||
if clip_data:
|
||||
self.clips.append(clip_data)
|
||||
|
||||
def _extract_clip_data(self, elem: ET.Element, clip_type: str) -> Optional[Dict[str, Any]]:
|
||||
"""Extract clip metadata into a dictionary."""
|
||||
name = elem.get('name', 'Untitled')
|
||||
duration_str = elem.get('duration', '0s')
|
||||
start_str = elem.get('start', '0s')
|
||||
ref = elem.get('ref', '')
|
||||
|
||||
duration = TimeValue.from_timecode(duration_str, self.fps)
|
||||
|
||||
# Skip very short clips
|
||||
if duration.to_seconds() < 0.1:
|
||||
return None
|
||||
|
||||
# Extract keywords
|
||||
keywords = []
|
||||
for kw in elem.findall('keyword'):
|
||||
keywords.append(kw.get('value', ''))
|
||||
|
||||
# Check if favorited
|
||||
is_favorite = elem.get('rating', '') == '1' or elem.get('isFavorite', '') == '1'
|
||||
is_rejected = elem.get('rating', '') == '-1' or elem.get('isRejected', '') == '1'
|
||||
|
||||
return {
|
||||
'element': elem,
|
||||
'name': name,
|
||||
'type': clip_type,
|
||||
'ref': ref,
|
||||
'duration': duration,
|
||||
'start': TimeValue.from_timecode(start_str, self.fps),
|
||||
'keywords': keywords,
|
||||
'is_favorite': is_favorite,
|
||||
'is_rejected': is_rejected,
|
||||
'used': False, # Track if already used in rough cut
|
||||
}
|
||||
|
||||
def _index_resources(self) -> None:
|
||||
"""Index all resources (assets, formats)."""
|
||||
self.resources = {}
|
||||
self.formats = {}
|
||||
|
||||
for asset in self.root.findall('.//asset'):
|
||||
self.resources[asset.get('id', '')] = asset
|
||||
|
||||
for fmt in self.root.findall('.//format'):
|
||||
self.formats[fmt.get('id', '')] = fmt
|
||||
|
||||
def generate(
|
||||
self,
|
||||
output_path: str,
|
||||
target_duration: str = "00:03:00:00",
|
||||
pacing: str = "medium",
|
||||
segments: Optional[List[SegmentSpec]] = None,
|
||||
keywords: Optional[List[str]] = None,
|
||||
priority: str = "best",
|
||||
exclude_rejected: bool = True,
|
||||
favorites_only: bool = False,
|
||||
add_transitions: bool = False,
|
||||
transition_duration: str = "00:00:00:15"
|
||||
) -> RoughCutResult:
|
||||
"""
|
||||
Generate a rough cut.
|
||||
|
||||
Args:
|
||||
output_path: Where to save the rough cut FCPXML
|
||||
target_duration: Target timeline length (timecode or "3m30s")
|
||||
pacing: "slow", "medium", "fast", or "dynamic"
|
||||
segments: Optional segment breakdown with keywords per section
|
||||
keywords: Simple mode - just filter by these keywords
|
||||
priority: How to select clips - "best", "favorites", "longest", "shortest", "random"
|
||||
exclude_rejected: Skip rejected clips
|
||||
favorites_only: Only use favorited clips
|
||||
add_transitions: Add cross-dissolves between clips
|
||||
transition_duration: Length of transitions
|
||||
|
||||
Returns:
|
||||
RoughCutResult with output details
|
||||
"""
|
||||
target_time = self._parse_duration(target_duration)
|
||||
pacing_config = PacingConfig(pacing=pacing)
|
||||
|
||||
# Filter available clips
|
||||
available_clips = self._filter_clips(
|
||||
keywords=keywords,
|
||||
exclude_rejected=exclude_rejected,
|
||||
favorites_only=favorites_only
|
||||
)
|
||||
|
||||
if not available_clips:
|
||||
raise ValueError("No clips match the filter criteria")
|
||||
|
||||
# Build the sequence
|
||||
if segments:
|
||||
selected_clips = self._select_clips_by_segments(
|
||||
available_clips, segments, target_time, pacing_config
|
||||
)
|
||||
else:
|
||||
selected_clips = self._select_clips_simple(
|
||||
available_clips, target_time, pacing_config, priority
|
||||
)
|
||||
|
||||
# Generate output FCPXML
|
||||
actual_duration = self._build_output(
|
||||
selected_clips, output_path, add_transitions, transition_duration
|
||||
)
|
||||
|
||||
return RoughCutResult(
|
||||
output_path=output_path,
|
||||
clips_used=len(selected_clips),
|
||||
clips_available=len(available_clips),
|
||||
target_duration=target_time.to_seconds(),
|
||||
actual_duration=actual_duration,
|
||||
segments=len(segments) if segments else 1,
|
||||
average_clip_duration=actual_duration / len(selected_clips) if selected_clips else 0
|
||||
)
|
||||
|
||||
def _parse_duration(self, duration: str) -> TimeValue:
|
||||
"""Parse various duration formats."""
|
||||
# Handle shorthand like "3m30s" or "2m"
|
||||
if 'm' in duration and ':' not in duration:
|
||||
parts = duration.lower().replace('s', '').split('m')
|
||||
minutes = int(parts[0]) if parts[0] else 0
|
||||
seconds = float(parts[1]) if len(parts) > 1 and parts[1] else 0
|
||||
total_seconds = minutes * 60 + seconds
|
||||
return TimeValue.from_seconds(total_seconds, self.fps)
|
||||
|
||||
return TimeValue.from_timecode(duration, self.fps)
|
||||
|
||||
def _filter_clips(
|
||||
self,
|
||||
keywords: Optional[List[str]] = None,
|
||||
exclude_rejected: bool = True,
|
||||
favorites_only: bool = False
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Filter clips by criteria."""
|
||||
result = []
|
||||
|
||||
for clip in self.clips:
|
||||
# Skip rejected
|
||||
if exclude_rejected and clip['is_rejected']:
|
||||
continue
|
||||
|
||||
# Filter to favorites only
|
||||
if favorites_only and not clip['is_favorite']:
|
||||
continue
|
||||
|
||||
# Filter by keywords
|
||||
if keywords:
|
||||
clip_kws = set(k.lower() for k in clip['keywords'])
|
||||
filter_kws = set(k.lower() for k in keywords)
|
||||
if not clip_kws & filter_kws:
|
||||
continue
|
||||
|
||||
result.append(clip)
|
||||
|
||||
return result
|
||||
|
||||
def _select_clips_simple(
|
||||
self,
|
||||
clips: List[Dict[str, Any]],
|
||||
target_duration: TimeValue,
|
||||
pacing: PacingConfig,
|
||||
priority: str
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Select clips to fill target duration."""
|
||||
selected = []
|
||||
current_duration = TimeValue.zero()
|
||||
target_seconds = target_duration.to_seconds()
|
||||
min_clip, max_clip = pacing.get_duration_range()
|
||||
|
||||
# Sort clips by priority
|
||||
if priority == 'favorites':
|
||||
clips = sorted(clips, key=lambda c: (not c['is_favorite'], -c['duration'].to_seconds()))
|
||||
elif priority == 'longest':
|
||||
clips = sorted(clips, key=lambda c: -c['duration'].to_seconds())
|
||||
elif priority == 'shortest':
|
||||
clips = sorted(clips, key=lambda c: c['duration'].to_seconds())
|
||||
elif priority == 'random':
|
||||
random.shuffle(clips)
|
||||
else: # 'best' - mix of favorites first, then by duration
|
||||
clips = sorted(clips, key=lambda c: (not c['is_favorite'], -c['duration'].to_seconds()))
|
||||
|
||||
for clip in clips:
|
||||
if current_duration.to_seconds() >= target_seconds:
|
||||
break
|
||||
|
||||
remaining = target_seconds - current_duration.to_seconds()
|
||||
clip_dur = clip['duration'].to_seconds()
|
||||
|
||||
# Calculate usable duration for this clip
|
||||
if clip_dur > max_clip:
|
||||
use_duration = min(max_clip, remaining)
|
||||
elif clip_dur < min_clip:
|
||||
use_duration = clip_dur # Use short clips as-is
|
||||
else:
|
||||
use_duration = min(clip_dur, remaining)
|
||||
|
||||
if use_duration <= 0:
|
||||
continue
|
||||
|
||||
# Create selection with in/out points
|
||||
selection = {
|
||||
**clip,
|
||||
'use_duration': TimeValue.from_seconds(use_duration, self.fps),
|
||||
'in_point': clip['start'],
|
||||
'out_point': clip['start'] + TimeValue.from_seconds(use_duration, self.fps),
|
||||
}
|
||||
selected.append(selection)
|
||||
current_duration = current_duration + TimeValue.from_seconds(use_duration, self.fps)
|
||||
|
||||
return selected
|
||||
|
||||
def _select_clips_by_segments(
|
||||
self,
|
||||
clips: List[Dict[str, Any]],
|
||||
segments: List[SegmentSpec],
|
||||
target_duration: TimeValue,
|
||||
pacing: PacingConfig
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Select clips organized by segment structure."""
|
||||
selected = []
|
||||
|
||||
# Calculate duration per segment
|
||||
total_specified = sum(s.duration_seconds for s in segments if s.duration_seconds > 0)
|
||||
unspecified_count = sum(1 for s in segments if s.duration_seconds <= 0)
|
||||
|
||||
if unspecified_count > 0:
|
||||
remaining_duration = target_duration.to_seconds() - total_specified
|
||||
per_segment = max(0, remaining_duration / unspecified_count)
|
||||
else:
|
||||
per_segment = 0
|
||||
|
||||
for segment in segments:
|
||||
seg_duration = segment.duration_seconds if segment.duration_seconds > 0 else per_segment
|
||||
|
||||
# Filter clips for this segment
|
||||
if segment.keywords:
|
||||
seg_clips = [c for c in clips if
|
||||
set(k.lower() for k in c['keywords']) &
|
||||
set(k.lower() for k in segment.keywords)]
|
||||
else:
|
||||
seg_clips = clips.copy()
|
||||
|
||||
# Remove already used clips
|
||||
seg_clips = [c for c in seg_clips if not c.get('used_in_rough', False)]
|
||||
|
||||
# Select clips for segment
|
||||
seg_target = TimeValue.from_seconds(seg_duration, self.fps)
|
||||
seg_selected = self._select_clips_simple(
|
||||
seg_clips, seg_target, pacing, segment.priority
|
||||
)
|
||||
|
||||
# Mark as used on originals so they're excluded from later segments
|
||||
used_names = {sel['name'] for sel in seg_selected}
|
||||
for c in clips:
|
||||
if c['name'] in used_names:
|
||||
c['used_in_rough'] = True
|
||||
for sel in seg_selected:
|
||||
sel['segment'] = segment.name
|
||||
|
||||
selected.extend(seg_selected)
|
||||
|
||||
return selected
|
||||
|
||||
def _build_output(
|
||||
self,
|
||||
clips: List[Dict[str, Any]],
|
||||
output_path: str,
|
||||
add_transitions: bool,
|
||||
transition_duration: str
|
||||
) -> float:
|
||||
"""Build the output FCPXML."""
|
||||
# Create new FCPXML structure
|
||||
root = ET.Element('fcpxml', version='1.13')
|
||||
resources = ET.SubElement(root, 'resources')
|
||||
|
||||
# Copy relevant resources
|
||||
format_id = None
|
||||
for fmt_id, fmt in self.formats.items():
|
||||
resources.append(fmt)
|
||||
format_id = fmt_id
|
||||
break
|
||||
|
||||
# Copy needed assets — use media-rep form for new assets
|
||||
used_refs = set(c.get('ref') for c in clips if c.get('ref'))
|
||||
for ref in used_refs:
|
||||
if ref in self.resources:
|
||||
orig = self.resources[ref]
|
||||
# Extract src from either attribute or media-rep child
|
||||
src = orig.get('src', '')
|
||||
if not src:
|
||||
mr = orig.find('media-rep')
|
||||
if mr is not None:
|
||||
src = mr.get('src', '')
|
||||
_create_asset_element(
|
||||
resources,
|
||||
asset_id=orig.get('id', ref),
|
||||
name=orig.get('name', ''),
|
||||
src=src,
|
||||
duration=orig.get('duration', '0s'),
|
||||
start=orig.get('start', '0s'),
|
||||
has_video=orig.get('hasVideo', '1'),
|
||||
has_audio=orig.get('hasAudio', '1'),
|
||||
uid=orig.get('uid'),
|
||||
)
|
||||
|
||||
# Create library structure
|
||||
library = ET.SubElement(root, 'library',
|
||||
location="file:///Users/editor/Movies/RoughCut.fcpbundle/")
|
||||
event = ET.SubElement(library, 'event',
|
||||
name="Rough Cut", uid=str(uuid.uuid4()).upper())
|
||||
project = ET.SubElement(event, 'project',
|
||||
name="Rough Cut", uid=str(uuid.uuid4()).upper(),
|
||||
modDate=datetime.now().strftime("%Y-%m-%d %H:%M:%S -0500"))
|
||||
|
||||
# Calculate total duration
|
||||
total_duration = TimeValue.zero()
|
||||
for clip in clips:
|
||||
total_duration = total_duration + clip['use_duration']
|
||||
|
||||
# Create sequence
|
||||
sequence = ET.SubElement(project, 'sequence',
|
||||
format=format_id or "r1",
|
||||
duration=total_duration.to_fcpxml(),
|
||||
tcStart="0s", tcFormat="NDF",
|
||||
audioLayout="stereo", audioRate="48k")
|
||||
|
||||
spine = ET.SubElement(sequence, 'spine')
|
||||
|
||||
# Add clips to spine
|
||||
current_offset = TimeValue.zero()
|
||||
trans_dur = TimeValue.from_timecode(transition_duration, self.fps) if add_transitions else None
|
||||
|
||||
for i, clip in enumerate(clips):
|
||||
ET.SubElement(spine, 'asset-clip',
|
||||
ref=clip.get('ref', 'r1'),
|
||||
offset=current_offset.to_fcpxml(),
|
||||
name=clip['name'],
|
||||
start=clip['in_point'].to_fcpxml(),
|
||||
duration=clip['use_duration'].to_fcpxml(),
|
||||
format=format_id or "r1",
|
||||
tcFormat="NDF")
|
||||
|
||||
# Add transition before clip (except first)
|
||||
if add_transitions and i > 0 and trans_dur:
|
||||
half_trans = trans_dur * 0.5
|
||||
trans_offset = current_offset - half_trans
|
||||
|
||||
transition = ET.Element('transition',
|
||||
name="Cross Dissolve",
|
||||
offset=trans_offset.to_fcpxml(),
|
||||
duration=trans_dur.to_fcpxml())
|
||||
ET.SubElement(transition, 'filter-video',
|
||||
name="Cross Dissolve")
|
||||
spine.insert(-1, transition)
|
||||
|
||||
current_offset = current_offset + clip['use_duration']
|
||||
|
||||
# Write output
|
||||
write_fcpxml(root, output_path)
|
||||
|
||||
return total_duration.to_seconds()
|
||||
|
||||
# ========================================================================
|
||||
# MONTAGE GENERATION (v0.3.0)
|
||||
# ========================================================================
|
||||
|
||||
def generate_montage(
|
||||
self,
|
||||
output_path: str,
|
||||
target_duration: str,
|
||||
pacing_curve: str = "accelerating",
|
||||
start_duration: float = 2.0,
|
||||
end_duration: float = 0.5,
|
||||
keywords: Optional[List[str]] = None,
|
||||
exclude_rejected: bool = True,
|
||||
add_transitions: bool = False
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Generate a rapid-fire montage with dynamic pacing curves.
|
||||
|
||||
Creates montages where clip duration varies over time:
|
||||
- accelerating: Starts slow (2s), ends fast (0.5s) - builds energy
|
||||
- decelerating: Starts fast, ends slow - winds down
|
||||
- pyramid: Slow → fast → slow - dramatic arc
|
||||
- constant: Same duration throughout
|
||||
|
||||
Args:
|
||||
output_path: Where to save the montage FCPXML
|
||||
target_duration: Total montage length (e.g., "30s", "00:00:30:00")
|
||||
pacing_curve: "accelerating", "decelerating", "pyramid", or "constant"
|
||||
start_duration: Clip duration at start (seconds)
|
||||
end_duration: Clip duration at end (seconds)
|
||||
keywords: Filter clips by keywords
|
||||
exclude_rejected: Skip rejected clips
|
||||
add_transitions: Add quick dissolves between clips
|
||||
|
||||
Returns:
|
||||
Dict with output_path, clips_used, actual_duration, pacing_curve
|
||||
"""
|
||||
# Parse pacing curve
|
||||
curve_map = {
|
||||
'accelerating': PacingCurve.ACCELERATING,
|
||||
'decelerating': PacingCurve.DECELERATING,
|
||||
'pyramid': PacingCurve.PYRAMID,
|
||||
'constant': PacingCurve.CONSTANT
|
||||
}
|
||||
curve = curve_map.get(pacing_curve, PacingCurve.ACCELERATING)
|
||||
|
||||
config = MontageConfig(
|
||||
target_duration=self._parse_duration(target_duration).to_seconds(),
|
||||
pacing_curve=curve,
|
||||
start_duration=start_duration,
|
||||
end_duration=end_duration,
|
||||
min_duration=0.2,
|
||||
max_duration=max(start_duration, end_duration) + 1.0
|
||||
)
|
||||
|
||||
# Filter available clips
|
||||
available_clips = self._filter_clips(
|
||||
keywords=keywords,
|
||||
exclude_rejected=exclude_rejected
|
||||
)
|
||||
|
||||
if not available_clips:
|
||||
raise ValueError("No clips match the filter criteria")
|
||||
|
||||
# Select clips with pacing curve
|
||||
selected_clips = self._select_clips_for_montage(available_clips, config)
|
||||
|
||||
# Build output
|
||||
actual_duration = self._build_output(
|
||||
selected_clips, output_path, add_transitions, "00:00:00:06"
|
||||
)
|
||||
|
||||
return {
|
||||
'output_path': output_path,
|
||||
'clips_used': len(selected_clips),
|
||||
'clips_available': len(available_clips),
|
||||
'target_duration': config.target_duration,
|
||||
'actual_duration': actual_duration,
|
||||
'pacing_curve': pacing_curve,
|
||||
'start_clip_duration': selected_clips[0]['use_duration'].to_seconds() if selected_clips else 0,
|
||||
'end_clip_duration': selected_clips[-1]['use_duration'].to_seconds() if selected_clips else 0
|
||||
}
|
||||
|
||||
def _select_clips_for_montage(
|
||||
self,
|
||||
clips: List[Dict[str, Any]],
|
||||
config: MontageConfig
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Select clips with dynamic pacing based on montage config."""
|
||||
selected = []
|
||||
current_duration = 0.0
|
||||
target = config.target_duration
|
||||
clip_index = 0
|
||||
|
||||
# Shuffle clips for variety
|
||||
available = clips.copy()
|
||||
random.shuffle(available)
|
||||
|
||||
while current_duration < target and clip_index < len(available):
|
||||
# Calculate position in montage (0.0 to 1.0)
|
||||
position = current_duration / target if target > 0 else 0
|
||||
|
||||
# Get target clip duration for this position
|
||||
target_clip_duration = config.get_duration_at_position(position)
|
||||
|
||||
# Find best matching clip
|
||||
clip = available[clip_index]
|
||||
clip_dur = clip['duration'].to_seconds()
|
||||
|
||||
# Determine actual duration to use
|
||||
remaining = target - current_duration
|
||||
use_duration = min(target_clip_duration, clip_dur, remaining)
|
||||
use_duration = max(use_duration, config.min_duration)
|
||||
|
||||
if use_duration <= 0:
|
||||
clip_index += 1
|
||||
continue
|
||||
|
||||
# Create selection
|
||||
selection = {
|
||||
**clip,
|
||||
'use_duration': TimeValue.from_seconds(use_duration, self.fps),
|
||||
'in_point': clip['start'],
|
||||
'out_point': clip['start'] + TimeValue.from_seconds(use_duration, self.fps),
|
||||
'montage_position': position
|
||||
}
|
||||
selected.append(selection)
|
||||
current_duration += use_duration
|
||||
clip_index += 1
|
||||
|
||||
return selected
|
||||
|
||||
# ========================================================================
|
||||
# A/B ROLL GENERATION (v0.3.0)
|
||||
# ========================================================================
|
||||
|
||||
def generate_ab_roll(
|
||||
self,
|
||||
output_path: str,
|
||||
target_duration: str,
|
||||
a_keywords: List[str],
|
||||
b_keywords: List[str],
|
||||
a_duration: str = "5s",
|
||||
b_duration: str = "3s",
|
||||
start_with: str = "a",
|
||||
exclude_rejected: bool = True,
|
||||
add_transitions: bool = True
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Generate classic documentary-style A/B roll edit.
|
||||
|
||||
Alternates between A-roll (main content, usually interviews) and
|
||||
B-roll (cutaway footage, usually visuals).
|
||||
|
||||
Args:
|
||||
output_path: Where to save the FCPXML
|
||||
target_duration: Total duration (e.g., "3m", "00:03:00:00")
|
||||
a_keywords: Keywords for A-roll clips (e.g., ["interview", "talking"])
|
||||
b_keywords: Keywords for B-roll clips (e.g., ["broll", "cutaway"])
|
||||
a_duration: How long each A-roll segment should be
|
||||
b_duration: How long each B-roll cutaway should be
|
||||
start_with: Start with "a" or "b" roll
|
||||
exclude_rejected: Skip rejected clips
|
||||
add_transitions: Add cross-dissolves between clips
|
||||
|
||||
Returns:
|
||||
Dict with output details including a_segments and b_segments counts
|
||||
"""
|
||||
target_time = self._parse_duration(target_duration)
|
||||
a_dur = self._parse_duration(a_duration)
|
||||
b_dur = self._parse_duration(b_duration)
|
||||
|
||||
# Filter A and B clips
|
||||
a_clips = self._filter_clips(keywords=a_keywords, exclude_rejected=exclude_rejected)
|
||||
b_clips = self._filter_clips(keywords=b_keywords, exclude_rejected=exclude_rejected)
|
||||
|
||||
if not a_clips:
|
||||
raise ValueError(f"No A-roll clips found with keywords: {a_keywords}")
|
||||
if not b_clips:
|
||||
raise ValueError(f"No B-roll clips found with keywords: {b_keywords}")
|
||||
|
||||
# Build alternating sequence
|
||||
selected_clips = self._build_ab_sequence(
|
||||
a_clips, b_clips, target_time, a_dur, b_dur, start_with
|
||||
)
|
||||
|
||||
# Build output
|
||||
actual_duration = self._build_output(
|
||||
selected_clips, output_path, add_transitions, "00:00:00:12"
|
||||
)
|
||||
|
||||
a_count = sum(1 for c in selected_clips if c.get('roll_type') == 'A')
|
||||
b_count = sum(1 for c in selected_clips if c.get('roll_type') == 'B')
|
||||
|
||||
return {
|
||||
'output_path': output_path,
|
||||
'clips_used': len(selected_clips),
|
||||
'a_clips_available': len(a_clips),
|
||||
'b_clips_available': len(b_clips),
|
||||
'a_segments': a_count,
|
||||
'b_segments': b_count,
|
||||
'target_duration': target_time.to_seconds(),
|
||||
'actual_duration': actual_duration,
|
||||
'a_duration_setting': a_duration,
|
||||
'b_duration_setting': b_duration
|
||||
}
|
||||
|
||||
def _build_ab_sequence(
|
||||
self,
|
||||
a_clips: List[Dict[str, Any]],
|
||||
b_clips: List[Dict[str, Any]],
|
||||
target_duration: TimeValue,
|
||||
a_dur: TimeValue,
|
||||
b_dur: TimeValue,
|
||||
start_with: str
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Build alternating A/B sequence."""
|
||||
selected = []
|
||||
current_duration = TimeValue.zero()
|
||||
target_seconds = target_duration.to_seconds()
|
||||
|
||||
# Track which clips we've used
|
||||
a_index = 0
|
||||
b_index = 0
|
||||
current_roll = start_with.lower()
|
||||
|
||||
while current_duration.to_seconds() < target_seconds:
|
||||
remaining = target_seconds - current_duration.to_seconds()
|
||||
|
||||
if current_roll == 'a':
|
||||
if a_index >= len(a_clips):
|
||||
a_index = 0 # Loop if needed
|
||||
|
||||
clip = a_clips[a_index]
|
||||
target_dur = min(a_dur.to_seconds(), remaining)
|
||||
clip_dur = clip['duration'].to_seconds()
|
||||
use_dur = min(target_dur, clip_dur)
|
||||
|
||||
selection = {
|
||||
**clip,
|
||||
'use_duration': TimeValue.from_seconds(use_dur, self.fps),
|
||||
'in_point': clip['start'],
|
||||
'out_point': clip['start'] + TimeValue.from_seconds(use_dur, self.fps),
|
||||
'roll_type': 'A'
|
||||
}
|
||||
selected.append(selection)
|
||||
current_duration = current_duration + TimeValue.from_seconds(use_dur, self.fps)
|
||||
a_index += 1
|
||||
current_roll = 'b'
|
||||
|
||||
else: # B-roll
|
||||
if b_index >= len(b_clips):
|
||||
b_index = 0 # Loop if needed
|
||||
|
||||
clip = b_clips[b_index]
|
||||
target_dur = min(b_dur.to_seconds(), remaining)
|
||||
clip_dur = clip['duration'].to_seconds()
|
||||
use_dur = min(target_dur, clip_dur)
|
||||
|
||||
selection = {
|
||||
**clip,
|
||||
'use_duration': TimeValue.from_seconds(use_dur, self.fps),
|
||||
'in_point': clip['start'],
|
||||
'out_point': clip['start'] + TimeValue.from_seconds(use_dur, self.fps),
|
||||
'roll_type': 'B'
|
||||
}
|
||||
selected.append(selection)
|
||||
current_duration = current_duration + TimeValue.from_seconds(use_dur, self.fps)
|
||||
b_index += 1
|
||||
current_roll = 'a'
|
||||
|
||||
# Safety check to prevent infinite loop
|
||||
if len(selected) > 1000:
|
||||
break
|
||||
|
||||
return selected
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# CONVENIENCE FUNCTIONS
|
||||
# ============================================================================
|
||||
|
||||
def generate_rough_cut(
|
||||
source_fcpxml: str,
|
||||
output_path: str,
|
||||
target_duration: str = "3m",
|
||||
pacing: str = "medium",
|
||||
keywords: Optional[List[str]] = None,
|
||||
priority: str = "best"
|
||||
) -> RoughCutResult:
|
||||
"""
|
||||
One-liner rough cut generation.
|
||||
|
||||
Example:
|
||||
result = generate_rough_cut(
|
||||
"source_clips.fcpxml",
|
||||
"rough_cut.fcpxml",
|
||||
target_duration="3m",
|
||||
pacing="fast",
|
||||
keywords=["broll", "action"]
|
||||
)
|
||||
"""
|
||||
generator = RoughCutGenerator(source_fcpxml)
|
||||
return generator.generate(
|
||||
output_path=output_path,
|
||||
target_duration=target_duration,
|
||||
pacing=pacing,
|
||||
keywords=keywords,
|
||||
priority=priority
|
||||
)
|
||||
|
||||
|
||||
def generate_segmented_rough_cut(
|
||||
source_fcpxml: str,
|
||||
output_path: str,
|
||||
segments: List[Dict[str, Any]],
|
||||
pacing: str = "medium",
|
||||
add_transitions: bool = True
|
||||
) -> RoughCutResult:
|
||||
"""
|
||||
Generate rough cut with defined segments.
|
||||
|
||||
Example:
|
||||
result = generate_segmented_rough_cut(
|
||||
"interview.fcpxml",
|
||||
"rough_cut.fcpxml",
|
||||
segments=[
|
||||
{"name": "Intro", "keywords": ["intro"], "duration": 30},
|
||||
{"name": "Main", "keywords": ["interview"], "duration": 180},
|
||||
{"name": "Outro", "keywords": ["outro"], "duration": 20},
|
||||
],
|
||||
pacing="medium",
|
||||
add_transitions=True
|
||||
)
|
||||
"""
|
||||
segment_specs = []
|
||||
for seg in segments:
|
||||
segment_specs.append(SegmentSpec(
|
||||
name=seg.get('name', 'Segment'),
|
||||
keywords=seg.get('keywords', []),
|
||||
duration_seconds=seg.get('duration', 0),
|
||||
priority=seg.get('priority', 'best')
|
||||
))
|
||||
|
||||
# Calculate total duration
|
||||
total = sum(seg.get('duration', 0) for seg in segments)
|
||||
|
||||
generator = RoughCutGenerator(source_fcpxml)
|
||||
return generator.generate(
|
||||
output_path=output_path,
|
||||
target_duration=f"{total}s",
|
||||
pacing=pacing,
|
||||
segments=segment_specs,
|
||||
add_transitions=add_transitions
|
||||
)
|
||||
Executable
+113
@@ -0,0 +1,113 @@
|
||||
"""
|
||||
Safe XML parsing — defused against XXE, billion laughs, and entity expansion.
|
||||
|
||||
Centralizes all XML parsing so every entry point (parser, writer, export,
|
||||
rough_cut) uses the same hardened functions. Drop-in replacements for
|
||||
ET.parse() and ET.fromstring().
|
||||
|
||||
Blocks:
|
||||
- External entity injection (XXE): file:///etc/passwd, http:// callbacks
|
||||
- Billion laughs / entity expansion: exponential DTD bombs
|
||||
- DTD retrieval: remote DTD loading
|
||||
|
||||
Security note: All flags are set explicitly rather than relying on library
|
||||
defaults — this ensures protection survives dependency upgrades that might
|
||||
change default behavior.
|
||||
|
||||
IMPORTANT: Never use stdlib xml.etree.ElementTree.parse() or .fromstring()
|
||||
directly in this codebase. Always import from this module instead.
|
||||
"""
|
||||
|
||||
import xml.etree.ElementTree as ET
|
||||
from xml.dom.minidom import Document
|
||||
|
||||
import defusedxml.ElementTree as _safe_ET
|
||||
import defusedxml.minidom as _safe_minidom
|
||||
|
||||
# Explicit security flags — pinned so a defusedxml upgrade that changes
|
||||
# defaults cannot silently weaken the boundary.
|
||||
#
|
||||
# forbid_dtd is False because FCPXML files legitimately include
|
||||
# <!DOCTYPE fcpxml> — blocking all DTDs would reject every real FCP export.
|
||||
# forbid_entities and forbid_external block the dangerous payloads
|
||||
# (entity expansion bombs, external entity reads, remote DTD fetches).
|
||||
_SECURITY_FLAGS = {
|
||||
"forbid_dtd": False,
|
||||
"forbid_entities": True,
|
||||
"forbid_external": True,
|
||||
}
|
||||
|
||||
|
||||
def safe_parse(source: str) -> ET.ElementTree:
|
||||
"""Parse an XML file with XXE and entity-expansion protection.
|
||||
|
||||
All DTD processing, entity definitions, and external references are
|
||||
rejected outright. Returns a standard ElementTree so downstream code
|
||||
is unchanged.
|
||||
"""
|
||||
return _safe_ET.parse(source, **_SECURITY_FLAGS)
|
||||
|
||||
|
||||
def safe_fromstring(text: str) -> ET.Element:
|
||||
"""Parse an XML string with XXE and entity-expansion protection.
|
||||
|
||||
All DTD processing, entity definitions, and external references are
|
||||
rejected outright. Returns a standard Element so downstream code is
|
||||
unchanged.
|
||||
"""
|
||||
return _safe_ET.fromstring(text, **_SECURITY_FLAGS)
|
||||
|
||||
|
||||
def safe_parse_string(text: str) -> Document:
|
||||
"""Parse an XML string into a minidom Document with XXE protection.
|
||||
|
||||
Drop-in replacement for xml.dom.minidom.parseString(). Used in the
|
||||
pretty-print path (writer.py, export.py) to maintain defense-in-depth
|
||||
even when re-parsing XML that was already produced by safe_parse().
|
||||
|
||||
Why this matters: if a future refactor passes unsanitized XML through
|
||||
serialize_xml(), stdlib minidom would silently process external
|
||||
entities and DTD bombs. Using defusedxml.minidom closes that gap.
|
||||
"""
|
||||
return _safe_minidom.parseString(text)
|
||||
|
||||
|
||||
def serialize_xml(root: ET.Element, filepath: str, doctype: str = "") -> str:
|
||||
"""Pretty-print an ElementTree root and write to disk.
|
||||
|
||||
Single serialization pipeline shared by all XML output paths
|
||||
(write_fcpxml for FCPXML, DaVinciExporter for XMEML / simplified exports).
|
||||
Consolidates the ET.tostring → minidom → toprettyxml → strip blanks →
|
||||
replace declaration → write-to-file sequence that was previously
|
||||
duplicated across writer.py and export.py.
|
||||
|
||||
Args:
|
||||
root: The XML root Element to serialize.
|
||||
filepath: Destination file path.
|
||||
doctype: DOCTYPE declaration to insert after the XML declaration.
|
||||
Example: ``'<!DOCTYPE fcpxml>'`` or ``'<!DOCTYPE xmeml>'``.
|
||||
If empty, no DOCTYPE is inserted.
|
||||
|
||||
Returns:
|
||||
The filepath written to.
|
||||
"""
|
||||
xml_str = ET.tostring(root, encoding='unicode')
|
||||
dom = safe_parse_string(xml_str)
|
||||
pretty_xml = dom.toprettyxml(indent=" ")
|
||||
lines = [line for line in pretty_xml.split('\n') if line.strip()]
|
||||
final_xml = '\n'.join(lines)
|
||||
|
||||
if doctype:
|
||||
final_xml = final_xml.replace(
|
||||
'<?xml version="1.0" ?>',
|
||||
f'<?xml version="1.0" encoding="UTF-8"?>\n{doctype}',
|
||||
)
|
||||
else:
|
||||
final_xml = final_xml.replace(
|
||||
'<?xml version="1.0" ?>',
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
)
|
||||
|
||||
with open(filepath, 'w', encoding='utf-8') as f:
|
||||
f.write(final_xml)
|
||||
return filepath
|
||||
Executable
+387
@@ -0,0 +1,387 @@
|
||||
"""
|
||||
Template System — Pre-built timeline structures for common editing patterns.
|
||||
|
||||
Templates define slot-based layouts that can be filled with clips to generate
|
||||
complete FCPXML timelines. Each template has named slots (video, audio, title,
|
||||
gap) with duration constraints and lane assignments.
|
||||
"""
|
||||
|
||||
import uuid
|
||||
import xml.etree.ElementTree as ET
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from .models import TimeValue
|
||||
from .writer import (
|
||||
_create_asset_element,
|
||||
_sanitize_xml_value,
|
||||
write_fcpxml,
|
||||
)
|
||||
|
||||
# ============================================================================
|
||||
# DATA CLASSES
|
||||
# ============================================================================
|
||||
|
||||
@dataclass
|
||||
class TemplateSlot:
|
||||
"""A slot in a template that can be filled with a clip.
|
||||
|
||||
Attributes:
|
||||
name: Unique slot name (e.g. "intro", "main", "music_bed").
|
||||
slot_type: Type of content: "video", "audio", "title", or "gap".
|
||||
min_duration: Minimum duration in seconds (0 = no minimum).
|
||||
max_duration: Maximum duration in seconds (0 = no maximum).
|
||||
default_duration: Default duration if no clip duration specified.
|
||||
lane: Lane assignment (0 = primary spine, positive = above, negative = below).
|
||||
role: Audio/video role (e.g. "music", "dialogue", "titles").
|
||||
required: Whether this slot must be filled.
|
||||
"""
|
||||
name: str
|
||||
slot_type: str = "video"
|
||||
min_duration: float = 0.0
|
||||
max_duration: float = 0.0
|
||||
default_duration: float = 5.0
|
||||
lane: int = 0
|
||||
role: str = ""
|
||||
required: bool = True
|
||||
|
||||
|
||||
@dataclass
|
||||
class Template:
|
||||
"""A timeline template with named slots.
|
||||
|
||||
Attributes:
|
||||
name: Template identifier (e.g. "intro_outro").
|
||||
description: Human-readable description.
|
||||
slots: Ordered list of template slots.
|
||||
"""
|
||||
name: str
|
||||
description: str
|
||||
slots: List[TemplateSlot] = field(default_factory=list)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ClipSpec:
|
||||
"""Specification for filling a template slot.
|
||||
|
||||
Provide either asset_id (for existing assets) or src (for new media).
|
||||
|
||||
Attributes:
|
||||
asset_id: Reference to an existing asset in the FCPXML.
|
||||
src: File path for new media.
|
||||
name: Clip display name.
|
||||
duration: Override duration in seconds (uses slot default if not set).
|
||||
"""
|
||||
asset_id: Optional[str] = None
|
||||
src: Optional[str] = None
|
||||
name: str = "Untitled"
|
||||
duration: Optional[float] = None
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# BUILTIN TEMPLATES
|
||||
# ============================================================================
|
||||
|
||||
BUILTIN_TEMPLATES: Dict[str, Template] = {
|
||||
"intro_outro": Template(
|
||||
name="intro_outro",
|
||||
description=(
|
||||
"Title card + main content + end card with optional music bed. "
|
||||
"Classic YouTube/corporate structure."
|
||||
),
|
||||
slots=[
|
||||
TemplateSlot(
|
||||
name="intro_card", slot_type="video",
|
||||
default_duration=5.0, max_duration=15.0,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="main_content", slot_type="video",
|
||||
default_duration=60.0, min_duration=5.0,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="end_card", slot_type="video",
|
||||
default_duration=5.0, max_duration=15.0,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="music_bed", slot_type="audio",
|
||||
default_duration=0.0, lane=-1, role="music",
|
||||
required=False,
|
||||
),
|
||||
],
|
||||
),
|
||||
"lower_thirds": Template(
|
||||
name="lower_thirds",
|
||||
description=(
|
||||
"Main content with title overlay positions at lane +1. "
|
||||
"Useful for interview graphics, name supers."
|
||||
),
|
||||
slots=[
|
||||
TemplateSlot(
|
||||
name="main_content", slot_type="video",
|
||||
default_duration=60.0, min_duration=5.0,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="lower_third_1", slot_type="title",
|
||||
default_duration=4.0, max_duration=10.0,
|
||||
lane=1, role="titles", required=False,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="lower_third_2", slot_type="title",
|
||||
default_duration=4.0, max_duration=10.0,
|
||||
lane=1, role="titles", required=False,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="lower_third_3", slot_type="title",
|
||||
default_duration=4.0, max_duration=10.0,
|
||||
lane=1, role="titles", required=False,
|
||||
),
|
||||
],
|
||||
),
|
||||
"music_video": Template(
|
||||
name="music_video",
|
||||
description=(
|
||||
"A/B roll structure with music bed. "
|
||||
"Alternating performance and cutaway shots over a music track."
|
||||
),
|
||||
slots=[
|
||||
TemplateSlot(
|
||||
name="a_roll_1", slot_type="video",
|
||||
default_duration=8.0,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="b_roll_1", slot_type="video",
|
||||
default_duration=4.0,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="a_roll_2", slot_type="video",
|
||||
default_duration=8.0,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="b_roll_2", slot_type="video",
|
||||
default_duration=4.0,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="a_roll_3", slot_type="video",
|
||||
default_duration=8.0,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="b_roll_3", slot_type="video",
|
||||
default_duration=4.0,
|
||||
),
|
||||
TemplateSlot(
|
||||
name="music_bed", slot_type="audio",
|
||||
default_duration=0.0, lane=-1, role="music",
|
||||
required=False,
|
||||
),
|
||||
],
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# PUBLIC API
|
||||
# ============================================================================
|
||||
|
||||
def list_templates() -> List[Dict[str, Any]]:
|
||||
"""Return all available templates with their slot definitions.
|
||||
|
||||
Returns:
|
||||
List of dicts with template name, description, and slot details.
|
||||
"""
|
||||
result = []
|
||||
for name, tmpl in BUILTIN_TEMPLATES.items():
|
||||
result.append({
|
||||
'name': tmpl.name,
|
||||
'description': tmpl.description,
|
||||
'slots': [
|
||||
{
|
||||
'name': s.name,
|
||||
'slot_type': s.slot_type,
|
||||
'default_duration': s.default_duration,
|
||||
'min_duration': s.min_duration,
|
||||
'max_duration': s.max_duration,
|
||||
'lane': s.lane,
|
||||
'role': s.role,
|
||||
'required': s.required,
|
||||
}
|
||||
for s in tmpl.slots
|
||||
],
|
||||
})
|
||||
return result
|
||||
|
||||
|
||||
def apply_template(
|
||||
template_name: str,
|
||||
clips_map: Dict[str, ClipSpec],
|
||||
output_path: str,
|
||||
fps: float = 24.0,
|
||||
) -> str:
|
||||
"""Fill a template with clips and write the resulting FCPXML.
|
||||
|
||||
Args:
|
||||
template_name: Name of a builtin template (e.g. "intro_outro").
|
||||
clips_map: Dict mapping slot names to ClipSpec objects.
|
||||
output_path: Where to write the generated FCPXML.
|
||||
fps: Frame rate for the output timeline.
|
||||
|
||||
Returns:
|
||||
The output file path.
|
||||
|
||||
Raises:
|
||||
ValueError: If template not found or required slots are missing.
|
||||
"""
|
||||
template = BUILTIN_TEMPLATES.get(template_name)
|
||||
if template is None:
|
||||
available = ', '.join(BUILTIN_TEMPLATES.keys())
|
||||
raise ValueError(
|
||||
f"Template '{template_name}' not found. Available: {available}"
|
||||
)
|
||||
|
||||
# Validate required slots
|
||||
for slot in template.slots:
|
||||
if slot.required and slot.name not in clips_map:
|
||||
raise ValueError(
|
||||
f"Required slot '{slot.name}' not filled in template '{template_name}'."
|
||||
)
|
||||
|
||||
# Build FCPXML
|
||||
fps_int = int(fps)
|
||||
root = ET.Element('fcpxml', version='1.13')
|
||||
resources = ET.SubElement(root, 'resources')
|
||||
|
||||
# Format resource
|
||||
format_id = 'r1'
|
||||
ET.SubElement(resources, 'format',
|
||||
id=format_id,
|
||||
name=f"FFVideoFormat1080p{fps_int}",
|
||||
frameDuration=f"1/{fps_int}s",
|
||||
width="1920", height="1080")
|
||||
|
||||
# Create assets for each filled slot
|
||||
asset_map: Dict[str, str] = {} # slot_name → asset_id
|
||||
asset_counter = 2
|
||||
for slot_name, spec in clips_map.items():
|
||||
if spec.asset_id:
|
||||
asset_map[slot_name] = spec.asset_id
|
||||
elif spec.src:
|
||||
aid = f"r{asset_counter}"
|
||||
asset_counter += 1
|
||||
slot = next((s for s in template.slots if s.name == slot_name), None)
|
||||
dur = spec.duration or (slot.default_duration if slot else 5.0)
|
||||
dur_tv = TimeValue.from_seconds(dur, fps)
|
||||
_create_asset_element(
|
||||
resources, aid, spec.name, spec.src,
|
||||
duration=dur_tv.to_fcpxml(),
|
||||
has_video="1" if (not slot or slot.slot_type != "audio") else "0",
|
||||
has_audio="1",
|
||||
)
|
||||
asset_map[slot_name] = aid
|
||||
|
||||
# Library/event/project structure
|
||||
library = ET.SubElement(root, 'library',
|
||||
location="file:///Users/editor/Movies/Template.fcpbundle/")
|
||||
event = ET.SubElement(library, 'event',
|
||||
name=template_name, uid=str(uuid.uuid4()).upper())
|
||||
project_elem = ET.SubElement(event, 'project',
|
||||
name=template.name,
|
||||
uid=str(uuid.uuid4()).upper(),
|
||||
modDate=datetime.now().strftime("%Y-%m-%d %H:%M:%S -0500"))
|
||||
|
||||
# Calculate total spine duration (only lane-0 slots)
|
||||
total_duration = TimeValue.zero()
|
||||
for slot in template.slots:
|
||||
if slot.lane != 0:
|
||||
continue
|
||||
spec = clips_map.get(slot.name)
|
||||
if spec:
|
||||
dur = spec.duration or slot.default_duration
|
||||
else:
|
||||
dur = slot.default_duration
|
||||
total_duration = total_duration + TimeValue.from_seconds(dur, fps)
|
||||
|
||||
sequence = ET.SubElement(project_elem, 'sequence',
|
||||
format=format_id,
|
||||
duration=total_duration.to_fcpxml(),
|
||||
tcStart="0s", tcFormat="NDF",
|
||||
audioLayout="stereo", audioRate="48k")
|
||||
spine = ET.SubElement(sequence, 'spine')
|
||||
|
||||
# Fill spine slots (lane 0) and track connected slots
|
||||
current_offset = TimeValue.zero()
|
||||
spine_clips: Dict[str, ET.Element] = {} # slot_name → spine clip element
|
||||
|
||||
for slot in template.slots:
|
||||
if slot.lane != 0:
|
||||
continue
|
||||
|
||||
spec = clips_map.get(slot.name)
|
||||
dur = (spec.duration if spec and spec.duration else slot.default_duration)
|
||||
dur_tv = TimeValue.from_seconds(dur, fps)
|
||||
aid = asset_map.get(slot.name, format_id)
|
||||
|
||||
if spec:
|
||||
clip_elem = ET.SubElement(spine, 'asset-clip',
|
||||
ref=aid,
|
||||
offset=current_offset.to_fcpxml(),
|
||||
name=_sanitize_xml_value(spec.name, 512),
|
||||
start="0s",
|
||||
duration=dur_tv.to_fcpxml(),
|
||||
format=format_id,
|
||||
tcFormat="NDF")
|
||||
else:
|
||||
# Gap placeholder for unfilled optional slot
|
||||
clip_elem = ET.SubElement(spine, 'gap',
|
||||
name=slot.name,
|
||||
offset=current_offset.to_fcpxml(),
|
||||
duration=dur_tv.to_fcpxml())
|
||||
|
||||
spine_clips[slot.name] = clip_elem
|
||||
current_offset = current_offset + dur_tv
|
||||
|
||||
# Fill connected slots (lane != 0) — attach to first spine clip
|
||||
first_spine_clip = None
|
||||
for s in template.slots:
|
||||
if s.lane == 0 and s.name in spine_clips:
|
||||
first_spine_clip = spine_clips[s.name]
|
||||
break
|
||||
|
||||
for slot in template.slots:
|
||||
if slot.lane == 0:
|
||||
continue
|
||||
spec = clips_map.get(slot.name)
|
||||
if not spec:
|
||||
continue
|
||||
|
||||
dur = spec.duration or slot.default_duration
|
||||
dur_tv = TimeValue.from_seconds(dur, fps)
|
||||
aid = asset_map.get(slot.name, format_id)
|
||||
|
||||
parent = first_spine_clip if first_spine_clip is not None else spine
|
||||
|
||||
if slot.slot_type == "audio":
|
||||
# Audio: spans total if duration is 0
|
||||
if slot.default_duration == 0.0 and spec.duration is None:
|
||||
dur_tv = total_duration
|
||||
connected = ET.SubElement(parent, 'asset-clip',
|
||||
ref=aid,
|
||||
lane=str(slot.lane),
|
||||
offset="0s",
|
||||
name=_sanitize_xml_value(spec.name, 512),
|
||||
start="0s",
|
||||
duration=dur_tv.to_fcpxml(),
|
||||
audioRole=slot.role or "music")
|
||||
else:
|
||||
# Title or video overlay
|
||||
connected = ET.SubElement(parent, 'asset-clip',
|
||||
ref=aid,
|
||||
lane=str(slot.lane),
|
||||
offset="0s",
|
||||
name=_sanitize_xml_value(spec.name, 512),
|
||||
start="0s",
|
||||
duration=dur_tv.to_fcpxml())
|
||||
if slot.role:
|
||||
connected.set('videoRole', slot.role)
|
||||
|
||||
write_fcpxml(root, output_path)
|
||||
return output_path
|
||||
@@ -0,0 +1,826 @@
|
||||
"""Text measurement and block layout for kinetic-typography subtitles.
|
||||
|
||||
Lays a sentence's words out as a compact block — words packed onto lines,
|
||||
lines stacked and centered — so each word can be emitted as its own positioned
|
||||
``<title>`` without ever overlapping a sibling.
|
||||
|
||||
CALIBRATION
|
||||
-----------
|
||||
Every constant here is derived from a real Final Cut export the user built by
|
||||
hand and sent back ("Exemplo Letra.fcpxmld", project 2160x3840, template
|
||||
"Essencial - Título"). The five hand-placed words of its first sentence:
|
||||
|
||||
word position fontSize kerning
|
||||
Toda -251.828 -109 170 2.72
|
||||
a 2.236 -101.135 128 —
|
||||
minha 267.319 -95 151 2.416
|
||||
vida, 71.0395 -233.65 128 2.048
|
||||
assim, 210.88 -100 128 2.048
|
||||
|
||||
Three facts fall out of those numbers, and they are why this module can place
|
||||
words at all:
|
||||
|
||||
1. **Position uses the same unit as fontSize.** Center-to-center distances on
|
||||
line 1 are 254.06 (Toda→a) and 265.08 (a→minha). Half-width sums from the
|
||||
Helvetica metrics below at those sizes give 245.1 and 265.0 — matching to
|
||||
within a few units, which is the slop of dragging by hand. A different unit
|
||||
would have shown up as a constant ratio; there is none.
|
||||
2. **The canvas is 1080x1920 points** — half of the 2160x3840 frame, because
|
||||
Final Cut positions in points over 2x media. Line 1 spans -460.3 to +495.7,
|
||||
which fills that width with small side margins, exactly as the reference
|
||||
frame looks.
|
||||
3. **y grows upward.** "vida," (line 2) sits at -233.65 against line 1's ~-101.
|
||||
|
||||
Widths are still ESTIMATES — Final Cut renders the real glyphs — so they are
|
||||
computed generously. Overestimating costs a little empty space; underestimating
|
||||
makes two words collide, which is the one failure this module exists to
|
||||
prevent.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional, Sequence
|
||||
|
||||
from .font_metrics import METRICS, VERTICAL_METRICS
|
||||
|
||||
# Fallback advance widths (standard Helvetica AFM), used only for a font the
|
||||
# embedded metrics do not cover. Real per-variant metrics live in
|
||||
# font_metrics.METRICS and are preferred — see measure_text.
|
||||
_HELVETICA_WIDTHS = {
|
||||
' ': 0.278, '!': 0.278, '"': 0.355, '#': 0.556, '$': 0.556, '%': 0.889,
|
||||
'&': 0.667, "'": 0.191, '(': 0.333, ')': 0.333, '*': 0.389, '+': 0.584,
|
||||
',': 0.278, '-': 0.333, '.': 0.278, '/': 0.278, ':': 0.278, ';': 0.278,
|
||||
'<': 0.584, '=': 0.584, '>': 0.584, '?': 0.556, '@': 1.015, '[': 0.278,
|
||||
'\\': 0.278, ']': 0.278, '^': 0.469, '_': 0.556, '`': 0.333, '{': 0.334,
|
||||
'|': 0.260, '}': 0.334, '~': 0.584,
|
||||
'A': 0.667, 'B': 0.667, 'C': 0.722, 'D': 0.722, 'E': 0.667, 'F': 0.611,
|
||||
'G': 0.778, 'H': 0.722, 'I': 0.278, 'J': 0.500, 'K': 0.667, 'L': 0.556,
|
||||
'M': 0.833, 'N': 0.722, 'O': 0.778, 'P': 0.667, 'Q': 0.778, 'R': 0.722,
|
||||
'S': 0.667, 'T': 0.611, 'U': 0.722, 'V': 0.667, 'W': 0.944, 'X': 0.667,
|
||||
'Y': 0.667, 'Z': 0.611,
|
||||
'a': 0.556, 'b': 0.556, 'c': 0.500, 'd': 0.556, 'e': 0.556, 'f': 0.278,
|
||||
'g': 0.556, 'h': 0.556, 'i': 0.222, 'j': 0.222, 'k': 0.500, 'l': 0.222,
|
||||
'm': 0.833, 'n': 0.556, 'o': 0.556, 'p': 0.556, 'q': 0.556, 'r': 0.333,
|
||||
's': 0.500, 't': 0.278, 'u': 0.556, 'v': 0.500, 'w': 0.722, 'x': 0.500,
|
||||
'y': 0.500, 'z': 0.500,
|
||||
}
|
||||
_FALLBACK_WIDTH = 0.556
|
||||
|
||||
# Accented Portuguese letters advance like their base letter.
|
||||
_ACCENT_BASE = str.maketrans(
|
||||
'áàâãäéèêëíìîïóòôõöúùûüçñÁÀÂÃÄÉÈÊËÍÌÎÏÓÒÔÕÖÚÙÛÜÇÑ',
|
||||
'aaaaaeeeeiiiiooooouuuucnAAAAAEEEEIIIIOOOOOUUUUCN',
|
||||
)
|
||||
|
||||
# Bold thickens every stem. Only applied on the fallback path — the embedded
|
||||
# metrics already carry the bold variant's own advances.
|
||||
_BOLD_FACTOR = 1.06
|
||||
|
||||
# Cushion over the computed advance. Small, because the embedded metrics are
|
||||
# exact: it covers the renderer's own rounding and any glyph outside the table,
|
||||
# nothing more.
|
||||
_SAFETY_MARGIN = 1.02
|
||||
|
||||
# The frame is 2160x3840 but Final Cut positions in points over 2x media, so
|
||||
# the coordinate canvas is half that. See the calibration note above.
|
||||
POINT_SCALE = 0.5
|
||||
|
||||
# The canvas the reference sizes were chosen against: 3840px tall at 2x. Other
|
||||
# formats scale off this, so a 170pt word keeps the same share of frame height
|
||||
# (8.9%) on a landscape timeline as it has on the user's vertical one.
|
||||
REFERENCE_CANVAS_HEIGHT = 3840.0 * POINT_SCALE
|
||||
|
||||
# Reference values read off the calibration export.
|
||||
REFERENCE_FONT_SIZE_LARGE = 170
|
||||
REFERENCE_FONT_SIZE_MEDIUM = 128
|
||||
REFERENCE_FONT_SIZE_ALT = 151
|
||||
REFERENCE_KERNING = 2.048
|
||||
# Center of the hand-placed block: line 1 at y≈-101, line 2 at y≈-233.65.
|
||||
REFERENCE_BLOCK_CENTER_Y = -167.0
|
||||
# Gap between words on a line, as a fraction of the larger neighbour's font
|
||||
# size. The reference export's own gaps work out to ~10-12 points at 170pt
|
||||
# (0.06), but that leaves only a few points of slack once glyph-metric error is
|
||||
# accounted for — one bad estimate and two words touch. This is deliberately
|
||||
# roomier: still a tight typographic block, with margin that survives the
|
||||
# estimate being off.
|
||||
REFERENCE_WORD_GAP_RATIO = 0.14
|
||||
|
||||
# Text occupies roughly cap-height, not the full em box, so a line's visual
|
||||
# height is well under its font size. The reference export puts line 1 (max
|
||||
# 170pt) and line 2 (128pt) 132.65 apart; with this ratio their half-heights
|
||||
# sum to 111.75, leaving the ~20pt of breathing room below. Using the full em
|
||||
# box instead would space the lines ~40% further apart than the user did.
|
||||
_CAP_HEIGHT_RATIO = 0.75
|
||||
REFERENCE_LINE_GAP = 20.0
|
||||
|
||||
# Progressive composition sets its lines much tighter than the word-by-word
|
||||
# block: in the reference the small grotesque lines almost touch the display
|
||||
# italic between them. Small but never negative — overlapping boxes is the one
|
||||
# failure this module exists to prevent.
|
||||
REFERENCE_BLOCK_LINE_GAP = 8.0
|
||||
# How far a body line slides toward its side of the emphasis line, as a
|
||||
# fraction of the slack between the two widths. 1.0 would flush it against the
|
||||
# emphasis line's edge; the reference leaves a little air.
|
||||
REFERENCE_STAGGER_RATIO = 0.8
|
||||
|
||||
|
||||
def metrics_for(font: Optional[str], face: Optional[str] = None) -> Optional[Dict]:
|
||||
"""The embedded advance table for *font*/*face*, or None if uncovered.
|
||||
|
||||
Falls back from the exact "family-face" key to the bare family, so an
|
||||
unknown face still measures against the right family rather than a
|
||||
generic table.
|
||||
"""
|
||||
if not font:
|
||||
return None
|
||||
family = font.strip().lower()
|
||||
if face:
|
||||
exact = METRICS.get(f"{family}-{face.strip().lower()}")
|
||||
if exact:
|
||||
return exact
|
||||
return METRICS.get(family)
|
||||
|
||||
|
||||
def char_width_ratio(ch: str, table: Optional[Dict] = None) -> float:
|
||||
"""Return *ch*'s advance width as a fraction of the font size."""
|
||||
if table is not None and ch in table:
|
||||
return table[ch]
|
||||
base = ch.translate(_ACCENT_BASE)
|
||||
if table is not None and base in table:
|
||||
return table[base]
|
||||
return _HELVETICA_WIDTHS.get(base, _FALLBACK_WIDTH)
|
||||
|
||||
|
||||
def measure_text(
|
||||
text: str,
|
||||
font_size: float,
|
||||
*,
|
||||
bold: bool = False,
|
||||
kerning: float = REFERENCE_KERNING,
|
||||
font: Optional[str] = None,
|
||||
face: Optional[str] = None,
|
||||
) -> float:
|
||||
"""Width of *text* rendered at *font_size*, in canvas points.
|
||||
|
||||
Measured against the real advance widths of the macOS font when *font*
|
||||
names one this module carries metrics for, which is the case for every
|
||||
font the subtitle rhythm uses. Otherwise falls back to generic Helvetica
|
||||
advances, which is an estimate.
|
||||
|
||||
``kerning`` is Final Cut's per-character tracking, in the same unit as
|
||||
font size — the calibration export carries 2.048 to 2.72.
|
||||
"""
|
||||
if not text:
|
||||
return 0.0
|
||||
table = metrics_for(font, face)
|
||||
width = sum(char_width_ratio(ch, table) for ch in text) * font_size
|
||||
width += kerning * len(text)
|
||||
if bold and table is None:
|
||||
# The embedded tables already carry the bold variant's own advances;
|
||||
# only the generic fallback needs a correction factor.
|
||||
width *= _BOLD_FACTOR
|
||||
return width * _SAFETY_MARGIN
|
||||
|
||||
|
||||
# Glyph classes for the vertical ink extent of a line. A line's real top and
|
||||
# bottom depend on WHICH characters it contains: "sua legenda" reaches the
|
||||
# x-height and dips to the descender of its g; "que vão" adds the tilde above.
|
||||
# Measuring the class actually present keeps the stack as tight as the
|
||||
# reference without ever letting two lines touch.
|
||||
_ACCENTED_UPPER = set('ÁÀÂÃÄÉÈÊËÍÌÎÏÓÒÔÕÖÚÙÛÜÑÇ')
|
||||
_ACCENTED_LOWER = set('áàâãäéèêëíìîïóòôõöúùûüñ')
|
||||
_ASCENDERS = set('bdfhklt')
|
||||
_DESCENDERS = set('gjpqyçÇ')
|
||||
_CAPS = set('ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789')
|
||||
# Used when a font carries no measured vertical metrics: Helvetica's, which
|
||||
# are typical for a grotesque and tighter than a display serif's, plus a
|
||||
# cushion so an unmeasured display face still clears its neighbour.
|
||||
_FALLBACK_VERTICAL = {
|
||||
'ascent': 0.98, 'descent': -0.25, 'cap': 0.72, 'x_height': 0.53,
|
||||
'ascender': 0.73, 'accent_upper': 0.95, 'accent_lower': 0.80,
|
||||
'descender': -0.22,
|
||||
}
|
||||
_UNMEASURED_VERTICAL_CUSHION = 1.08
|
||||
|
||||
|
||||
def vertical_metrics_for(font: Optional[str], face: Optional[str] = None) -> tuple:
|
||||
"""``(metrics, measured)`` for *font*/*face* — the same lookup as metrics_for."""
|
||||
family = (font or '').strip().lower()
|
||||
if face:
|
||||
exact = VERTICAL_METRICS.get(f"{family}-{face.strip().lower()}")
|
||||
if exact:
|
||||
return exact, True
|
||||
table = VERTICAL_METRICS.get(family)
|
||||
if table:
|
||||
return table, True
|
||||
return _FALLBACK_VERTICAL, False
|
||||
|
||||
|
||||
def ink_extent(
|
||||
text: str,
|
||||
font_size: float,
|
||||
*,
|
||||
font: Optional[str] = None,
|
||||
face: Optional[str] = None,
|
||||
) -> tuple:
|
||||
"""``(top, bottom)`` of the rendered ink, in points around the title's y.
|
||||
|
||||
Final Cut centres the LINE BOX — ascent to descent — on the title's
|
||||
Position when vertical alignment is Middle, so the ink sits off-centre by
|
||||
however asymmetric the font is. Both values are returned relative to that
|
||||
centre: positive up, negative down.
|
||||
"""
|
||||
metrics, measured = vertical_metrics_for(font, face)
|
||||
baseline = -(metrics['ascent'] + metrics['descent']) / 2
|
||||
|
||||
top = metrics['x_height']
|
||||
bottom = 0.0
|
||||
for ch in text:
|
||||
if ch in _ACCENTED_UPPER:
|
||||
top = max(top, metrics['accent_upper'])
|
||||
elif ch in _ACCENTED_LOWER:
|
||||
top = max(top, metrics['accent_lower'])
|
||||
elif ch in _CAPS:
|
||||
top = max(top, metrics['cap'])
|
||||
elif ch in _ASCENDERS:
|
||||
top = max(top, metrics['ascender'])
|
||||
if ch in _DESCENDERS:
|
||||
bottom = min(bottom, metrics['descender'])
|
||||
if not measured:
|
||||
top *= _UNMEASURED_VERTICAL_CUSHION
|
||||
bottom *= _UNMEASURED_VERTICAL_CUSHION
|
||||
return (baseline + top) * font_size, (baseline + bottom) * font_size
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlacedWord:
|
||||
"""One word positioned inside a laid-out block.
|
||||
|
||||
``x``/``y`` are the word's CENTER in canvas points, origin at frame center,
|
||||
y growing upward — the convention Final Cut's position param uses, and the
|
||||
value that goes straight into the title's "Posição" param.
|
||||
"""
|
||||
text: str
|
||||
font_size: float
|
||||
italic: bool
|
||||
x: float
|
||||
y: float
|
||||
width: float
|
||||
height: float
|
||||
line_index: int
|
||||
source: Optional[Dict] = None
|
||||
look: Optional[object] = None # the WordLook this word was set with
|
||||
kerning: float = REFERENCE_KERNING # scaled to the frame, as font_size is
|
||||
|
||||
@property
|
||||
def left(self) -> float:
|
||||
return self.x - self.width / 2
|
||||
|
||||
@property
|
||||
def right(self) -> float:
|
||||
return self.x + self.width / 2
|
||||
|
||||
@property
|
||||
def bottom(self) -> float:
|
||||
return self.y - self.height / 2
|
||||
|
||||
@property
|
||||
def top(self) -> float:
|
||||
return self.y + self.height / 2
|
||||
|
||||
def overlaps(self, other: 'PlacedWord') -> bool:
|
||||
"""True if this word's box intersects *other*'s."""
|
||||
return (
|
||||
self.left < other.right
|
||||
and other.left < self.right
|
||||
and self.bottom < other.top
|
||||
and other.bottom < self.top
|
||||
)
|
||||
|
||||
def position_param(self) -> str:
|
||||
"""The value for the title's "Posição" param, as FCP writes it."""
|
||||
return f"{self.x:g} {self.y:g}"
|
||||
|
||||
# The rest of this block is the interface a placed unit shares with
|
||||
# PlacedBlock, so the writer emits titles from either without caring
|
||||
# whether the composition is per word or per phrase.
|
||||
|
||||
@property
|
||||
def words(self) -> List[Dict]:
|
||||
return [self.source] if self.source else []
|
||||
|
||||
@property
|
||||
def start(self) -> float:
|
||||
return float(self.source.get('start', 0.0)) if self.source else 0.0
|
||||
|
||||
@property
|
||||
def end(self) -> float:
|
||||
return float(self.source.get('end', 0.0)) if self.source else 0.0
|
||||
|
||||
@property
|
||||
def font(self) -> str:
|
||||
return getattr(self.look, 'font', None) or 'Helvetica Neue'
|
||||
|
||||
@property
|
||||
def face(self) -> Optional[str]:
|
||||
return getattr(self.look, 'face', None)
|
||||
|
||||
@property
|
||||
def color(self) -> str:
|
||||
return getattr(self.look, 'color', '1 1 1 1')
|
||||
|
||||
|
||||
@dataclass
|
||||
class BlockLayout:
|
||||
"""A laid-out group of words, plus whatever did not fit."""
|
||||
placed: List[PlacedWord] = field(default_factory=list)
|
||||
overflow: List[Dict] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def fitted_count(self) -> int:
|
||||
return len(self.placed)
|
||||
|
||||
|
||||
@dataclass
|
||||
class LayoutBox:
|
||||
"""The usable area words may occupy, in canvas points.
|
||||
|
||||
Defaults reproduce the calibration export: a block centered slightly below
|
||||
frame center, spanning most of the width of a 2160x3840 vertical frame.
|
||||
"""
|
||||
width: float = 1080.0 * 0.92
|
||||
height: float = 1920.0 * 0.22
|
||||
center_y: float = REFERENCE_BLOCK_CENTER_Y
|
||||
font_scale: float = 1.0
|
||||
|
||||
@classmethod
|
||||
def for_frame(
|
||||
cls,
|
||||
frame_width: float,
|
||||
frame_height: float,
|
||||
*,
|
||||
side_margin: float = 0.04,
|
||||
band_height: float = 0.22,
|
||||
center_y: Optional[float] = None,
|
||||
) -> 'LayoutBox':
|
||||
"""Build a box for a frame of *frame_width* x *frame_height* pixels.
|
||||
|
||||
Pixels are converted to canvas points via ``POINT_SCALE``. Type sizes
|
||||
and the block's height scale with the frame, so the same rhythm reads
|
||||
proportionally on any format rather than overflowing a shorter one.
|
||||
``center_y`` defaults to the calibration export's height, scaled.
|
||||
"""
|
||||
w = frame_width * POINT_SCALE
|
||||
h = frame_height * POINT_SCALE
|
||||
scale = h / REFERENCE_CANVAS_HEIGHT
|
||||
return cls(
|
||||
width=w * (1.0 - 2 * side_margin),
|
||||
height=h * band_height,
|
||||
center_y=(
|
||||
REFERENCE_BLOCK_CENTER_Y * scale if center_y is None else center_y
|
||||
),
|
||||
font_scale=scale,
|
||||
)
|
||||
|
||||
|
||||
def look_for(index: int, style):
|
||||
"""The look for the word at *index* within its sentence.
|
||||
|
||||
Prefers ``style.look_for`` (WordStyle's rhythm of WordLook entries, which
|
||||
carries size, colour, font and face together). Falls back to the older
|
||||
parallel-pattern attributes so a bare stand-in style still lays out.
|
||||
"""
|
||||
resolver = getattr(style, 'look_for', None)
|
||||
if callable(resolver):
|
||||
return resolver(index)
|
||||
|
||||
class _Fallback:
|
||||
pass
|
||||
|
||||
look = _Fallback()
|
||||
sizes = getattr(style, 'size_pattern', None)
|
||||
look.font_size = (
|
||||
float(sizes[index % len(sizes)]) if sizes
|
||||
else float(getattr(style, 'font_size', REFERENCE_FONT_SIZE_MEDIUM))
|
||||
)
|
||||
italics = getattr(style, 'italic_pattern', None)
|
||||
look.italic = bool(italics[index % len(italics)]) if italics else False
|
||||
look.font = getattr(style, 'font', 'Helvetica Neue')
|
||||
look.face = None
|
||||
look.color = getattr(style, 'active_color', '1 1 1 1')
|
||||
look.kerning = float(getattr(style, 'kerning', REFERENCE_KERNING))
|
||||
return look
|
||||
|
||||
|
||||
def rhythm_font_size(index: int, style) -> float:
|
||||
"""Font size for the word at *index* within its sentence."""
|
||||
return float(look_for(index, style).font_size)
|
||||
|
||||
|
||||
def rhythm_italic(index: int, style) -> bool:
|
||||
"""Whether the word at *index* within its sentence is italic."""
|
||||
return bool(look_for(index, style).italic)
|
||||
|
||||
|
||||
def layout_sentence(
|
||||
words: Sequence[Dict],
|
||||
style,
|
||||
box: Optional[LayoutBox] = None,
|
||||
*,
|
||||
line_gap: float = REFERENCE_LINE_GAP,
|
||||
word_gap_ratio: float = REFERENCE_WORD_GAP_RATIO,
|
||||
) -> BlockLayout:
|
||||
"""Lay *words* out as a centered, line-wrapped block inside *box*.
|
||||
|
||||
Words are packed left to right until the line no longer fits ``box.width``,
|
||||
then a new line opens. Lines are stacked, the stack centered on
|
||||
``box.center_y``, and every line centered horizontally — a compact block
|
||||
with no word ever overlapping another.
|
||||
|
||||
Words that would push the block past ``box.height`` come back in
|
||||
``BlockLayout.overflow`` instead of being placed. The caller starts a fresh
|
||||
block with them, which is what keeps a long sentence from spilling off
|
||||
screen.
|
||||
|
||||
``word_gap_ratio`` sizes the gap between neighbours off the larger of the
|
||||
two font sizes, so a 170pt word is not separated by the same sliver as a
|
||||
128pt one.
|
||||
"""
|
||||
if box is None:
|
||||
box = LayoutBox()
|
||||
|
||||
bold = bool(getattr(style, 'bold', False))
|
||||
default_kerning = float(getattr(style, 'kerning', REFERENCE_KERNING))
|
||||
# Type, spacing and gaps all scale together, or a shorter frame would get
|
||||
# reference-sized words that never fit.
|
||||
scale = float(getattr(box, 'font_scale', 1.0)) or 1.0
|
||||
line_gap *= scale
|
||||
|
||||
def gap_between(left: Dict, right: Dict) -> float:
|
||||
"""Space between two neighbouring words, off the larger of the two."""
|
||||
return max(left['font_size'], right['font_size']) * word_gap_ratio
|
||||
|
||||
tokens = []
|
||||
for w in words:
|
||||
text = str(w.get('word', '')).strip()
|
||||
if not text:
|
||||
continue
|
||||
look = look_for(len(tokens), style)
|
||||
size = float(look.font_size) * scale
|
||||
kerning = float(getattr(look, 'kerning', default_kerning)) * scale
|
||||
tokens.append({
|
||||
'source': w,
|
||||
'text': text,
|
||||
'look': look,
|
||||
'kerning': kerning,
|
||||
'font_size': size,
|
||||
'italic': bool(look.italic),
|
||||
'width': measure_text(
|
||||
text, size, bold=bold, kerning=kerning,
|
||||
font=getattr(look, 'font', None) or getattr(style, 'font', None),
|
||||
face=getattr(look, 'face', None),
|
||||
),
|
||||
'height': size * _CAP_HEIGHT_RATIO,
|
||||
})
|
||||
if not tokens:
|
||||
return BlockLayout()
|
||||
|
||||
# Pack into lines. A word wider than the whole box still gets its own line
|
||||
# rather than being dropped — losing a spoken word is worse than one line
|
||||
# running wide.
|
||||
lines: List[List[Dict]] = []
|
||||
current: List[Dict] = []
|
||||
current_width = 0.0
|
||||
for tok in tokens:
|
||||
gap = gap_between(current[-1], tok) if current else 0.0
|
||||
projected = current_width + gap + tok['width']
|
||||
if current and projected > box.width:
|
||||
lines.append(current)
|
||||
current = [tok]
|
||||
current_width = tok['width']
|
||||
else:
|
||||
current.append(tok)
|
||||
current_width = projected
|
||||
if current:
|
||||
lines.append(current)
|
||||
|
||||
# Keep the leading lines that fit the band; the rest overflow into a new
|
||||
# block. Consecutive lines are half-height + gap + half-height apart, so a
|
||||
# tall word only costs what it actually occupies.
|
||||
line_heights = [max(t['height'] for t in line) for line in lines]
|
||||
kept = 0
|
||||
total_height = 0.0
|
||||
for i, h in enumerate(line_heights):
|
||||
advance = h if not kept else (line_heights[i - 1] + h) / 2 + line_gap
|
||||
if kept and total_height + advance > box.height:
|
||||
break
|
||||
total_height += advance
|
||||
kept += 1
|
||||
kept = max(kept, 1) # always place one line, or the caller never advances
|
||||
|
||||
result = BlockLayout()
|
||||
for line in lines[kept:]:
|
||||
result.overflow.extend(tok['source'] for tok in line)
|
||||
|
||||
# Center the stack: the first line's center sits half the total span above
|
||||
# box.center_y, measuring the span between line CENTERS.
|
||||
span = sum(
|
||||
(line_heights[i - 1] + line_heights[i]) / 2 + line_gap
|
||||
for i in range(1, kept)
|
||||
)
|
||||
cursor_y = box.center_y + span / 2
|
||||
|
||||
for line_index, line in enumerate(lines[:kept]):
|
||||
if line_index:
|
||||
cursor_y -= (
|
||||
(line_heights[line_index - 1] + line_heights[line_index]) / 2
|
||||
+ line_gap
|
||||
)
|
||||
gaps = [gap_between(a, b) for a, b in zip(line, line[1:])]
|
||||
line_width = sum(tok['width'] for tok in line) + sum(gaps)
|
||||
cursor_x = -line_width / 2
|
||||
for position, tok in enumerate(line):
|
||||
if position:
|
||||
cursor_x += gaps[position - 1]
|
||||
result.placed.append(PlacedWord(
|
||||
text=tok['text'],
|
||||
font_size=tok['font_size'],
|
||||
italic=tok['italic'],
|
||||
x=cursor_x + tok['width'] / 2,
|
||||
y=cursor_y,
|
||||
width=tok['width'],
|
||||
height=tok['height'],
|
||||
line_index=line_index,
|
||||
source=tok['source'],
|
||||
look=tok['look'],
|
||||
kerning=tok['kerning'],
|
||||
))
|
||||
cursor_x += tok['width']
|
||||
|
||||
return result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# PROGRESSIVE COMPOSITION (block-per-phrase)
|
||||
# ---------------------------------------------------------------------------
|
||||
# The look the user asked for (reference: fernandoluz.d reel, 2026-08-17):
|
||||
#
|
||||
# [ que vão ]
|
||||
# [ melhorar ]
|
||||
# [ sua legenda ]
|
||||
#
|
||||
# One title per BLOCK, not per word. Supporting words are set small in a
|
||||
# grotesque; the sentence's key word is set large in a display italic, on its
|
||||
# own line. Blocks appear as their first word is spoken and stay on screen, so
|
||||
# the sentence assembles itself; they all clear together.
|
||||
|
||||
# Function words are never the emphasis — "que", "de", "uma" set 2.5x larger
|
||||
# than the rest reads as a mistake, not as a design.
|
||||
STOPWORDS_PT = frozenset("""
|
||||
a as o os um uma uns umas de do da dos das em no na nos nas por para pra pro
|
||||
com sem sob sobre e ou mas que se ao aos à às pelo pela pelos pelas num numa
|
||||
eu tu ele ela nos vos eles elas me te lhe nos vos lhes meu minha seu sua teu
|
||||
tua nosso nossa este esta esse essa aquele aquela isso isto aquilo já não sim
|
||||
muito mais menos tão como quando onde quem qual quais é foi ser estar tem ter
|
||||
vai vou vão era são está estão dos aqui ali lá então porque assim
|
||||
""".split())
|
||||
|
||||
|
||||
def pick_emphasis_index(texts: Sequence[str]) -> int:
|
||||
"""Index of the word to set as the block's emphasis.
|
||||
|
||||
The longest content word, since length is the best proxy available for
|
||||
"the word this sentence is about" without a language model. Ties break
|
||||
toward the middle of the sentence, which is where a designer puts the
|
||||
hero word. A sentence of nothing but function words emphasises its
|
||||
longest word anyway rather than emphasising nothing.
|
||||
"""
|
||||
if not texts:
|
||||
return 0
|
||||
cleaned = [t.strip(".,!?;:…\"'()").lower() for t in texts]
|
||||
middle = (len(texts) - 1) / 2
|
||||
content = [i for i, t in enumerate(cleaned) if t and t not in STOPWORDS_PT]
|
||||
pool = content or list(range(len(texts)))
|
||||
return max(pool, key=lambda i: (len(cleaned[i]), -abs(i - middle)))
|
||||
|
||||
|
||||
@dataclass
|
||||
class PlacedBlock:
|
||||
"""One title's worth of text, positioned as a line of the composition."""
|
||||
text: str
|
||||
words: List[Dict]
|
||||
font: str
|
||||
face: Optional[str]
|
||||
font_size: float
|
||||
color: str
|
||||
kerning: float
|
||||
x: float
|
||||
y: float
|
||||
width: float
|
||||
height: float
|
||||
line_index: int
|
||||
emphasis: bool = False
|
||||
# Where the rendered ink actually reaches, relative to y (see ink_extent).
|
||||
ink_top: float = 0.0
|
||||
ink_bottom: float = 0.0
|
||||
|
||||
@property
|
||||
def left(self) -> float:
|
||||
return self.x - self.width / 2
|
||||
|
||||
@property
|
||||
def right(self) -> float:
|
||||
return self.x + self.width / 2
|
||||
|
||||
@property
|
||||
def bottom(self) -> float:
|
||||
return self.y + self.ink_bottom
|
||||
|
||||
@property
|
||||
def top(self) -> float:
|
||||
return self.y + self.ink_top
|
||||
|
||||
@property
|
||||
def start(self) -> float:
|
||||
"""When this block is spoken — its first word's start, in seconds."""
|
||||
return min(float(w.get('start', 0.0)) for w in self.words)
|
||||
|
||||
@property
|
||||
def end(self) -> float:
|
||||
"""When this block finishes being spoken, in seconds."""
|
||||
return max(float(w.get('end', 0.0)) for w in self.words)
|
||||
|
||||
def position_param(self) -> str:
|
||||
"""The value for the title's "Posição" param, as FCP writes it."""
|
||||
return f"{self.x:g} {self.y:g}"
|
||||
|
||||
def overlaps(self, other: 'PlacedBlock') -> bool:
|
||||
return (
|
||||
self.left < other.right
|
||||
and other.left < self.right
|
||||
and self.bottom < other.top
|
||||
and other.bottom < self.top
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Composition:
|
||||
"""A laid-out group of blocks, plus whatever did not fit."""
|
||||
blocks: List[PlacedBlock] = field(default_factory=list)
|
||||
overflow: List[Dict] = field(default_factory=list)
|
||||
|
||||
|
||||
def compose_sentence(
|
||||
words: Sequence[Dict],
|
||||
style,
|
||||
box: Optional[LayoutBox] = None,
|
||||
*,
|
||||
line_gap: float = REFERENCE_BLOCK_LINE_GAP,
|
||||
stagger_ratio: float = REFERENCE_STAGGER_RATIO,
|
||||
) -> Composition:
|
||||
"""Lay a sentence out as stacked blocks, one title per line.
|
||||
|
||||
The emphasis word takes a line of its own, set in the display face; the
|
||||
words before and after it fill the lines above and below, wrapped at
|
||||
``box.width`` and set in the body face. Body lines are staggered — pushed
|
||||
toward opposite edges of the emphasis line — which is what makes the
|
||||
composition read as diagrammed rather than as a centred caption.
|
||||
|
||||
Lines that would push the stack past ``box.height`` come back in
|
||||
``Composition.overflow`` for the caller to place as the next composition.
|
||||
"""
|
||||
if box is None:
|
||||
box = LayoutBox()
|
||||
scale = float(getattr(box, 'font_scale', 1.0)) or 1.0
|
||||
|
||||
entries = [
|
||||
(w, str(w.get('word', '')).strip())
|
||||
for w in words
|
||||
if str(w.get('word', '')).strip()
|
||||
]
|
||||
if not entries:
|
||||
return Composition()
|
||||
|
||||
texts = [t for _, t in entries]
|
||||
emphasis_index = pick_emphasis_index(texts)
|
||||
emphasis_look = style.look_for_emphasis()
|
||||
body_look = style.look_for_body()
|
||||
|
||||
def measure(text: str, look) -> tuple:
|
||||
size = float(look.font_size) * scale
|
||||
kerning = float(getattr(look, 'kerning', REFERENCE_KERNING)) * scale
|
||||
width = measure_text(
|
||||
text, size,
|
||||
bold=bool(getattr(style, 'bold', False)),
|
||||
kerning=kerning,
|
||||
font=getattr(look, 'font', None),
|
||||
face=getattr(look, 'face', None),
|
||||
)
|
||||
return size, kerning, width
|
||||
|
||||
# Split into lines: everything before the emphasis, the emphasis alone,
|
||||
# everything after. Body runs wrap at the box width so a long lead-in
|
||||
# becomes two lines instead of running off frame.
|
||||
def body_lines(run: List[tuple]) -> List[List[tuple]]:
|
||||
out: List[List[tuple]] = []
|
||||
current: List[tuple] = []
|
||||
for item in run:
|
||||
trial = current + [item]
|
||||
text = ' '.join(t for _, t in trial)
|
||||
if current and measure(text, body_look)[2] > box.width:
|
||||
out.append(current)
|
||||
current = [item]
|
||||
else:
|
||||
current = trial
|
||||
if current:
|
||||
out.append(current)
|
||||
return out
|
||||
|
||||
lines: List[tuple] = [] # (run, look, is_emphasis)
|
||||
for run in body_lines(entries[:emphasis_index]):
|
||||
lines.append((run, body_look, False))
|
||||
lines.append(([entries[emphasis_index]], emphasis_look, True))
|
||||
for run in body_lines(entries[emphasis_index + 1:]):
|
||||
lines.append((run, body_look, False))
|
||||
|
||||
measured = []
|
||||
for run, look, is_emphasis in lines:
|
||||
text = ' '.join(t for _, t in run)
|
||||
size, kerning, width = measure(text, look)
|
||||
# Stack on the real ink each line contains, not on a nominal
|
||||
# cap-height: the display italic's accents and descenders run well
|
||||
# past it, and a nominal box lets them collide with the neighbour.
|
||||
top, bottom = ink_extent(
|
||||
text, size,
|
||||
font=getattr(look, 'font', None), face=getattr(look, 'face', None),
|
||||
)
|
||||
measured.append({
|
||||
'run': run, 'look': look, 'emphasis': is_emphasis, 'text': text,
|
||||
'font_size': size, 'kerning': kerning, 'width': width,
|
||||
'ink_top': top, 'ink_bottom': bottom, 'height': top - bottom,
|
||||
})
|
||||
|
||||
# Keep the leading lines that fit the band; the rest become the next
|
||||
# composition. The emphasis line must survive — a block of only body text
|
||||
# loses the whole point of the look — so if it does not fit, everything
|
||||
# from the emphasis on overflows together.
|
||||
gap = line_gap * scale
|
||||
kept = 0
|
||||
total = 0.0
|
||||
for line in measured:
|
||||
advance = line['height'] if not kept else line['height'] + gap
|
||||
if kept and total + advance > box.height:
|
||||
break
|
||||
total += advance
|
||||
kept += 1
|
||||
kept = max(kept, 1)
|
||||
if not any(line['emphasis'] for line in measured[:kept]):
|
||||
kept = min(kept, next(
|
||||
i for i, line in enumerate(measured) if line['emphasis']
|
||||
)) or 1
|
||||
|
||||
result = Composition()
|
||||
for line in measured[kept:]:
|
||||
result.overflow.extend(w for w, _ in line['run'])
|
||||
|
||||
visible = measured[:kept]
|
||||
# Stack the ink boxes edge to edge with exactly *gap* between them, then
|
||||
# centre the whole stack on the band. Because the boxes are the real ink,
|
||||
# "no overlap" is a property of the arithmetic, not of a safety factor.
|
||||
stack_height = (
|
||||
sum(line['height'] for line in visible) + gap * (len(visible) - 1)
|
||||
)
|
||||
edge = box.center_y + stack_height / 2
|
||||
|
||||
# Body lines hang off the emphasis line's edges, alternating sides in
|
||||
# reading order — the first body line to the left, the next to the right.
|
||||
anchor = max(line['width'] for line in visible)
|
||||
side = -1
|
||||
for index, line in enumerate(visible):
|
||||
if index:
|
||||
edge -= gap
|
||||
cursor_y = edge - line['ink_top']
|
||||
edge = cursor_y + line['ink_bottom']
|
||||
if line['emphasis']:
|
||||
x = 0.0
|
||||
else:
|
||||
x = side * (anchor - line['width']) / 2 * stagger_ratio
|
||||
side = -side
|
||||
look = line['look']
|
||||
result.blocks.append(PlacedBlock(
|
||||
text=line['text'],
|
||||
words=[w for w, _ in line['run']],
|
||||
font=getattr(look, 'font', 'Helvetica Neue'),
|
||||
face=getattr(look, 'face', None),
|
||||
font_size=line['font_size'],
|
||||
color=getattr(look, 'color', '1 1 1 1'),
|
||||
kerning=line['kerning'],
|
||||
x=x,
|
||||
y=cursor_y,
|
||||
width=line['width'],
|
||||
height=line['height'],
|
||||
ink_top=line['ink_top'],
|
||||
ink_bottom=line['ink_bottom'],
|
||||
line_index=index,
|
||||
emphasis=line['emphasis'],
|
||||
))
|
||||
return result
|
||||
Executable
+288
@@ -0,0 +1,288 @@
|
||||
"""Transcript intelligence — local Whisper transcription + text-driven editing.
|
||||
|
||||
v0.13 slice 1: transcript-based editing. Transcription runs locally via
|
||||
faster-whisper (optional ``[transcribe]`` extra) with word-level timestamps;
|
||||
without it, or when media is missing/unreadable, ``transcribe`` returns
|
||||
``None`` so callers degrade to an install hint instead of crashing — the
|
||||
same contract as ``media_intel.detect_beats``.
|
||||
|
||||
The matching helpers below are pure functions over word lists so they are
|
||||
fully testable without any model installed, and so tools can accept
|
||||
pre-computed transcripts (from a previous ``transcribe_media`` run) instead
|
||||
of re-transcribing.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import List, Optional, Sequence, Tuple
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Model names are used to resolve (and download) model weights, so they are
|
||||
# validated against an allowlist, not trusted.
|
||||
ALLOWED_MODELS = (
|
||||
"tiny", "tiny.en", "base", "base.en", "small", "small.en",
|
||||
"medium", "medium.en", "large-v2", "large-v3", "distil-large-v3",
|
||||
)
|
||||
|
||||
# Conservative by default: interjections that are near-universally filler.
|
||||
# "like" / "so" / "actually" are speech, not noise, unless the user opts in.
|
||||
DEFAULT_FILLERS = ("um", "uh", "uhh", "umm", "erm", "ehm", "mmm", "hmm", "mhm")
|
||||
|
||||
_NORM_RE = re.compile(r"[^\w']+")
|
||||
|
||||
|
||||
def normalize_word(word: str) -> str:
|
||||
"""Lowercase a word and strip punctuation so matching survives Whisper's
|
||||
tokenization quirks (leading spaces, trailing commas, case)."""
|
||||
return _NORM_RE.sub("", word.lower())
|
||||
|
||||
|
||||
def find_phrase_spans(
|
||||
words: Sequence[dict], phrase: str
|
||||
) -> List[Tuple[float, float]]:
|
||||
"""Find every occurrence of ``phrase`` in a word-level transcript.
|
||||
|
||||
``words`` is a sequence of ``{"word", "start", "end"}`` dicts in source
|
||||
seconds. Matching is case- and punctuation-insensitive. Returns
|
||||
``(start, end)`` source-second ranges spanning first to last matched word.
|
||||
"""
|
||||
target = [normalize_word(w) for w in phrase.split()]
|
||||
target = [t for t in target if t]
|
||||
if not target:
|
||||
return []
|
||||
normed = [normalize_word(w.get("word", "")) for w in words]
|
||||
spans: List[Tuple[float, float]] = []
|
||||
i = 0
|
||||
n, m = len(normed), len(target)
|
||||
while i <= n - m:
|
||||
if normed[i:i + m] == target:
|
||||
spans.append((float(words[i]["start"]), float(words[i + m - 1]["end"])))
|
||||
i += m
|
||||
else:
|
||||
i += 1
|
||||
return spans
|
||||
|
||||
|
||||
def find_filler_spans(
|
||||
words: Sequence[dict], fillers: Sequence[str] = DEFAULT_FILLERS
|
||||
) -> List[Tuple[float, float]]:
|
||||
"""Find filler-word occurrences (single- or multi-word fillers)."""
|
||||
spans: List[Tuple[float, float]] = []
|
||||
for filler in fillers:
|
||||
spans.extend(find_phrase_spans(words, filler))
|
||||
spans.sort()
|
||||
return spans
|
||||
|
||||
|
||||
def merge_ranges(
|
||||
ranges: Sequence[Tuple[float, float]], min_gap: float = 0.0
|
||||
) -> List[Tuple[float, float]]:
|
||||
"""Merge overlapping (or nearly touching, within ``min_gap``) ranges."""
|
||||
if not ranges:
|
||||
return []
|
||||
ordered = sorted(ranges)
|
||||
merged = [list(ordered[0])]
|
||||
for start, end in ordered[1:]:
|
||||
if start <= merged[-1][1] + min_gap:
|
||||
merged[-1][1] = max(merged[-1][1], end)
|
||||
else:
|
||||
merged.append([start, end])
|
||||
return [(s, e) for s, e in merged]
|
||||
|
||||
|
||||
def invert_ranges(
|
||||
ranges: Sequence[Tuple[float, float]], window_start: float, window_end: float
|
||||
) -> List[Tuple[float, float]]:
|
||||
"""Complement of ``ranges`` within ``[window_start, window_end]`` —
|
||||
turns keep-ranges into cut-ranges for keep_only mode."""
|
||||
if window_end <= window_start:
|
||||
return []
|
||||
kept = merge_ranges(
|
||||
[(max(s, window_start), min(e, window_end)) for s, e in ranges
|
||||
if min(e, window_end) > max(s, window_start)]
|
||||
)
|
||||
if not kept:
|
||||
return [(window_start, window_end)]
|
||||
out: List[Tuple[float, float]] = []
|
||||
cursor = window_start
|
||||
for start, end in kept:
|
||||
if start > cursor:
|
||||
out.append((cursor, start))
|
||||
cursor = max(cursor, end)
|
||||
if cursor < window_end:
|
||||
out.append((cursor, window_end))
|
||||
return out
|
||||
|
||||
|
||||
def transcribe(
|
||||
path: str, model_size: str = "base", language: Optional[str] = None
|
||||
) -> Optional[dict]:
|
||||
"""Transcribe an audio/video file locally with word-level timestamps.
|
||||
|
||||
Requires the optional ``[transcribe]`` extra (faster-whisper). Returns
|
||||
``None`` when the model is unavailable or the file is missing/unreadable.
|
||||
|
||||
The model weights are resolved from the configured models directory (see
|
||||
``model_manager.get_models_dir``), so a model selected/downloaded through
|
||||
the app is found without an implicit download to the default HF cache.
|
||||
|
||||
Returns:
|
||||
``{"language": str, "duration": float, "text": str,
|
||||
"segments": [{"text", "start", "end", "start_fmt", "end_fmt"}, ...],
|
||||
"words": [{"word", "start", "end", "confidence"}, ...]}``
|
||||
"""
|
||||
if model_size not in ALLOWED_MODELS:
|
||||
raise ValueError(
|
||||
f"model_size must be one of {', '.join(ALLOWED_MODELS)}, got {model_size!r}"
|
||||
)
|
||||
file_path = Path(path)
|
||||
if not file_path.is_file():
|
||||
return None
|
||||
try:
|
||||
# Resolve the configured models root and point HF at it *before* the
|
||||
# first faster_whisper import, so downloads/loads go to our folder.
|
||||
from .model_manager import get_models_dir
|
||||
|
||||
models_dir = get_models_dir()
|
||||
hf_home = models_dir / "hf_home"
|
||||
hf_home.mkdir(parents=True, exist_ok=True)
|
||||
os.environ["HF_HOME"] = str(hf_home)
|
||||
os.environ["HUGGINGFACE_HUB_CACHE"] = str(hf_home / "hub")
|
||||
|
||||
from faster_whisper import WhisperModel
|
||||
except ImportError:
|
||||
logger.info("faster-whisper not installed; transcription unavailable")
|
||||
return None
|
||||
try:
|
||||
model = WhisperModel(
|
||||
model_size,
|
||||
compute_type="int8",
|
||||
download_root=str(models_dir),
|
||||
)
|
||||
segments_iter, info = model.transcribe(
|
||||
str(file_path),
|
||||
language=language,
|
||||
word_timestamps=True,
|
||||
vad_filter=True,
|
||||
)
|
||||
segments: List[dict] = []
|
||||
words: List[dict] = []
|
||||
for seg in segments_iter:
|
||||
start = float(seg.start)
|
||||
end = float(seg.end)
|
||||
segments.append(
|
||||
{
|
||||
"text": seg.text.strip(),
|
||||
"start": start,
|
||||
"end": end,
|
||||
"start_fmt": format_timestamp(start),
|
||||
"end_fmt": format_timestamp(end),
|
||||
}
|
||||
)
|
||||
for w in seg.words or []:
|
||||
ws = float(w.start)
|
||||
we = float(w.end)
|
||||
words.append(
|
||||
{
|
||||
"word": w.word.strip(),
|
||||
"start": ws,
|
||||
"end": we,
|
||||
"confidence": float(w.probability),
|
||||
}
|
||||
)
|
||||
except Exception:
|
||||
logger.warning("whisper transcription failed for %s", file_path)
|
||||
return None
|
||||
return {
|
||||
"language": info.language,
|
||||
"duration": float(info.duration),
|
||||
"text": " ".join(s["text"] for s in segments),
|
||||
"segments": segments,
|
||||
"words": words,
|
||||
}
|
||||
|
||||
|
||||
def format_timestamp(seconds: float) -> str:
|
||||
"""Format float seconds as ``HH:MM:SS.mmm`` (e.g. ``00:01:23.450``)."""
|
||||
total_ms = max(0, round((seconds or 0.0) * 1000))
|
||||
hours, rem_ms = divmod(total_ms, 3_600_000)
|
||||
minutes, rem_ms = divmod(rem_ms, 60_000)
|
||||
secs, ms = divmod(rem_ms, 1000)
|
||||
return f"{hours:02d}:{minutes:02d}:{secs:02d}.{ms:03d}"
|
||||
|
||||
|
||||
def group_words_by_segment(
|
||||
words: Sequence[dict],
|
||||
segments: Sequence[dict],
|
||||
) -> List[List[dict]]:
|
||||
"""Group flat *words* into sentences using *segments*' time windows.
|
||||
|
||||
``transcribe()`` returns ``segments`` and ``words`` as sibling flat lists —
|
||||
the words are flattened out of the segments and the association is lost in
|
||||
serialisation, leaving only the time ranges to rejoin them by. A word
|
||||
belongs to the segment whose ``[start, end)`` contains its ``start``.
|
||||
|
||||
Words falling in no segment (rounding at a boundary, or a gap between
|
||||
segments) attach to the group being built rather than being dropped —
|
||||
losing a spoken word would silently drop it from the subtitles.
|
||||
|
||||
With no usable *segments*, every word comes back as a single group, which
|
||||
the caller can still split on its own terms.
|
||||
"""
|
||||
kept = [
|
||||
s for s in (segments or [])
|
||||
if s.get('start') is not None and s.get('end') is not None
|
||||
]
|
||||
ordered = sorted(kept, key=lambda s: float(s['start']))
|
||||
if not ordered:
|
||||
return [list(words)] if words else []
|
||||
|
||||
groups: List[List[dict]] = []
|
||||
current: List[dict] = []
|
||||
current_index: Optional[int] = None
|
||||
cursor = 0
|
||||
|
||||
for w in words:
|
||||
start = float(w.get('start', 0.0))
|
||||
# Segments and words are both chronological, so the search only ever
|
||||
# moves forward.
|
||||
while (
|
||||
cursor + 1 < len(ordered)
|
||||
and start >= float(ordered[cursor]['end'])
|
||||
and start >= float(ordered[cursor + 1]['start'])
|
||||
):
|
||||
cursor += 1
|
||||
|
||||
seg = ordered[cursor]
|
||||
inside = float(seg['start']) <= start < float(seg['end'])
|
||||
index = cursor if inside else current_index
|
||||
|
||||
if current and index != current_index and inside:
|
||||
groups.append(current)
|
||||
current = []
|
||||
if inside or current_index is None:
|
||||
current_index = index if index is not None else cursor
|
||||
current.append(w)
|
||||
|
||||
if current:
|
||||
groups.append(current)
|
||||
return groups
|
||||
|
||||
|
||||
def segments_to_srt(segments: Sequence[dict]) -> str:
|
||||
"""Render transcript segments as an SRT string (for captions import)."""
|
||||
|
||||
def stamp(seconds: float) -> str:
|
||||
ms = int(round(seconds * 1000))
|
||||
h, rem = divmod(ms, 3600000)
|
||||
m, rem = divmod(rem, 60000)
|
||||
s, ms = divmod(rem, 1000)
|
||||
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
|
||||
|
||||
blocks = []
|
||||
for i, seg in enumerate(segments, 1):
|
||||
blocks.append(f"{i}\n{stamp(seg['start'])} --> {stamp(seg['end'])}\n{seg['text']}\n")
|
||||
return "\n".join(blocks)
|
||||
Executable
+3822
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user