Files
gart/code/fcpxml/voice_actions.py

249 lines
8.7 KiB
Python

"""Voice actions — the editing decisions produced from a voice timeline.
This is the contract between *deciding* and *applying*. Whoever makes the
editorial call — the deterministic rules engine, or a model reading the
voice timeline JSON — emits the same list of actions, and one applier turns
it into FCPXML. Nothing that produces actions ever touches XML.
Every action's ``start``/``end`` is in **original source seconds**, matching
the voice timeline. That matters: cuts shift everything after them, so if
decisions were expressed in post-cut time they would silently land in the
wrong place the moment a cut was added. Keeping one origin and resolving the
shift at apply time (:func:`shift_after_cuts`) removes that whole class of bug.
Actions arriving from a model are untrusted input: :func:`parse_actions`
validates and reports what it rejected rather than raising, so one malformed
row never discards a whole edit.
"""
from dataclasses import dataclass, field
from typing import Any, List, Optional, Sequence, Tuple
# What an action can ask for. Deliberately small — each maps onto one
# existing writer capability, so no new XML knowledge lives here.
ACTION_KINDS = ("cut", "zoom", "text", "marker")
# Bounds for a zoom's scale factor. Below 1.0 is a pull-back, not a punch-in;
# above 3x the image falls apart on any normal footage.
MIN_ZOOM_SCALE = 1.0
MAX_ZOOM_SCALE = 3.0
MAX_TEXT_LENGTH = 120
@dataclass
class VoiceAction:
"""One editing decision, in original source time."""
kind: str
start: float
end: float
params: dict = field(default_factory=dict)
reason: str = ""
speaker: str = ""
@property
def duration(self) -> float:
return max(0.0, self.end - self.start)
def as_dict(self) -> dict:
return {
"kind": self.kind,
"start": round(self.start, 3),
"end": round(self.end, 3),
"params": self.params,
"reason": self.reason,
"speaker": self.speaker,
}
def _validate_one(raw: Any, index: int) -> Tuple[Optional[VoiceAction], str]:
"""Turn one raw row into a VoiceAction, or explain why it can't be."""
where = f"action[{index}]"
if not isinstance(raw, dict):
return None, f"{where}: expected an object, got {type(raw).__name__}"
kind = str(raw.get("kind", "")).strip().lower()
if kind not in ACTION_KINDS:
return None, f"{where}: unknown kind {raw.get('kind')!r} (expected one of {', '.join(ACTION_KINDS)})"
try:
start = float(raw.get("start"))
end = float(raw.get("end"))
except (TypeError, ValueError):
return None, f"{where}: start/end must be numbers (seconds)"
if start < 0:
return None, f"{where}: start is negative ({start})"
if end <= start:
return None, f"{where}: end ({end}) must be after start ({start})"
params = raw.get("params")
params = dict(params) if isinstance(params, dict) else {}
if kind == "zoom":
try:
scale = float(params.get("scale", 1.3))
except (TypeError, ValueError):
return None, f"{where}: zoom scale must be a number"
if not (MIN_ZOOM_SCALE <= scale <= MAX_ZOOM_SCALE):
return None, (
f"{where}: zoom scale {scale} outside {MIN_ZOOM_SCALE}-{MAX_ZOOM_SCALE}"
)
params["scale"] = scale
if kind == "text":
content = str(params.get("content", "")).strip()
if not content:
return None, f"{where}: text action needs params.content"
params["content"] = content[:MAX_TEXT_LENGTH]
return (
VoiceAction(
kind=kind,
start=start,
end=end,
params=params,
reason=str(raw.get("reason", "")),
speaker=str(raw.get("speaker", "")),
),
"",
)
def parse_actions(data: Any) -> Tuple[List[VoiceAction], List[str]]:
"""Validate a decision list into actions, collecting rejections.
Accepts either a bare list of actions or ``{"actions": [...]}`` — the
shape a model is most likely to return. Returns ``(actions, errors)``;
a row that fails validation is reported and skipped, never fatal.
"""
if isinstance(data, dict):
data = data.get("actions", [])
if not isinstance(data, Sequence) or isinstance(data, (str, bytes)):
return [], ["expected a list of actions, or an object with an 'actions' list"]
actions: List[VoiceAction] = []
errors: List[str] = []
for i, raw in enumerate(data):
action, error = _validate_one(raw, i)
if action is not None:
actions.append(action)
else:
errors.append(error)
return actions, errors
def speaker_cut_actions(
timeline: dict,
speaker_ids: Sequence[str],
padding: float = 0.15,
) -> List[VoiceAction]:
"""Cut actions removing everything the given speakers say.
The everyday case on a testimonial shoot: an interviewer or a crew
member talks over the take, and only the subject should survive the
edit. ``padding`` trims slightly *inside* each segment rather than
around it — speech boundaries from a transcript are approximate, and
eating into the neighbouring silence is far safer than clipping the
first syllable of the person being kept.
"""
wanted = {str(s) for s in speaker_ids}
actions: List[VoiceAction] = []
for segment in timeline.get("segments", []):
if str(segment.get("speaker", "")) not in wanted:
continue
start = float(segment.get("start", 0.0)) + padding
end = float(segment.get("end", 0.0)) - padding
if end <= start:
continue
actions.append(
VoiceAction(
kind="cut",
start=start,
end=end,
reason=f"fala de {segment.get('speaker')}",
speaker=str(segment.get("speaker", "")),
)
)
return actions
def merge_cut_ranges(actions: Sequence[VoiceAction]) -> List[Tuple[float, float]]:
"""The cut actions as merged, sorted, non-overlapping source ranges."""
cuts = sorted((a.start, a.end) for a in actions if a.kind == "cut")
merged: List[Tuple[float, float]] = []
for start, end in cuts:
if merged and start <= merged[-1][1]:
merged[-1] = (merged[-1][0], max(merged[-1][1], end))
else:
merged.append((start, end))
return merged
def shift_after_cuts(
time: float, cuts: Sequence[Tuple[float, float]]
) -> Optional[float]:
"""Where source ``time`` lands once ``cuts`` are removed.
Returns ``None`` when the time falls *inside* a cut — the material it
referred to no longer exists, so the action that pointed at it must be
dropped rather than silently slid onto neighbouring content.
``cuts`` must be merged and sorted (see :func:`merge_cut_ranges`).
"""
shift = 0.0
for start, end in cuts:
if time < start:
break
if time < end:
return None
shift += end - start
return time - shift
def resolve_actions(
actions: Sequence[VoiceAction],
) -> Tuple[List[Tuple[float, float]], List[VoiceAction], List[VoiceAction]]:
"""Split a decision list into what to cut and what to place afterwards.
Returns ``(cut_ranges, placed, dropped)``. Non-cut actions are moved onto
their post-cut times; any that pointed into removed material land in
``dropped`` so the caller can report them instead of losing them quietly.
"""
cut_ranges = merge_cut_ranges(actions)
placed: List[VoiceAction] = []
dropped: List[VoiceAction] = []
for action in actions:
if action.kind == "cut":
continue
new_start = shift_after_cuts(action.start, cut_ranges)
if new_start is None:
dropped.append(action)
continue
if action.kind == "marker":
# A marker is a point, not a span: it survives as long as its own
# instant does. Requiring its nominal end to survive too would
# drop exactly the markers worth keeping — the ones flagging a
# join, which sit right against a cut edge by definition.
new_end = new_start + action.duration
else:
new_end = shift_after_cuts(action.end, cut_ranges)
if new_end is None:
dropped.append(action)
continue
if new_end <= new_start:
dropped.append(action)
continue
placed.append(
VoiceAction(
kind=action.kind,
start=new_start,
end=new_end,
params=action.params,
reason=action.reason,
speaker=action.speaker,
)
)
return cut_ranges, placed, dropped